mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1000 @@
|
|
|
1
|
+
"""Look at a file, or a public https address, before writing a Recipe against it.
|
|
2
|
+
|
|
3
|
+
`peek` is the first thing a person tries and the first thing a driving agent runs. It answers one
|
|
4
|
+
question -- what is in there -- and it answers it without building anything, approving anything, or
|
|
5
|
+
writing a single byte to disk.
|
|
6
|
+
|
|
7
|
+
Ordinary tabular bytes go through :func:`parse_tabular_bytes_for_display`. Reader-pinned bytes go
|
|
8
|
+
through the exact Toolbox coordinate's warm-up and decode inside the same clean room as a Build;
|
|
9
|
+
for a URL, retrieval and decode are one operation so container bytes never return to this process.
|
|
10
|
+
That is deliberate and load-bearing: the size budgets, archive and executable refusal, strict
|
|
11
|
+
UTF-8 rule, and row and cell limits come along for free, with no second unconfined parser. There is
|
|
12
|
+
no ``open()``, no ``csv`` import, and no ``json`` import in this module, and
|
|
13
|
+
`tests/test_ux_peek.py` reads this file's own source to keep it that way. Local bytes are fetched
|
|
14
|
+
through :mod:`ux.plain_file`, the one open-then-fstat rule every command that takes a path from a
|
|
15
|
+
person shares, so peek has no reading rule of its own either.
|
|
16
|
+
|
|
17
|
+
The one difference is the display rendering. A Build's Reader refuses a value the canonical rule
|
|
18
|
+
cannot hold exactly -- a date, a time, a duration, a decimal, a not-a-number -- because what a
|
|
19
|
+
Build reads gets sealed. A peek seals nothing and writes nothing, so it renders those values into
|
|
20
|
+
their exact text instead of refusing them, and reports the column as the kind of column it really
|
|
21
|
+
is. That is why peek opens a Build's own ``table.parquet``, and every timestamped source the
|
|
22
|
+
product exists to work with, rather than turning them away.
|
|
23
|
+
|
|
24
|
+
What peek reports is deliberately small: the columns, the type observed in each, how many values
|
|
25
|
+
are missing, how many are distinct, the lowest and highest, and the first few rows. Nothing here
|
|
26
|
+
infers a type, coerces a value, or refuses a messy cell. Peek exists to look at data nobody has
|
|
27
|
+
cleaned yet, so a peek that refuses messy data would not be a peek.
|
|
28
|
+
|
|
29
|
+
Column types are reported in the product's own vocabulary -- ``string``, ``int64``, ``float64``,
|
|
30
|
+
``boolean``, ``date``, ``timestamp_utc`` -- which is the set a Recipe accepts and the set
|
|
31
|
+
``mr-data show`` and ``mr-data inventory`` print. That is not an inference: it is the observed kind
|
|
32
|
+
under the name the rest of the product gives it, so a type read out of a peek is a type ``author``
|
|
33
|
+
takes. A column that holds no single kind, or a kind none of the six covers, is never dressed up as
|
|
34
|
+
one of them; see :func:`plain_column_type`.
|
|
35
|
+
|
|
36
|
+
``timestamp_utc`` is the narrowest of the six and is not a synonym for "a timestamp": it is bound
|
|
37
|
+
to ``timestamp[us, tz=UTC]``, and a Recipe refuses the type unless the timezone it declares is UTC.
|
|
38
|
+
So a column whose timestamps carry no zone, or carry one that is not UTC, is reported as the kind
|
|
39
|
+
it is rather than as that one -- the sample rows printed two lines below say the same thing, and
|
|
40
|
+
the two may not disagree.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import math
|
|
46
|
+
import os
|
|
47
|
+
import tempfile
|
|
48
|
+
from collections.abc import Mapping
|
|
49
|
+
from dataclasses import dataclass, replace
|
|
50
|
+
from datetime import datetime, timedelta
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
from typing import Any
|
|
53
|
+
from urllib.parse import urlsplit
|
|
54
|
+
|
|
55
|
+
from mostlyright.data_harness.acquisition.http import (
|
|
56
|
+
PinnedHttpsRetriever,
|
|
57
|
+
PinnedTransport,
|
|
58
|
+
RetrievalLimits,
|
|
59
|
+
StdlibPinnedTransport,
|
|
60
|
+
reader_retrieval_limits,
|
|
61
|
+
)
|
|
62
|
+
from mostlyright.data_harness.acquisition.parsing import (
|
|
63
|
+
_FORMAT_MEDIA_TYPES,
|
|
64
|
+
_FORMAT_SUFFIXES,
|
|
65
|
+
ParsedTable,
|
|
66
|
+
ParseLimits,
|
|
67
|
+
column_types,
|
|
68
|
+
parse_tabular_bytes_for_display,
|
|
69
|
+
)
|
|
70
|
+
from mostlyright.data_harness.acquisition.sandbox import CrawlerSandbox, SandboxResult
|
|
71
|
+
from mostlyright.data_harness.acquisition.url_policy import (
|
|
72
|
+
AcquisitionSecurityError,
|
|
73
|
+
EgressPolicy,
|
|
74
|
+
Resolver,
|
|
75
|
+
SystemResolver,
|
|
76
|
+
_normalize_hostname,
|
|
77
|
+
)
|
|
78
|
+
from mostlyright.data_harness.canonical import sha256_bytes
|
|
79
|
+
from mostlyright.data_harness.readers.contracts import ReaderError, ReaderPin, identifier, semver
|
|
80
|
+
from mostlyright.data_harness.readers.registry import TOOLBOX
|
|
81
|
+
from mostlyright.data_harness.readers.samples import DecodedFacts, ReaderSample, warm_up
|
|
82
|
+
from mostlyright.data_harness.ux.path_kind import UNKNOWN_KIND, name_of_kind, presence_at
|
|
83
|
+
from mostlyright.data_harness.ux.plain_file import (
|
|
84
|
+
DANGLING,
|
|
85
|
+
MISSING,
|
|
86
|
+
PlainFileRefusal,
|
|
87
|
+
open_plain_file,
|
|
88
|
+
read_bounded,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
# How many rows come back by default, and the ceiling on asking for more. A peek is a look, not an
|
|
92
|
+
# export: anything larger is a Build.
|
|
93
|
+
DEFAULT_SAMPLE_ROWS = 5
|
|
94
|
+
MAX_SAMPLE_ROWS = 100
|
|
95
|
+
|
|
96
|
+
# Derived from the Reader's own allowlist rather than re-typed here, so the list of formats peek
|
|
97
|
+
# offers cannot drift from the list of formats the harness can actually open.
|
|
98
|
+
PEEK_FORMATS: tuple[str, ...] = tuple(sorted(_FORMAT_SUFFIXES))
|
|
99
|
+
_SUFFIX_FORMATS: dict[str, str] = {
|
|
100
|
+
suffix: data_format
|
|
101
|
+
for data_format, suffixes in _FORMAT_SUFFIXES.items()
|
|
102
|
+
for suffix in sorted(suffixes)
|
|
103
|
+
}
|
|
104
|
+
_MEDIA_FORMATS: dict[str, str] = {
|
|
105
|
+
media_type: data_format
|
|
106
|
+
for data_format, media_types in _FORMAT_MEDIA_TYPES.items()
|
|
107
|
+
for media_type in sorted(media_types)
|
|
108
|
+
}
|
|
109
|
+
_READABLE = ", ".join(PEEK_FORMATS)
|
|
110
|
+
_JSON_READER_COORDINATE = ("json.tabular", "1.0.0")
|
|
111
|
+
_INHERITED_MEMORY_CGROUP_ROOT_FD = 3
|
|
112
|
+
_EXTERNAL_NETWORK_POLICY_ATTESTATION_ENV = "MOSTLYRIGHT_EXTERNAL_NETWORK_POLICY_ATTESTATION"
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _reader_sandbox(staging_root: Path) -> CrawlerSandbox:
|
|
116
|
+
memory_fd = None
|
|
117
|
+
if os.sys.platform == "linux":
|
|
118
|
+
try:
|
|
119
|
+
os.fstat(_INHERITED_MEMORY_CGROUP_ROOT_FD)
|
|
120
|
+
inherited = os.get_inheritable(_INHERITED_MEMORY_CGROUP_ROOT_FD)
|
|
121
|
+
except OSError:
|
|
122
|
+
pass
|
|
123
|
+
else:
|
|
124
|
+
if inherited:
|
|
125
|
+
memory_fd = _INHERITED_MEMORY_CGROUP_ROOT_FD
|
|
126
|
+
return CrawlerSandbox(
|
|
127
|
+
staging_root=staging_root,
|
|
128
|
+
external_network_policy_attestation=os.environ.get(
|
|
129
|
+
_EXTERNAL_NETWORK_POLICY_ATTESTATION_ENV
|
|
130
|
+
),
|
|
131
|
+
memory_cgroup_root_fd=memory_fd,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _reader_pin(
|
|
136
|
+
coordinate: tuple[str, str] | None,
|
|
137
|
+
options: Mapping[str, Any],
|
|
138
|
+
) -> tuple[Any, dict[str, Any], str]:
|
|
139
|
+
family = TOOLBOX.resolve(*(coordinate or _JSON_READER_COORDINATE))
|
|
140
|
+
admitted = dict(family.validate_options(options))
|
|
141
|
+
resolved = ReaderPin(family.family_id, family.family_version, admitted)
|
|
142
|
+
return (
|
|
143
|
+
family,
|
|
144
|
+
{
|
|
145
|
+
"family_id": family.family_id,
|
|
146
|
+
"family_version": family.family_version,
|
|
147
|
+
"decode_options": admitted,
|
|
148
|
+
},
|
|
149
|
+
resolved.options_digest,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _normalise_reader_request(
|
|
154
|
+
coordinate: tuple[str, str] | None,
|
|
155
|
+
options: Mapping[str, Any] | None,
|
|
156
|
+
data_format: str | None,
|
|
157
|
+
) -> tuple[tuple[str, str] | None, Mapping[str, Any] | None]:
|
|
158
|
+
"""Choose Reader mode once; an exact coordinate can never fall through to direct parsing."""
|
|
159
|
+
|
|
160
|
+
reader_requested = coordinate is not None or options is not None
|
|
161
|
+
if not reader_requested:
|
|
162
|
+
return None, None
|
|
163
|
+
if data_format is not None:
|
|
164
|
+
raise ReaderError(
|
|
165
|
+
"READER_OPTIONS",
|
|
166
|
+
"peek.format",
|
|
167
|
+
"data_format is for direct tabular preview and cannot be combined with a Reader",
|
|
168
|
+
)
|
|
169
|
+
if options is not None and not isinstance(options, Mapping):
|
|
170
|
+
raise ReaderError("READER_OPTIONS", "peek.reader_options", "must be an object")
|
|
171
|
+
if coordinate is None:
|
|
172
|
+
resolved_coordinate = _JSON_READER_COORDINATE
|
|
173
|
+
else:
|
|
174
|
+
if not isinstance(coordinate, tuple) or len(coordinate) != 2:
|
|
175
|
+
raise ReaderError(
|
|
176
|
+
"READER_CONTRACT",
|
|
177
|
+
"peek.reader",
|
|
178
|
+
"must be an exact (family, version) Reader coordinate",
|
|
179
|
+
)
|
|
180
|
+
resolved_coordinate = (
|
|
181
|
+
identifier(coordinate[0], "peek.reader.family"),
|
|
182
|
+
semver(coordinate[1], "peek.reader.version"),
|
|
183
|
+
)
|
|
184
|
+
return resolved_coordinate, options or {}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _require_reader_result(
|
|
188
|
+
result: SandboxResult,
|
|
189
|
+
*,
|
|
190
|
+
operation: str,
|
|
191
|
+
family: Any,
|
|
192
|
+
pin: Mapping[str, Any],
|
|
193
|
+
options_digest: str,
|
|
194
|
+
retrieval: bool,
|
|
195
|
+
) -> None:
|
|
196
|
+
"""Bind a sandbox answer to the exact decode the coordinator requested."""
|
|
197
|
+
|
|
198
|
+
expected_identity = (
|
|
199
|
+
family.family_id,
|
|
200
|
+
family.family_version,
|
|
201
|
+
options_digest,
|
|
202
|
+
)
|
|
203
|
+
returned_identity = (
|
|
204
|
+
result.decode_family_id,
|
|
205
|
+
result.decode_family_version,
|
|
206
|
+
result.decode_options_digest,
|
|
207
|
+
)
|
|
208
|
+
if (
|
|
209
|
+
result.operation != operation
|
|
210
|
+
or returned_identity != expected_identity
|
|
211
|
+
or result.decode_flags is None
|
|
212
|
+
or result.content is None
|
|
213
|
+
or result.parsed is None
|
|
214
|
+
or pin.get("family_id") != family.family_id
|
|
215
|
+
or pin.get("family_version") != family.family_version
|
|
216
|
+
):
|
|
217
|
+
raise AcquisitionSecurityError(
|
|
218
|
+
"SANDBOX_RESULT",
|
|
219
|
+
"Reader preview result does not bind the exact requested Reader decode",
|
|
220
|
+
)
|
|
221
|
+
if retrieval and (
|
|
222
|
+
result.fetched_content_sha256 is None
|
|
223
|
+
or not result.final_url
|
|
224
|
+
or result.transport_evidence_digest is None
|
|
225
|
+
):
|
|
226
|
+
raise AcquisitionSecurityError(
|
|
227
|
+
"SANDBOX_RESULT",
|
|
228
|
+
"Reader URL preview result omits required retrieval evidence",
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _warm_up_reader(sandbox: CrawlerSandbox, family: Any) -> None:
|
|
233
|
+
"""Prove the exact Reader coordinate on its packaged sample through this clean room."""
|
|
234
|
+
|
|
235
|
+
def decode(sample: ReaderSample) -> DecodedFacts:
|
|
236
|
+
sample_pin = ReaderPin(
|
|
237
|
+
sample.family_id,
|
|
238
|
+
sample.family_version,
|
|
239
|
+
dict(sample.decode_options),
|
|
240
|
+
)
|
|
241
|
+
result = sandbox.decode_and_parse(
|
|
242
|
+
request_id=f"peek.warmup.{sample.family_id}.{sample.family_version}".lower(),
|
|
243
|
+
content=sample.content,
|
|
244
|
+
reader_pin={
|
|
245
|
+
"family_id": sample.family_id,
|
|
246
|
+
"family_version": sample.family_version,
|
|
247
|
+
"decode_options": dict(sample.decode_options),
|
|
248
|
+
},
|
|
249
|
+
output_format=family.output_format,
|
|
250
|
+
limits=ParseLimits(),
|
|
251
|
+
)
|
|
252
|
+
_require_reader_result(
|
|
253
|
+
result,
|
|
254
|
+
operation="decode_and_parse",
|
|
255
|
+
family=family,
|
|
256
|
+
pin={
|
|
257
|
+
"family_id": sample.family_id,
|
|
258
|
+
"family_version": sample.family_version,
|
|
259
|
+
"decode_options": dict(sample.decode_options),
|
|
260
|
+
},
|
|
261
|
+
options_digest=sample_pin.options_digest,
|
|
262
|
+
retrieval=False,
|
|
263
|
+
)
|
|
264
|
+
assert result.content is not None
|
|
265
|
+
assert result.parsed is not None
|
|
266
|
+
assert result.decode_flags is not None
|
|
267
|
+
return DecodedFacts(
|
|
268
|
+
sha256_bytes(result.content),
|
|
269
|
+
len(result.parsed.rows),
|
|
270
|
+
tuple(result.parsed.columns),
|
|
271
|
+
tuple(result.decode_flags),
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
try:
|
|
275
|
+
warm_up(family.family_id, family.family_version, decode=decode)
|
|
276
|
+
except ReaderError as error:
|
|
277
|
+
# A preview must distinguish "this certified Reader is unavailable on this host" from
|
|
278
|
+
# "the Reader or source is invalid". Warm-up intentionally wraps runtime failures for
|
|
279
|
+
# Build execution, but peek is the capability-inspection surface: preserve the exact
|
|
280
|
+
# typed platform refusal there so discovery does not misclassify the source as unfit.
|
|
281
|
+
cause = error.__cause__
|
|
282
|
+
if isinstance(cause, AcquisitionSecurityError) and cause.code in {
|
|
283
|
+
"SANDBOX_MEMORY_BOUNDARY",
|
|
284
|
+
"SANDBOX_OS_BOUNDARY",
|
|
285
|
+
}:
|
|
286
|
+
raise cause from error
|
|
287
|
+
raise
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _peek_reader_bytes(
|
|
291
|
+
content: bytes,
|
|
292
|
+
*,
|
|
293
|
+
family: Any,
|
|
294
|
+
pin: Mapping[str, Any],
|
|
295
|
+
options_digest: str,
|
|
296
|
+
limits: ParseLimits,
|
|
297
|
+
sandbox: CrawlerSandbox,
|
|
298
|
+
) -> ParsedTable:
|
|
299
|
+
_warm_up_reader(sandbox, family)
|
|
300
|
+
result = sandbox.decode_and_parse(
|
|
301
|
+
request_id=f"peek.{family.family_id}.{family.family_version}",
|
|
302
|
+
content=content,
|
|
303
|
+
reader_pin=pin,
|
|
304
|
+
output_format=family.output_format,
|
|
305
|
+
limits=limits,
|
|
306
|
+
)
|
|
307
|
+
_require_reader_result(
|
|
308
|
+
result,
|
|
309
|
+
operation="decode_and_parse",
|
|
310
|
+
family=family,
|
|
311
|
+
pin=pin,
|
|
312
|
+
options_digest=options_digest,
|
|
313
|
+
retrieval=False,
|
|
314
|
+
)
|
|
315
|
+
assert result.parsed is not None
|
|
316
|
+
return result.parsed
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
# ------------------------------------------------------------------------------------------------
|
|
320
|
+
# What a column's type is called
|
|
321
|
+
# ------------------------------------------------------------------------------------------------
|
|
322
|
+
#
|
|
323
|
+
# The Reader reports what it saw in Python's own words -- `str`, `int`, `float`, `int|str`. Those
|
|
324
|
+
# are not words this product uses anywhere else: `mr-data show` prints `string` and `float64`,
|
|
325
|
+
# `mr-data inventory` publishes the same six, and a Recipe accepts only those six. peek is the
|
|
326
|
+
# look-before-you-author step, so an agent that carries a peek answer straight into `author` was
|
|
327
|
+
# being handed a value the Recipe contract rejects, with a refusal that names neither the offending
|
|
328
|
+
# value nor a legal one.
|
|
329
|
+
#
|
|
330
|
+
# So the observed kind is rendered into the product's own column-type vocabulary here. Nothing is
|
|
331
|
+
# inferred or coerced: this is the same fact under the name the rest of the product gives it. A
|
|
332
|
+
# column that holds no single kind, or a kind no Recipe column type covers, is never dressed up as
|
|
333
|
+
# one -- it gets a named marker instead, and the markers are written out below rather than being
|
|
334
|
+
# whatever text happened to come back.
|
|
335
|
+
OBSERVED_LOGICAL_TYPES: dict[str, str] = {
|
|
336
|
+
"str": "string",
|
|
337
|
+
"int": "int64",
|
|
338
|
+
"float": "float64",
|
|
339
|
+
"bool": "boolean",
|
|
340
|
+
"date": "date",
|
|
341
|
+
"datetime": "timestamp_utc",
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
# `timestamp_utc` is not this product's word for "a timestamp". It is bound to
|
|
345
|
+
# `timestamp[us, tz=UTC]` (`preparation/contracts.PHYSICAL_TYPES`), a Recipe refuses the type
|
|
346
|
+
# unless its declared timezone is UTC (`recipe.py`), and the contract reader accepts only UTC
|
|
347
|
+
# instants. A column of naive timestamps, or of timestamps at some other offset, is none of those
|
|
348
|
+
# -- so reporting `timestamp_utc` for every datetime column asserted UTC about values the sample
|
|
349
|
+
# rows two lines above show are not, and handed an agent a type `author` then refuses.
|
|
350
|
+
#
|
|
351
|
+
# The Reader names the kind after the Python type it rendered, and every ``datetime.datetime``
|
|
352
|
+
# renders under one name, so the zone is read back off the rendered text here. These two markers
|
|
353
|
+
# are observed kinds like any other: no Recipe column type covers them, so
|
|
354
|
+
# :func:`plain_column_type` reports them as exactly that, and a mixed column lists them beside the
|
|
355
|
+
# kinds it holds.
|
|
356
|
+
TIMESTAMP_NO_ZONE = "timestamp with no time zone"
|
|
357
|
+
TIMESTAMP_OTHER_ZONE = "timestamp at an offset other than UTC"
|
|
358
|
+
|
|
359
|
+
_OBSERVED_DATETIME = "datetime"
|
|
360
|
+
|
|
361
|
+
# The three answers that are not a column type, spelled out. `null` inside a column that also holds
|
|
362
|
+
# one kind is not one of them: how many values are missing is already its own reported number, so a
|
|
363
|
+
# column of whole numbers with gaps in it is a whole-number column.
|
|
364
|
+
NO_ROWS = "no rows to read"
|
|
365
|
+
ALL_MISSING = "missing in every row"
|
|
366
|
+
MIXED_PREFIX = "mixed: "
|
|
367
|
+
NOT_A_COLUMN_TYPE_PREFIX = "no Recipe column type for this: "
|
|
368
|
+
|
|
369
|
+
# The literal the Reader uses for a column it saw no rows of, and for an absent value.
|
|
370
|
+
_OBSERVED_EMPTY = "empty"
|
|
371
|
+
_OBSERVED_NULL = "null"
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def plain_column_type(observed: str) -> str:
|
|
375
|
+
"""The Reader's observed kind, in the vocabulary the rest of the product uses.
|
|
376
|
+
|
|
377
|
+
``observed`` is what :func:`acquisition.parsing.column_types` returns: one Python type name, or
|
|
378
|
+
several joined by ``|`` for a column holding more than one kind, or ``empty``.
|
|
379
|
+
"""
|
|
380
|
+
|
|
381
|
+
if observed == _OBSERVED_EMPTY:
|
|
382
|
+
return NO_ROWS
|
|
383
|
+
kinds = [kind for kind in observed.split("|") if kind != _OBSERVED_NULL]
|
|
384
|
+
if not kinds:
|
|
385
|
+
return ALL_MISSING
|
|
386
|
+
named = sorted(
|
|
387
|
+
OBSERVED_LOGICAL_TYPES[kind] if kind in OBSERVED_LOGICAL_TYPES else kind for kind in kinds
|
|
388
|
+
)
|
|
389
|
+
if len(named) > 1:
|
|
390
|
+
return MIXED_PREFIX + " and ".join(named)
|
|
391
|
+
if kinds[0] not in OBSERVED_LOGICAL_TYPES:
|
|
392
|
+
return NOT_A_COLUMN_TYPE_PREFIX + named[0]
|
|
393
|
+
return named[0]
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
@dataclass(frozen=True)
|
|
397
|
+
class PeekColumn:
|
|
398
|
+
"""One column as it is, before anyone has cleaned it."""
|
|
399
|
+
|
|
400
|
+
name: str
|
|
401
|
+
type: str
|
|
402
|
+
null_count: int
|
|
403
|
+
distinct_count: int
|
|
404
|
+
minimum: str | None
|
|
405
|
+
maximum: str | None
|
|
406
|
+
|
|
407
|
+
def to_dict(self) -> dict[str, Any]:
|
|
408
|
+
return {
|
|
409
|
+
"name": self.name,
|
|
410
|
+
"type": self.type,
|
|
411
|
+
"null_count": self.null_count,
|
|
412
|
+
"distinct_count": self.distinct_count,
|
|
413
|
+
"minimum": self.minimum,
|
|
414
|
+
"maximum": self.maximum,
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
@dataclass(frozen=True)
|
|
419
|
+
class PeekResult:
|
|
420
|
+
"""What one look found: where it looked, what the columns are, and the first rows."""
|
|
421
|
+
|
|
422
|
+
source: str
|
|
423
|
+
origin: str
|
|
424
|
+
data_format: str
|
|
425
|
+
input_sha256: str
|
|
426
|
+
schema_digest: str
|
|
427
|
+
row_count: int
|
|
428
|
+
columns: tuple[PeekColumn, ...]
|
|
429
|
+
sample: tuple[tuple[Any, ...], ...]
|
|
430
|
+
media_type: str | None = None
|
|
431
|
+
final_url: str | None = None
|
|
432
|
+
content_sha256: str | None = None
|
|
433
|
+
fetched_from: str | None = None
|
|
434
|
+
note: str | None = None
|
|
435
|
+
reader_coordinate: str | None = None
|
|
436
|
+
reader_options_digest: str | None = None
|
|
437
|
+
|
|
438
|
+
def to_dict(self) -> dict[str, Any]:
|
|
439
|
+
"""The one payload both renderings are built from.
|
|
440
|
+
|
|
441
|
+
No key here is taken from the data. A column is called ``column 1`` and carries its real
|
|
442
|
+
name as a value, and a sample row is the row's values in column order. That is deliberate:
|
|
443
|
+
the plain rendering gives a payload key its plain label, so a column named ``max_temp_c``
|
|
444
|
+
used as a key would be shown to a person as ``Max temp c`` -- a name the file does not
|
|
445
|
+
have. Facts about the data go in values, where nothing rewrites them.
|
|
446
|
+
"""
|
|
447
|
+
|
|
448
|
+
payload: dict[str, Any] = {
|
|
449
|
+
"status": "peeked",
|
|
450
|
+
"source": self.source,
|
|
451
|
+
"origin": self.origin,
|
|
452
|
+
"data_format": self.data_format,
|
|
453
|
+
"input_sha256": self.input_sha256,
|
|
454
|
+
"schema_digest": self.schema_digest,
|
|
455
|
+
"row_count": self.row_count,
|
|
456
|
+
"columns": [column.name for column in self.columns],
|
|
457
|
+
"schema": self._schema_payload(),
|
|
458
|
+
"sample": self._sample_payload(),
|
|
459
|
+
}
|
|
460
|
+
if self.origin == "url":
|
|
461
|
+
payload["url"] = self.final_url
|
|
462
|
+
payload["fetched_from"] = self.fetched_from
|
|
463
|
+
payload["content_sha256"] = self.content_sha256
|
|
464
|
+
if self.media_type is not None:
|
|
465
|
+
payload["media_type"] = self.media_type
|
|
466
|
+
payload["saved_to_disk"] = False
|
|
467
|
+
if self.note is not None:
|
|
468
|
+
payload["note"] = self.note
|
|
469
|
+
if self.reader_coordinate is not None:
|
|
470
|
+
payload["reader_coordinate"] = self.reader_coordinate
|
|
471
|
+
payload["reader_options_digest"] = self.reader_options_digest
|
|
472
|
+
payload["reader_execution_status"] = "available"
|
|
473
|
+
return payload
|
|
474
|
+
|
|
475
|
+
def _schema_payload(self) -> dict[str, dict[str, Any]]:
|
|
476
|
+
width = len(str(max(len(self.columns), 1)))
|
|
477
|
+
return {
|
|
478
|
+
f"column {index:0{width}d}": column.to_dict()
|
|
479
|
+
for index, column in enumerate(self.columns, start=1)
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
def _sample_payload(self) -> dict[str, list[Any]]:
|
|
483
|
+
"""Each row as its values in column order, the way a table has always been written."""
|
|
484
|
+
|
|
485
|
+
width = len(str(max(len(self.sample), 1)))
|
|
486
|
+
return {
|
|
487
|
+
f"row {index:0{width}d}": [_json_safe(value) for value in row]
|
|
488
|
+
for index, row in enumerate(self.sample, start=1)
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def peek_bytes(
|
|
493
|
+
content: bytes,
|
|
494
|
+
*,
|
|
495
|
+
filename: str,
|
|
496
|
+
data_format: str | None = None,
|
|
497
|
+
sample_rows: int = DEFAULT_SAMPLE_ROWS,
|
|
498
|
+
limits: ParseLimits | None = None,
|
|
499
|
+
source: str | None = None,
|
|
500
|
+
reader_coordinate: tuple[str, str] | None = None,
|
|
501
|
+
reader_options: Mapping[str, Any] | None = None,
|
|
502
|
+
reader_sandbox: CrawlerSandbox | None = None,
|
|
503
|
+
) -> PeekResult:
|
|
504
|
+
"""Look at exact bytes. The only byte reader in this module, by design."""
|
|
505
|
+
|
|
506
|
+
reader_coordinate, reader_options = _normalise_reader_request(
|
|
507
|
+
reader_coordinate, reader_options, data_format
|
|
508
|
+
)
|
|
509
|
+
rows_wanted = _bounded_sample_rows(sample_rows)
|
|
510
|
+
original_sha256 = sha256_bytes(content)
|
|
511
|
+
resolved_reader_coordinate = None
|
|
512
|
+
reader_options_digest = None
|
|
513
|
+
if reader_options is not None:
|
|
514
|
+
family, pin, reader_options_digest = _reader_pin(reader_coordinate, reader_options)
|
|
515
|
+
resolved_reader_coordinate = f"{family.family_id}@{family.family_version}"
|
|
516
|
+
if reader_sandbox is not None:
|
|
517
|
+
table = _peek_reader_bytes(
|
|
518
|
+
content,
|
|
519
|
+
family=family,
|
|
520
|
+
pin=pin,
|
|
521
|
+
options_digest=reader_options_digest,
|
|
522
|
+
limits=limits or ParseLimits(),
|
|
523
|
+
sandbox=reader_sandbox,
|
|
524
|
+
)
|
|
525
|
+
else:
|
|
526
|
+
with tempfile.TemporaryDirectory(prefix="mr-peek-reader-") as temporary:
|
|
527
|
+
table = _peek_reader_bytes(
|
|
528
|
+
content,
|
|
529
|
+
family=family,
|
|
530
|
+
pin=pin,
|
|
531
|
+
options_digest=reader_options_digest,
|
|
532
|
+
limits=limits or ParseLimits(),
|
|
533
|
+
sandbox=_reader_sandbox(Path(temporary).resolve()),
|
|
534
|
+
)
|
|
535
|
+
else:
|
|
536
|
+
resolved_format, media_type = _format_for(filename, data_format)
|
|
537
|
+
table = parse_tabular_bytes_for_display(
|
|
538
|
+
content,
|
|
539
|
+
data_format=resolved_format,
|
|
540
|
+
media_type=media_type,
|
|
541
|
+
filename=filename,
|
|
542
|
+
limits=limits or ParseLimits(),
|
|
543
|
+
)
|
|
544
|
+
return PeekResult(
|
|
545
|
+
source=source if source is not None else filename,
|
|
546
|
+
origin="file",
|
|
547
|
+
data_format=table.data_format,
|
|
548
|
+
input_sha256=original_sha256,
|
|
549
|
+
schema_digest=table.schema_digest,
|
|
550
|
+
row_count=len(table.rows),
|
|
551
|
+
columns=_summarise(table),
|
|
552
|
+
sample=table.rows[:rows_wanted],
|
|
553
|
+
reader_coordinate=resolved_reader_coordinate if reader_options is not None else None,
|
|
554
|
+
reader_options_digest=reader_options_digest,
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def peek_path(
|
|
559
|
+
path: Path | str,
|
|
560
|
+
*,
|
|
561
|
+
data_format: str | None = None,
|
|
562
|
+
sample_rows: int = DEFAULT_SAMPLE_ROWS,
|
|
563
|
+
limits: ParseLimits | None = None,
|
|
564
|
+
reader_coordinate: tuple[str, str] | None = None,
|
|
565
|
+
reader_options: Mapping[str, Any] | None = None,
|
|
566
|
+
reader_sandbox: CrawlerSandbox | None = None,
|
|
567
|
+
) -> PeekResult:
|
|
568
|
+
"""Look at one file on this machine.
|
|
569
|
+
|
|
570
|
+
The size is checked before the bytes are loaded, so a file too large to open is refused rather
|
|
571
|
+
than read into memory first -- and it is checked on the open descriptor, through
|
|
572
|
+
:mod:`ux.plain_file`, which is the rule the rest of this tree opens anything with. Asking the
|
|
573
|
+
path instead got two answers wrong at once. ``st_size`` is 0 for a pipe and a character device,
|
|
574
|
+
so the cap governed nothing there, and ``Path.is_file`` is false for both, so a peek at either
|
|
575
|
+
one was refused with the sentence written for a folder -- a statement about what is at the path
|
|
576
|
+
that is simply not true, and advice ("check the spelling") aimed at a caller who typed the path
|
|
577
|
+
they meant. The bound also has to govern the read that follows it, so a file that grows between
|
|
578
|
+
the two is caught by the read rather than trusted from a moment earlier.
|
|
579
|
+
"""
|
|
580
|
+
|
|
581
|
+
reader_coordinate, reader_options = _normalise_reader_request(
|
|
582
|
+
reader_coordinate, reader_options, data_format
|
|
583
|
+
)
|
|
584
|
+
target = Path(path)
|
|
585
|
+
rows_wanted = _bounded_sample_rows(sample_rows)
|
|
586
|
+
budgets = limits or ParseLimits()
|
|
587
|
+
try:
|
|
588
|
+
descriptor, size = open_plain_file(target)
|
|
589
|
+
except PlainFileRefusal as refusal:
|
|
590
|
+
# Two codes rather than two wordings of one, because the two findings contradict each
|
|
591
|
+
# other: `PEEK_TARGET` says nothing is there and tells the caller to check the spelling,
|
|
592
|
+
# which is advice about a mistake somebody pointing at a real pipe did not make.
|
|
593
|
+
# `OUTPUT_PARENT_ABSENT` and `OUTPUT_PARENT_INVALID` are two codes for the same reason.
|
|
594
|
+
# Raised here rather than returned from a helper: the sweep that proves every typed code
|
|
595
|
+
# names its fix reads `raise` statements out of the live source.
|
|
596
|
+
if refusal.reason == DANGLING:
|
|
597
|
+
# A dangling link is present even though opening its target raises
|
|
598
|
+
# `FileNotFoundError`, so it must not be translated as an absent input. The noun comes
|
|
599
|
+
# from the `st_mode` retained from the non-following lookup, keeping the sentence and
|
|
600
|
+
# finding bound to one observation.
|
|
601
|
+
raise AcquisitionSecurityError(
|
|
602
|
+
"PEEK_TARGET_DANGLING",
|
|
603
|
+
f"there is {name_of_kind(refusal.mode) or UNKNOWN_KIND} at {refusal.path}, "
|
|
604
|
+
"and it leads nowhere",
|
|
605
|
+
) from None
|
|
606
|
+
if refusal.reason == MISSING:
|
|
607
|
+
raise AcquisitionSecurityError(
|
|
608
|
+
"PEEK_TARGET",
|
|
609
|
+
f"there is {presence_at(target) or UNKNOWN_KIND} at {target} to look at",
|
|
610
|
+
) from None
|
|
611
|
+
raise AcquisitionSecurityError(
|
|
612
|
+
"PEEK_TARGET_NOT_A_FILE",
|
|
613
|
+
f"{target} is not a file: a peek reads a file, not a folder, a pipe, or a device",
|
|
614
|
+
) from None
|
|
615
|
+
try:
|
|
616
|
+
# Resolve the format before the read as well, so an unreadable file is named as unreadable
|
|
617
|
+
# rather than loaded and then refused. Opening the descriptor has loaded nothing yet.
|
|
618
|
+
if reader_options is None:
|
|
619
|
+
_format_for(target.name, data_format)
|
|
620
|
+
if size > budgets.max_input_bytes:
|
|
621
|
+
raise AcquisitionSecurityError(
|
|
622
|
+
"PEEK_SIZE",
|
|
623
|
+
f"{target} holds {size} bytes, above the {budgets.max_input_bytes} byte "
|
|
624
|
+
"limit on anything this harness opens",
|
|
625
|
+
)
|
|
626
|
+
content = read_bounded(descriptor, max_bytes=budgets.max_input_bytes)
|
|
627
|
+
finally:
|
|
628
|
+
os.close(descriptor)
|
|
629
|
+
if len(content) > budgets.max_input_bytes:
|
|
630
|
+
raise AcquisitionSecurityError(
|
|
631
|
+
"PEEK_SIZE",
|
|
632
|
+
f"{target} holds more than {budgets.max_input_bytes} bytes, above the limit on "
|
|
633
|
+
"anything this harness opens",
|
|
634
|
+
)
|
|
635
|
+
return peek_bytes(
|
|
636
|
+
content,
|
|
637
|
+
filename=target.name,
|
|
638
|
+
data_format=data_format,
|
|
639
|
+
sample_rows=rows_wanted,
|
|
640
|
+
limits=budgets,
|
|
641
|
+
source=str(target),
|
|
642
|
+
reader_coordinate=reader_coordinate,
|
|
643
|
+
reader_options=reader_options,
|
|
644
|
+
reader_sandbox=reader_sandbox,
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def peek_url(
|
|
649
|
+
url: str,
|
|
650
|
+
*,
|
|
651
|
+
allow_redirect_hosts: tuple[str, ...] = (),
|
|
652
|
+
data_format: str | None = None,
|
|
653
|
+
sample_rows: int = DEFAULT_SAMPLE_ROWS,
|
|
654
|
+
limits: ParseLimits | None = None,
|
|
655
|
+
transport: PinnedTransport | None = None,
|
|
656
|
+
resolver: Resolver | None = None,
|
|
657
|
+
reader_coordinate: tuple[str, str] | None = None,
|
|
658
|
+
reader_options: Mapping[str, Any] | None = None,
|
|
659
|
+
reader_sandbox: CrawlerSandbox | None = None,
|
|
660
|
+
) -> PeekResult:
|
|
661
|
+
"""Fetch one public https address and look at what came back, saving nothing.
|
|
662
|
+
|
|
663
|
+
The fetch goes through the harness's one Courier: the same address validation, the same
|
|
664
|
+
pinned-peer connection, the same redirect and byte budgets a Recipe's own fetches use. Nothing
|
|
665
|
+
here widens that; the only thing peek supplies is which single host is allowed.
|
|
666
|
+
|
|
667
|
+
``transport`` and ``resolver`` exist so a test can hand in a recorded exchange instead of
|
|
668
|
+
reaching the network. Neither is reachable from the command line.
|
|
669
|
+
"""
|
|
670
|
+
|
|
671
|
+
reader_coordinate, reader_options = _normalise_reader_request(
|
|
672
|
+
reader_coordinate, reader_options, data_format
|
|
673
|
+
)
|
|
674
|
+
rows_wanted = _bounded_sample_rows(sample_rows)
|
|
675
|
+
hostname = _hostname_of(url)
|
|
676
|
+
# The allowlist of one is derived from an explicit argument -- the address the person or their
|
|
677
|
+
# agent typed -- and never from a config file and never from the content of a source. That is
|
|
678
|
+
# what keeps the fail-closed egress model intact: peek narrows it to a single host rather than
|
|
679
|
+
# widening it, and `--allow-redirect-host` names any further host explicitly. Every entry
|
|
680
|
+
# is written in the egress rule's own spelling, the named host and the redirect hosts alike:
|
|
681
|
+
# an entry spelled any other way is either refused as non-canonical or silently never matched.
|
|
682
|
+
allowed_hostnames = (
|
|
683
|
+
hostname,
|
|
684
|
+
*(_normalize_hostname(host) for host in allow_redirect_hosts),
|
|
685
|
+
)
|
|
686
|
+
if reader_options is not None:
|
|
687
|
+
family, pin, reader_options_digest = _reader_pin(reader_coordinate, reader_options)
|
|
688
|
+
budgets = limits or ParseLimits()
|
|
689
|
+
|
|
690
|
+
def retrieve(sandbox: CrawlerSandbox) -> PeekResult:
|
|
691
|
+
_warm_up_reader(sandbox, family)
|
|
692
|
+
result = sandbox.retrieve_decode_and_parse(
|
|
693
|
+
request_id=f"peek.url.{family.family_id}.{family.family_version}",
|
|
694
|
+
url=url,
|
|
695
|
+
allowed_hostnames=allowed_hostnames,
|
|
696
|
+
reader_pin=pin,
|
|
697
|
+
output_format=family.output_format,
|
|
698
|
+
parse_limits=budgets,
|
|
699
|
+
retrieval_limits=reader_retrieval_limits(
|
|
700
|
+
RetrievalLimits(),
|
|
701
|
+
family_id=family.family_id,
|
|
702
|
+
family_version=family.family_version,
|
|
703
|
+
),
|
|
704
|
+
)
|
|
705
|
+
_require_reader_result(
|
|
706
|
+
result,
|
|
707
|
+
operation="retrieve_decode_and_parse",
|
|
708
|
+
family=family,
|
|
709
|
+
pin=pin,
|
|
710
|
+
options_digest=reader_options_digest,
|
|
711
|
+
retrieval=True,
|
|
712
|
+
)
|
|
713
|
+
assert result.parsed is not None
|
|
714
|
+
assert result.fetched_content_sha256 is not None
|
|
715
|
+
return PeekResult(
|
|
716
|
+
source=url,
|
|
717
|
+
origin="url",
|
|
718
|
+
data_format=result.parsed.data_format,
|
|
719
|
+
input_sha256=result.fetched_content_sha256,
|
|
720
|
+
schema_digest=result.parsed.schema_digest,
|
|
721
|
+
row_count=len(result.parsed.rows),
|
|
722
|
+
columns=_summarise(result.parsed),
|
|
723
|
+
sample=result.parsed.rows[:rows_wanted],
|
|
724
|
+
# The combined clean-room result exposes the normalized CSV media type, not the
|
|
725
|
+
# original response header. Do not fabricate source metadata from decode settings.
|
|
726
|
+
media_type=None,
|
|
727
|
+
final_url=result.final_url,
|
|
728
|
+
content_sha256=result.fetched_content_sha256,
|
|
729
|
+
fetched_from=hostname,
|
|
730
|
+
note=(
|
|
731
|
+
f"The fetched media type was enforced against {family.family_id}@"
|
|
732
|
+
f"{family.family_version} inside the "
|
|
733
|
+
"clean room; the original response header is not returned by that operation."
|
|
734
|
+
),
|
|
735
|
+
reader_coordinate=f"{family.family_id}@{family.family_version}",
|
|
736
|
+
reader_options_digest=reader_options_digest,
|
|
737
|
+
)
|
|
738
|
+
|
|
739
|
+
if reader_sandbox is not None:
|
|
740
|
+
return retrieve(reader_sandbox)
|
|
741
|
+
with tempfile.TemporaryDirectory(prefix="mr-peek-reader-") as temporary:
|
|
742
|
+
return retrieve(_reader_sandbox(Path(temporary).resolve()))
|
|
743
|
+
|
|
744
|
+
policy = EgressPolicy(allowed_hostnames=allowed_hostnames)
|
|
745
|
+
retriever = PinnedHttpsRetriever(
|
|
746
|
+
resolver=resolver if resolver is not None else SystemResolver(),
|
|
747
|
+
egress_policy=policy,
|
|
748
|
+
transport=transport if transport is not None else StdlibPinnedTransport(),
|
|
749
|
+
limits=RetrievalLimits(),
|
|
750
|
+
)
|
|
751
|
+
retrieved = retriever.retrieve(url)
|
|
752
|
+
resolved_format, filename, note = _format_for_response(
|
|
753
|
+
final_url=retrieved.final_url,
|
|
754
|
+
media_type=retrieved.media_type,
|
|
755
|
+
declared=data_format,
|
|
756
|
+
)
|
|
757
|
+
result = peek_bytes(
|
|
758
|
+
retrieved.content,
|
|
759
|
+
filename=filename,
|
|
760
|
+
data_format=resolved_format,
|
|
761
|
+
sample_rows=rows_wanted,
|
|
762
|
+
limits=limits,
|
|
763
|
+
reader_coordinate=reader_coordinate,
|
|
764
|
+
reader_options=reader_options,
|
|
765
|
+
)
|
|
766
|
+
return replace(
|
|
767
|
+
result,
|
|
768
|
+
source=url,
|
|
769
|
+
origin="url",
|
|
770
|
+
media_type=retrieved.media_type,
|
|
771
|
+
final_url=retrieved.final_url,
|
|
772
|
+
content_sha256=retrieved.content_sha256,
|
|
773
|
+
fetched_from=hostname,
|
|
774
|
+
note=note,
|
|
775
|
+
)
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
def _hostname_of(url: str) -> str:
|
|
779
|
+
"""The hostname to build the allowlist from. Validating the address is not this job.
|
|
780
|
+
|
|
781
|
+
``validate_public_https_url`` is the validator, and it runs on every hop inside the Courier.
|
|
782
|
+
Reading the host out here is only how the allowlist gets its single entry, so anything odd
|
|
783
|
+
about the address is still refused there rather than quietly accepted here.
|
|
784
|
+
|
|
785
|
+
The spelling is the egress rule's own: ``_normalize_hostname`` is what ``EgressPolicy`` measures
|
|
786
|
+
an entry against and what the validator normalizes every hop to, so this allowlist of one is
|
|
787
|
+
written in the same spelling it will later be compared in. Lower-casing the host here instead
|
|
788
|
+
was a second rule that agreed with that one on ordinary addresses and disagreed on two:
|
|
789
|
+
an internationalized address (whose entry was refused as non-canonical before a byte was
|
|
790
|
+
fetched) and a trailing-dot name (whose entry was stored unnormalized and never matched), both
|
|
791
|
+
turned away with a message about an allowlist a `peek` invocation does not have.
|
|
792
|
+
"""
|
|
793
|
+
|
|
794
|
+
try:
|
|
795
|
+
hostname = urlsplit(url).hostname
|
|
796
|
+
except ValueError:
|
|
797
|
+
hostname = None
|
|
798
|
+
if not hostname:
|
|
799
|
+
raise AcquisitionSecurityError(
|
|
800
|
+
"PEEK_TARGET",
|
|
801
|
+
f"{url} is not an address with a host in it",
|
|
802
|
+
)
|
|
803
|
+
return _normalize_hostname(hostname)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _format_for_response(
|
|
807
|
+
*,
|
|
808
|
+
final_url: str,
|
|
809
|
+
media_type: str,
|
|
810
|
+
declared: str | None,
|
|
811
|
+
) -> tuple[str, str, str | None]:
|
|
812
|
+
"""Decide what was served, and name a file the Reader's own suffix check can look at.
|
|
813
|
+
|
|
814
|
+
Order: what the caller said, then what the server said it was serving, then the name at the
|
|
815
|
+
end of the address. The media type only picks which Reader opens the bytes; that Reader then
|
|
816
|
+
re-checks the magic bytes, the encoding, and the budgets, so a content type that lies gets a
|
|
817
|
+
refusal rather than a wrong reading.
|
|
818
|
+
"""
|
|
819
|
+
|
|
820
|
+
basename = _basename_of(final_url)
|
|
821
|
+
suffix_format = _SUFFIX_FORMATS.get(_suffix_of(basename))
|
|
822
|
+
data_format = declared or _MEDIA_FORMATS.get(media_type) or suffix_format
|
|
823
|
+
if data_format is None:
|
|
824
|
+
raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(final_url))
|
|
825
|
+
if data_format not in PEEK_FORMATS:
|
|
826
|
+
raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(data_format))
|
|
827
|
+
if suffix_format is not None:
|
|
828
|
+
# The address names a file, so the Reader checks that name against the bytes as usual.
|
|
829
|
+
return data_format, basename, None
|
|
830
|
+
stand_in = f"download{_suffix_for(data_format)}"
|
|
831
|
+
reason = (
|
|
832
|
+
f"the server did not name a file, so this was read as {data_format} "
|
|
833
|
+
f"because that is what the media type ({media_type}) says it is"
|
|
834
|
+
)
|
|
835
|
+
return data_format, stand_in, reason
|
|
836
|
+
|
|
837
|
+
|
|
838
|
+
def _basename_of(url: str) -> str:
|
|
839
|
+
return urlsplit(url).path.rsplit("/", 1)[-1]
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _suffix_for(data_format: str) -> str:
|
|
843
|
+
"""The suffix the Reader accepts for this format, taken from the Reader's own table."""
|
|
844
|
+
|
|
845
|
+
return min(_FORMAT_SUFFIXES[data_format])
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def _summarise(table: ParsedTable) -> tuple[PeekColumn, ...]:
|
|
849
|
+
"""One :class:`PeekColumn` per column, over rows the Reader has already bounded."""
|
|
850
|
+
|
|
851
|
+
types = column_types(table)
|
|
852
|
+
columns: list[PeekColumn] = []
|
|
853
|
+
for index, name in enumerate(table.columns):
|
|
854
|
+
values = [row[index] for row in table.rows]
|
|
855
|
+
present = [value for value in values if value is not None]
|
|
856
|
+
minimum, maximum = _extremes(present)
|
|
857
|
+
columns.append(
|
|
858
|
+
PeekColumn(
|
|
859
|
+
name=name,
|
|
860
|
+
type=plain_column_type(_zoned(types[index], present)),
|
|
861
|
+
null_count=len(values) - len(present),
|
|
862
|
+
distinct_count=_distinct(present),
|
|
863
|
+
minimum=minimum,
|
|
864
|
+
maximum=maximum,
|
|
865
|
+
)
|
|
866
|
+
)
|
|
867
|
+
return tuple(columns)
|
|
868
|
+
|
|
869
|
+
|
|
870
|
+
def _zoned(observed: str, values: list[Any]) -> str:
|
|
871
|
+
"""``observed`` with its ``datetime`` kind replaced by what the values say about the zone.
|
|
872
|
+
|
|
873
|
+
Nothing is inferred and nothing is refused: the rendered text of a ``datetime.datetime`` is its
|
|
874
|
+
own ``isoformat``, and reading it back with ``datetime.fromisoformat`` -- the inverse of the
|
|
875
|
+
call that wrote it -- answers whether the value carried a zone and whether that zone was UTC.
|
|
876
|
+
A column whose datetimes do not all agree on UTC is reported as the kind it is, which is one no
|
|
877
|
+
Recipe column type covers. Text that will not read back is treated the same way: this says what
|
|
878
|
+
the column is, and ``timestamp_utc`` is a claim, not a default.
|
|
879
|
+
"""
|
|
880
|
+
|
|
881
|
+
if _OBSERVED_DATETIME not in observed.split("|"):
|
|
882
|
+
return observed
|
|
883
|
+
for value in values:
|
|
884
|
+
if type(value).__name__ != _OBSERVED_DATETIME:
|
|
885
|
+
continue
|
|
886
|
+
offset = _utc_offset(str(value))
|
|
887
|
+
if offset is None:
|
|
888
|
+
return observed.replace(_OBSERVED_DATETIME, TIMESTAMP_NO_ZONE)
|
|
889
|
+
if offset != timedelta(0):
|
|
890
|
+
return observed.replace(_OBSERVED_DATETIME, TIMESTAMP_OTHER_ZONE)
|
|
891
|
+
return observed
|
|
892
|
+
|
|
893
|
+
|
|
894
|
+
def _utc_offset(text: str) -> timedelta | None:
|
|
895
|
+
"""The offset from UTC the rendered timestamp carries, or ``None`` when it carries none."""
|
|
896
|
+
|
|
897
|
+
try:
|
|
898
|
+
return datetime.fromisoformat(text).utcoffset()
|
|
899
|
+
except ValueError:
|
|
900
|
+
return None
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def _extremes(values: list[Any]) -> tuple[str | None, str | None]:
|
|
904
|
+
"""The lowest and highest value, or nothing at all when they cannot be ordered.
|
|
905
|
+
|
|
906
|
+
A column holding more than one kind of value has no order, and saying so is more honest than
|
|
907
|
+
inventing one. It is never a refusal: unclean data is exactly what peek is for.
|
|
908
|
+
"""
|
|
909
|
+
|
|
910
|
+
if not values:
|
|
911
|
+
return None, None
|
|
912
|
+
try:
|
|
913
|
+
ordered = sorted(values)
|
|
914
|
+
except TypeError:
|
|
915
|
+
return None, None
|
|
916
|
+
return str(ordered[0]), str(ordered[-1])
|
|
917
|
+
|
|
918
|
+
|
|
919
|
+
def _distinct(values: list[Any]) -> int:
|
|
920
|
+
"""How many different values a column holds, counting only the ones that are there."""
|
|
921
|
+
|
|
922
|
+
try:
|
|
923
|
+
return len(set(values))
|
|
924
|
+
except TypeError:
|
|
925
|
+
return len({repr(value) for value in values})
|
|
926
|
+
|
|
927
|
+
|
|
928
|
+
def _bounded_sample_rows(sample_rows: int) -> int:
|
|
929
|
+
if type(sample_rows) is not int or not 0 <= sample_rows <= MAX_SAMPLE_ROWS:
|
|
930
|
+
raise AcquisitionSecurityError(
|
|
931
|
+
"PEEK_SAMPLE",
|
|
932
|
+
f"the number of rows to show must be a whole number from 0 to {MAX_SAMPLE_ROWS}",
|
|
933
|
+
)
|
|
934
|
+
return sample_rows
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
def _format_for(filename: str, declared: str | None) -> tuple[str, str]:
|
|
938
|
+
"""Pick the format and the media type that agree with the Reader's own allowlists."""
|
|
939
|
+
|
|
940
|
+
if declared is not None:
|
|
941
|
+
if declared not in PEEK_FORMATS:
|
|
942
|
+
raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(declared))
|
|
943
|
+
return declared, _media_type_for(declared)
|
|
944
|
+
data_format = _SUFFIX_FORMATS.get(_suffix_of(filename))
|
|
945
|
+
if data_format is None:
|
|
946
|
+
raise AcquisitionSecurityError("PEEK_FORMAT", _unreadable(filename))
|
|
947
|
+
return data_format, _media_type_for(data_format)
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def _media_type_for(data_format: str) -> str:
|
|
951
|
+
"""The media type the Reader expects for this format, taken from the Reader's own table."""
|
|
952
|
+
|
|
953
|
+
return min(_FORMAT_MEDIA_TYPES[data_format])
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
def _suffix_of(filename: str) -> str:
|
|
957
|
+
dot = filename.rfind(".")
|
|
958
|
+
return filename[dot:].lower() if dot > 0 else ""
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
def _unreadable(subject: str) -> str:
|
|
962
|
+
return (
|
|
963
|
+
f"{subject} is not one of the formats this build can read ({_READABLE}); "
|
|
964
|
+
"the set of readable formats grows as Readers are certified"
|
|
965
|
+
)
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def _json_safe(value: Any) -> Any:
|
|
969
|
+
"""Keep a cell as it is when it is already a plain value; otherwise show it as written.
|
|
970
|
+
|
|
971
|
+
A rendered value arrives here as a ``str`` subclass carrying the name of the kind it came from.
|
|
972
|
+
That name is a fact about the column and it is already reported there; on the value itself it
|
|
973
|
+
would only be a type nobody asked about, so the text is handed on as plain text.
|
|
974
|
+
"""
|
|
975
|
+
|
|
976
|
+
if isinstance(value, str):
|
|
977
|
+
return str(value)
|
|
978
|
+
if value is None or isinstance(value, bool | int):
|
|
979
|
+
return value
|
|
980
|
+
if isinstance(value, float):
|
|
981
|
+
return value if math.isfinite(value) else str(value)
|
|
982
|
+
return str(value)
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
__all__ = [
|
|
986
|
+
"ALL_MISSING",
|
|
987
|
+
"DEFAULT_SAMPLE_ROWS",
|
|
988
|
+
"MAX_SAMPLE_ROWS",
|
|
989
|
+
"MIXED_PREFIX",
|
|
990
|
+
"NOT_A_COLUMN_TYPE_PREFIX",
|
|
991
|
+
"NO_ROWS",
|
|
992
|
+
"OBSERVED_LOGICAL_TYPES",
|
|
993
|
+
"PEEK_FORMATS",
|
|
994
|
+
"PeekColumn",
|
|
995
|
+
"PeekResult",
|
|
996
|
+
"peek_bytes",
|
|
997
|
+
"peek_path",
|
|
998
|
+
"peek_url",
|
|
999
|
+
"plain_column_type",
|
|
1000
|
+
]
|