mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
"""The polars backend: ``PolarsBackend`` over an eager ``pl.DataFrame`` container.
|
|
2
|
+
|
|
3
|
+
The direct twin of ``PandasBackend`` with the engine swapped. ``PolarsBackend`` re-owns the
|
|
4
|
+
narrow ``clean -> join -> select`` triple behind the ``Backend`` protocol; every other
|
|
5
|
+
candidate byte is still produced by the shared layer (the one pyarrow sink, the canonical
|
|
6
|
+
schema, ``_coerce_native_rows``, the evidence serializer). Because the shared layer
|
|
7
|
+
downstream is unchanged, the backend's job is to reproduce the reference compute
|
|
8
|
+
byte-identically while genuinely exercising a polars object.
|
|
9
|
+
|
|
10
|
+
Design (the load-bearing execution posture):
|
|
11
|
+
|
|
12
|
+
- The row-shape transforms (``clean``/``join``/``select``) are computed **value-wise** in
|
|
13
|
+
plain Python, mirroring ``pipeline._apply_cleaning`` / ``_join`` / ``_select`` exactly so
|
|
14
|
+
unicode-whitespace, empty-vs-null, ordering, and the typed ``BuildError`` refusals match
|
|
15
|
+
the reference bit-for-bit. Every value-level cast routes through the shared
|
|
16
|
+
``pipeline._cast`` — never an engine-native parser, which is what keeps cast results and
|
|
17
|
+
cast refusals identical across engines.
|
|
18
|
+
- The op's output is then materialized as an **eager** ``pl.DataFrame`` — one explicit
|
|
19
|
+
polars dtype per column drawn from the closed ``_POLARS_BY_LOGICAL`` vocabulary aligned to
|
|
20
|
+
the sink's ``pipeline._ARROW_BY_LOGICAL`` map, constructed from the already-typed native
|
|
21
|
+
values (never inferred, never an engine-native cast token, never a ``pl.DataFrame`` handed
|
|
22
|
+
to the sink). Explicit ``pl.Int64``/``pl.Float64``/``pl.String``/``pl.Boolean``/``pl.Date``/
|
|
23
|
+
UTC ``pl.Datetime`` columns carry
|
|
24
|
+
true polars nulls (never ``NaN``) and keep an ``int64`` column ``int64`` under nulls (no
|
|
25
|
+
int->float promotion).
|
|
26
|
+
- The frame is used **only** as a typed shaping/materialization container: it is never a
|
|
27
|
+
lazy frame, is never collected, never runs a polars ``join``/``group_by``, never enables the
|
|
28
|
+
global string cache, and never uses a polars categorical type. Order is imposed
|
|
29
|
+
positionally in Python (strictly stronger than any polars ordering guarantee), so polars'
|
|
30
|
+
documented ordering/streaming/thread nondeterminism can never touch the digest path, and the
|
|
31
|
+
streaming engine is structurally unreachable.
|
|
32
|
+
- The frame is converted back to an ordered native ``list[dict]`` at the boundary
|
|
33
|
+
(``_rows_from_frame``): column order = the frame's schema order, and each cell is read
|
|
34
|
+
through polars' native ``to_dicts()`` so a polars null becomes Python ``None``, a
|
|
35
|
+
``pl.String`` cell (Arrow ``large_utf8`` internally) becomes a native ``str`` with no width
|
|
36
|
+
tag, a ``pl.Date`` cell becomes ``datetime.date``, and an ``pl.Int64``/``pl.Float64`` cell
|
|
37
|
+
becomes a native ``int``/``float`` — no engine scalar. Because the backend returns native
|
|
38
|
+
rows (not a ``pl.DataFrame`` or its Arrow table), the ``large_utf8`` distinction is gone at
|
|
39
|
+
the boundary and the shared sink's ``pa.string()`` imposition normalizes it with no sink
|
|
40
|
+
change. ``pipeline._coerce_native_rows`` remains the shared backstop.
|
|
41
|
+
|
|
42
|
+
``import polars`` is **function-local** (mirroring ``reference.py`` / ``pandas_backend.py``'s
|
|
43
|
+
lazy-engine-import ethos), so ``import backends`` and constructing ``PolarsBackend()`` at
|
|
44
|
+
registry load never import polars. The module is named ``polars_backend`` (not ``polars``) to
|
|
45
|
+
avoid the ``import polars`` self-shadow footgun.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
from typing import Any
|
|
51
|
+
|
|
52
|
+
from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class PolarsBackend:
|
|
56
|
+
"""polars engine for the ``clean -> join -> select`` triple over an eager frame."""
|
|
57
|
+
|
|
58
|
+
# A plain literal (not a lazy ``pipeline`` import), so constructing the backend at
|
|
59
|
+
# ``registry`` module load never imports ``pipeline`` or polars. The seam resolves the
|
|
60
|
+
# backend by this exact name; any drift fails closed at ``get_backend`` (BACKEND_UNKNOWN).
|
|
61
|
+
name: str = "polars"
|
|
62
|
+
|
|
63
|
+
def versions(self) -> dict[str, str]:
|
|
64
|
+
# Provenance only (engine-runtime.v2 ``backend_versions``). The polars import is
|
|
65
|
+
# function-local (only reached during a build, when polars is already dispatched), so
|
|
66
|
+
# ``import backends`` / constructing the backend stay polars-free. The version is read
|
|
67
|
+
# from the module attribute ``pl.__version__`` rather than the stdlib metadata-version
|
|
68
|
+
# helper: that helper's module name is one of the dynamic-import tokens banned under
|
|
69
|
+
# ``backends/`` by ``test_backend_registry.py``. For a pinned release the two
|
|
70
|
+
# spellings are identical.
|
|
71
|
+
import polars as pl
|
|
72
|
+
|
|
73
|
+
return {"polars": str(pl.__version__)}
|
|
74
|
+
|
|
75
|
+
def materialize_graph_rows(
|
|
76
|
+
self,
|
|
77
|
+
columns: tuple[str, ...],
|
|
78
|
+
rows: tuple[tuple[Any, ...], ...],
|
|
79
|
+
) -> tuple[tuple[Any, ...], ...]:
|
|
80
|
+
"""Exercise a Polars object frame while preserving contract-owned Python scalars."""
|
|
81
|
+
|
|
82
|
+
import polars as pl
|
|
83
|
+
|
|
84
|
+
if any(len(row) != len(columns) for row in rows):
|
|
85
|
+
raise ValueError("graph row width differs from its columns")
|
|
86
|
+
frame = pl.DataFrame(
|
|
87
|
+
[
|
|
88
|
+
pl.Series(column, [row[index] for row in rows], dtype=pl.Object)
|
|
89
|
+
for index, column in enumerate(columns)
|
|
90
|
+
]
|
|
91
|
+
)
|
|
92
|
+
return tuple(tuple(row) for row in frame.iter_rows())
|
|
93
|
+
|
|
94
|
+
def clean(
|
|
95
|
+
self,
|
|
96
|
+
rows: list[dict[str, Any]],
|
|
97
|
+
columns: list[str],
|
|
98
|
+
lineage: dict[str, dict[str, Any]],
|
|
99
|
+
step: CleaningStep,
|
|
100
|
+
) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
|
|
101
|
+
# Value-wise reproduction of ``pipeline._apply_cleaning`` (the DataFrame is a shaping
|
|
102
|
+
# container, not a transform engine): shared ``_require_columns`` refusals, Python
|
|
103
|
+
# shared ASCII-only trim, ``== ""`` -> None, positional
|
|
104
|
+
# rename with ``RENAME_COLLISION``, and every cast through the shared ``_cast``.
|
|
105
|
+
from mostlyright.data_harness import pipeline
|
|
106
|
+
|
|
107
|
+
output_rows = [dict(row) for row in rows]
|
|
108
|
+
output_columns = list(columns)
|
|
109
|
+
output_lineage = {
|
|
110
|
+
column: {**entry, "operations": list(entry["operations"])}
|
|
111
|
+
for column, entry in lineage.items()
|
|
112
|
+
}
|
|
113
|
+
if step.operation in {"trim", "empty_to_null"}:
|
|
114
|
+
target_columns = tuple(step.columns)
|
|
115
|
+
pipeline._require_columns(output_columns, target_columns, step.operation)
|
|
116
|
+
for row in output_rows:
|
|
117
|
+
for column in target_columns:
|
|
118
|
+
value = row[column]
|
|
119
|
+
if step.operation == "trim":
|
|
120
|
+
if value is not None and not isinstance(value, str):
|
|
121
|
+
raise pipeline.BuildError(
|
|
122
|
+
"CLEAN_TYPE", f"trim requires strings in {column!r}"
|
|
123
|
+
)
|
|
124
|
+
row[column] = None if value is None else pipeline._trim_ascii(value)
|
|
125
|
+
elif value == "":
|
|
126
|
+
row[column] = None
|
|
127
|
+
for column in target_columns:
|
|
128
|
+
output_lineage[column]["operations"].append(step.operation)
|
|
129
|
+
elif step.operation == "rename":
|
|
130
|
+
mapping = dict(step.columns)
|
|
131
|
+
pipeline._require_columns(output_columns, tuple(mapping), "rename")
|
|
132
|
+
final_columns = [mapping.get(column, column) for column in output_columns]
|
|
133
|
+
if len(final_columns) != len(set(final_columns)):
|
|
134
|
+
raise pipeline.BuildError("RENAME_COLLISION", "rename creates a column collision")
|
|
135
|
+
output_rows = [
|
|
136
|
+
{mapping.get(column, column): row[column] for column in output_columns}
|
|
137
|
+
for row in output_rows
|
|
138
|
+
]
|
|
139
|
+
new_lineage: dict[str, dict[str, Any]] = {}
|
|
140
|
+
for column in output_columns:
|
|
141
|
+
target = mapping.get(column, column)
|
|
142
|
+
entry = output_lineage[column]
|
|
143
|
+
if target != column:
|
|
144
|
+
entry["operations"].append(f"rename:{column}->{target}")
|
|
145
|
+
new_lineage[target] = entry
|
|
146
|
+
output_columns = final_columns
|
|
147
|
+
output_lineage = new_lineage
|
|
148
|
+
elif step.operation == "cast":
|
|
149
|
+
mapping = dict(step.columns)
|
|
150
|
+
pipeline._require_columns(output_columns, tuple(mapping), "cast")
|
|
151
|
+
for row_index, row in enumerate(output_rows, start=1):
|
|
152
|
+
for column, target in mapping.items():
|
|
153
|
+
row[column] = pipeline._cast(row[column], target, column, row_index)
|
|
154
|
+
for column, target in mapping.items():
|
|
155
|
+
output_lineage[column]["operations"].append(f"cast:{target}")
|
|
156
|
+
else: # pragma: no cover - contract parser makes this unreachable
|
|
157
|
+
raise pipeline.ContractError(
|
|
158
|
+
"plan.cleaning.operation",
|
|
159
|
+
"CLEAN_OPERATION_UNSUPPORTED",
|
|
160
|
+
f"unsupported operation: {step.operation}",
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
shaped = _rows_from_frame(
|
|
164
|
+
_frame_from_rows(
|
|
165
|
+
output_rows, output_columns, _logical_by_column(output_lineage, output_columns)
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
return shaped, output_columns, output_lineage
|
|
169
|
+
|
|
170
|
+
def join(
|
|
171
|
+
self,
|
|
172
|
+
plan: TablePlan,
|
|
173
|
+
rows_by_source: dict[str, list[dict[str, Any]]],
|
|
174
|
+
columns_by_source: dict[str, list[str]],
|
|
175
|
+
lineage_by_source: dict[str, dict[str, dict[str, Any]]],
|
|
176
|
+
) -> tuple[
|
|
177
|
+
list[dict[str, Any]],
|
|
178
|
+
list[str],
|
|
179
|
+
dict[str, dict[str, Any]],
|
|
180
|
+
dict[str, Any],
|
|
181
|
+
]:
|
|
182
|
+
# Value-wise reproduction of ``pipeline._join``. Left-source order is imposed
|
|
183
|
+
# positionally (iterate the left rows; look up the right by an equality index) — never
|
|
184
|
+
# a polars ``join``/``group_by``, so polars' ordering nondeterminism never touches the
|
|
185
|
+
# digest. Same guards and typed refusals; output columns ``[*left_columns,
|
|
186
|
+
# *right_nonkeys]``; merged lineage; and a ``report`` byte-identical to the reference
|
|
187
|
+
# one. The report counts stay native ``int`` because they are pure-Python
|
|
188
|
+
# ``len(...)``; an engine scalar there would change the evidence bytes.
|
|
189
|
+
from mostlyright.data_harness import pipeline
|
|
190
|
+
|
|
191
|
+
spec = plan.join
|
|
192
|
+
left_rows = rows_by_source[spec.left]
|
|
193
|
+
right_rows = rows_by_source[spec.right]
|
|
194
|
+
left_columns = columns_by_source[spec.left]
|
|
195
|
+
right_columns = columns_by_source[spec.right]
|
|
196
|
+
pipeline._require_columns(left_columns, spec.on, "join.left keys")
|
|
197
|
+
pipeline._require_columns(right_columns, spec.on, "join.right keys")
|
|
198
|
+
for key in spec.on:
|
|
199
|
+
left_types = {type(row[key]) for row in left_rows if row[key] is not None}
|
|
200
|
+
right_types = {type(row[key]) for row in right_rows if row[key] is not None}
|
|
201
|
+
if any(row[key] is None for row in left_rows + right_rows):
|
|
202
|
+
raise pipeline.BuildError("JOIN_NULL_KEY", f"join key {key!r} contains null")
|
|
203
|
+
if len(left_types) != 1 or left_types != right_types:
|
|
204
|
+
raise pipeline.BuildError(
|
|
205
|
+
"JOIN_KEY_TYPE", f"join key {key!r} types do not match exactly"
|
|
206
|
+
)
|
|
207
|
+
right_nonkeys = [column for column in right_columns if column not in spec.on]
|
|
208
|
+
collisions = set(left_columns) & set(right_nonkeys)
|
|
209
|
+
if collisions:
|
|
210
|
+
raise pipeline.BuildError(
|
|
211
|
+
"JOIN_COLUMN_COLLISION", f"join column collision: {sorted(collisions)}"
|
|
212
|
+
)
|
|
213
|
+
right_index: dict[tuple[Any, ...], dict[str, Any]] = {}
|
|
214
|
+
for row in right_rows:
|
|
215
|
+
key = tuple(row[column] for column in spec.on)
|
|
216
|
+
if key in right_index:
|
|
217
|
+
raise pipeline.BuildError("JOIN_CARDINALITY", "right join keys are not unique")
|
|
218
|
+
right_index[key] = row
|
|
219
|
+
if spec.cardinality == "one_to_one":
|
|
220
|
+
left_keys = [tuple(row[column] for column in spec.on) for row in left_rows]
|
|
221
|
+
if len(left_keys) != len(set(left_keys)):
|
|
222
|
+
raise pipeline.BuildError("JOIN_CARDINALITY", "left join keys are not unique")
|
|
223
|
+
output: list[dict[str, Any]] = []
|
|
224
|
+
unmatched = 0
|
|
225
|
+
for left_row in left_rows:
|
|
226
|
+
key = tuple(left_row[column] for column in spec.on)
|
|
227
|
+
right_row = right_index.get(key)
|
|
228
|
+
if right_row is None:
|
|
229
|
+
unmatched += 1
|
|
230
|
+
merged = {**left_row, **{column: None for column in right_nonkeys}}
|
|
231
|
+
else:
|
|
232
|
+
merged = {**left_row, **{column: right_row[column] for column in right_nonkeys}}
|
|
233
|
+
output.append(merged)
|
|
234
|
+
multiplier = len(output) / len(left_rows)
|
|
235
|
+
if unmatched:
|
|
236
|
+
raise pipeline.BuildError("JOIN_UNMATCHED", f"left join has {unmatched} unmatched rows")
|
|
237
|
+
if multiplier != 1.0:
|
|
238
|
+
raise pipeline.BuildError("JOIN_MULTIPLIER", f"join row multiplier is {multiplier}")
|
|
239
|
+
columns = [*left_columns, *right_nonkeys]
|
|
240
|
+
lineage = {
|
|
241
|
+
**lineage_by_source[spec.left],
|
|
242
|
+
**{column: lineage_by_source[spec.right][column] for column in right_nonkeys},
|
|
243
|
+
}
|
|
244
|
+
report = {
|
|
245
|
+
"schema_version": "join-evidence.v1",
|
|
246
|
+
"left_source": spec.left,
|
|
247
|
+
"right_source": spec.right,
|
|
248
|
+
"keys": list(spec.on),
|
|
249
|
+
"cardinality": spec.cardinality,
|
|
250
|
+
"left_rows": len(left_rows),
|
|
251
|
+
"right_rows": len(right_rows),
|
|
252
|
+
"output_rows": len(output),
|
|
253
|
+
"matched_left_rows": len(output) - unmatched,
|
|
254
|
+
"unmatched_left_rows": unmatched,
|
|
255
|
+
"row_multiplier": multiplier,
|
|
256
|
+
"stable_order": "left_source_order",
|
|
257
|
+
}
|
|
258
|
+
shaped = _rows_from_frame(
|
|
259
|
+
_frame_from_rows(output, columns, _logical_by_column(lineage, columns))
|
|
260
|
+
)
|
|
261
|
+
return shaped, columns, lineage, report
|
|
262
|
+
|
|
263
|
+
def select(
|
|
264
|
+
self,
|
|
265
|
+
plan: TablePlan,
|
|
266
|
+
rows: list[dict[str, Any]],
|
|
267
|
+
columns: list[str],
|
|
268
|
+
lineage: dict[str, dict[str, Any]],
|
|
269
|
+
) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
|
|
270
|
+
# Value-wise reproduction of ``pipeline._select``: project/order columns to
|
|
271
|
+
# ``plan.select`` (shared ``_require_columns`` for the missing-column refusal);
|
|
272
|
+
# ``selected_lineage`` in ``plan.select`` order.
|
|
273
|
+
from mostlyright.data_harness import pipeline
|
|
274
|
+
|
|
275
|
+
pipeline._require_columns(columns, plan.select, "select")
|
|
276
|
+
selected_rows = [{column: row[column] for column in plan.select} for row in rows]
|
|
277
|
+
selected_lineage = {column: lineage[column] for column in plan.select}
|
|
278
|
+
shaped = _rows_from_frame(
|
|
279
|
+
_frame_from_rows(
|
|
280
|
+
selected_rows,
|
|
281
|
+
list(plan.select),
|
|
282
|
+
_logical_by_column(selected_lineage, plan.select),
|
|
283
|
+
)
|
|
284
|
+
)
|
|
285
|
+
return shaped, selected_lineage
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _logical_by_column(lineage: dict[str, dict[str, Any]], columns: Any) -> dict[str, str]:
|
|
289
|
+
"""Map each column to its closed logical type via the shared ``_logical_type_for``.
|
|
290
|
+
|
|
291
|
+
The logical type is drawn only from the column's lineage ``operations`` (the last
|
|
292
|
+
``cast:<target>``, else ``"string"``) — never from an engine scalar's class name — so
|
|
293
|
+
the polars dtype the frame is built with matches the sink's imposed schema.
|
|
294
|
+
"""
|
|
295
|
+
|
|
296
|
+
from mostlyright.data_harness import pipeline
|
|
297
|
+
|
|
298
|
+
return {column: pipeline._logical_type_for(lineage[column]) for column in columns}
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _polars_dtype_for(logical: str) -> Any:
|
|
302
|
+
"""Return the explicit polars dtype for one authoritative local logical type.
|
|
303
|
+
|
|
304
|
+
``_POLARS_BY_LOGICAL`` is aligned one-to-one with the sink's closed
|
|
305
|
+
``pipeline._ARROW_BY_LOGICAL`` vocabulary, so the frame's per-column dtype is canonical and
|
|
306
|
+
can never be inferred/lenient. Built with a function-local ``pl`` reference so ``import
|
|
307
|
+
backends`` never imports polars.
|
|
308
|
+
"""
|
|
309
|
+
|
|
310
|
+
import polars as pl
|
|
311
|
+
|
|
312
|
+
_POLARS_BY_LOGICAL = {
|
|
313
|
+
"string": pl.String,
|
|
314
|
+
"int64": pl.Int64,
|
|
315
|
+
"float64": pl.Float64,
|
|
316
|
+
"boolean": pl.Boolean,
|
|
317
|
+
"date": pl.Date,
|
|
318
|
+
"timestamp_utc": pl.Datetime(time_unit="us", time_zone="UTC"),
|
|
319
|
+
}
|
|
320
|
+
return _POLARS_BY_LOGICAL[logical]
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _frame_from_rows(
|
|
324
|
+
rows: list[dict[str, Any]], columns: Any, logical_by_column: dict[str, str]
|
|
325
|
+
) -> Any:
|
|
326
|
+
"""Build an EAGER ``pl.DataFrame`` with an explicit per-column dtype.
|
|
327
|
+
|
|
328
|
+
Each column is constructed as a ``pl.Series(name, values, dtype=<explicit>)`` from the
|
|
329
|
+
already-typed native values — never inferred, never an engine-native cast token, never a
|
|
330
|
+
``pl.DataFrame`` handed to the sink. Explicit closed dtypes carry true polars nulls (never
|
|
331
|
+
``NaN``) and keep ``int64`` as ``int64`` under nulls (no float promotion). The zero-column
|
|
332
|
+
shape is unreachable in the contract
|
|
333
|
+
(every stage yields >=1 output column) but is handled defensively — an empty frame — so a
|
|
334
|
+
degenerate op cannot raise here.
|
|
335
|
+
"""
|
|
336
|
+
|
|
337
|
+
import polars as pl
|
|
338
|
+
|
|
339
|
+
column_order = list(columns)
|
|
340
|
+
if not column_order:
|
|
341
|
+
return pl.DataFrame()
|
|
342
|
+
series = [
|
|
343
|
+
pl.Series(
|
|
344
|
+
column,
|
|
345
|
+
[row[column] for row in rows],
|
|
346
|
+
dtype=_polars_dtype_for(logical_by_column[column]),
|
|
347
|
+
)
|
|
348
|
+
for column in column_order
|
|
349
|
+
]
|
|
350
|
+
return pl.DataFrame(series)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _rows_from_frame(frame: Any) -> list[dict[str, Any]]:
|
|
354
|
+
"""Convert an eager ``pl.DataFrame`` back to an ordered native ``list[dict]``.
|
|
355
|
+
|
|
356
|
+
Column (key) order follows the frame's schema order; each cell is read through polars'
|
|
357
|
+
native ``to_dicts()`` so a polars null becomes Python ``None`` and every scalar is a native
|
|
358
|
+
Python builtin — ``pl.String`` (Arrow ``large_utf8``) -> ``str`` with no width tag,
|
|
359
|
+
``pl.Int64`` -> ``int``, ``pl.Float64`` -> ``float``, ``pl.Date`` -> ``datetime.date`` — no
|
|
360
|
+
engine scalar, no polars categorical type. Returning native rows (not a ``pl.DataFrame`` /
|
|
361
|
+
its Arrow table) is what strips the ``large_utf8`` width tag, so the shared sink's imposed
|
|
362
|
+
``pa.string()`` normalizes it with no sink change. ``pipeline._coerce_native_rows`` remains
|
|
363
|
+
the shared backstop downstream.
|
|
364
|
+
"""
|
|
365
|
+
|
|
366
|
+
return frame.to_dicts()
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""The named backend protocol at the ``clean -> join -> select`` row-transform seam.
|
|
2
|
+
|
|
3
|
+
The protocol boundary is the inner ``clean -> join -> select`` triple over ``list[dict]``
|
|
4
|
+
of native Python scalars — not the outer ``pipeline._replay_candidate_derivations``
|
|
5
|
+
signature. Making the whole replay pluggable would force every backend to re-own the
|
|
6
|
+
Parquet sink, the casts, validation and evidence, which is the entire byte-parity risk
|
|
7
|
+
surface. Keeping the seam narrow means the shared layer still produces every candidate
|
|
8
|
+
byte no matter which engine is selected.
|
|
9
|
+
|
|
10
|
+
The three method signatures are the signatures of ``pipeline._apply_cleaning`` /
|
|
11
|
+
``pipeline._join`` / ``pipeline._select`` (only ``self`` is added), so a backend is a
|
|
12
|
+
drop-in for those functions and ``reference.py`` can delegate to them unchanged.
|
|
13
|
+
|
|
14
|
+
A backend must be byte-identical to the reference implementation, not merely equivalent:
|
|
15
|
+
same row order, same column order, same Python scalar types, same typed refusals in the
|
|
16
|
+
same order. Any divergence changes the sealed candidate digest.
|
|
17
|
+
|
|
18
|
+
This module imports only ``local_contracts`` for typing and never ``pipeline``, which
|
|
19
|
+
keeps ``import backends`` free of a module-load dependency on ``pipeline``. ``pipeline``
|
|
20
|
+
imports the registry, so a top-level ``pipeline`` import here would close a cycle.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from typing import Any, Protocol, runtime_checkable
|
|
26
|
+
|
|
27
|
+
from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@runtime_checkable
|
|
31
|
+
class Backend(Protocol):
|
|
32
|
+
"""A named engine for the ``clean -> join -> select`` row-transform triple."""
|
|
33
|
+
|
|
34
|
+
name: str
|
|
35
|
+
|
|
36
|
+
def versions(self) -> dict[str, str]:
|
|
37
|
+
"""Engine-library versions for provenance (``{}`` for a pure-Python backend)."""
|
|
38
|
+
...
|
|
39
|
+
|
|
40
|
+
def materialize_graph_rows(
|
|
41
|
+
self,
|
|
42
|
+
columns: tuple[str, ...],
|
|
43
|
+
rows: tuple[tuple[Any, ...], ...],
|
|
44
|
+
) -> tuple[tuple[Any, ...], ...]:
|
|
45
|
+
"""Round-trip one graph node through this engine's deterministic container."""
|
|
46
|
+
...
|
|
47
|
+
|
|
48
|
+
def clean(
|
|
49
|
+
self,
|
|
50
|
+
rows: list[dict[str, Any]],
|
|
51
|
+
columns: list[str],
|
|
52
|
+
lineage: dict[str, dict[str, Any]],
|
|
53
|
+
step: CleaningStep,
|
|
54
|
+
) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
|
|
55
|
+
"""Apply one cleaning step and return ``(rows, columns, lineage)``.
|
|
56
|
+
|
|
57
|
+
``step.operation`` is one of ``trim`` / ``empty_to_null`` / ``rename`` / ``cast``.
|
|
58
|
+
The inputs must not be mutated; return fresh structures. Row order is the input
|
|
59
|
+
order. ``columns`` is the output column order (only ``rename`` changes it, and it
|
|
60
|
+
renames in place, positionally). ``lineage`` keeps one entry per output column,
|
|
61
|
+
with the applied operation appended to that entry's ``operations`` list
|
|
62
|
+
(``rename:<from>-><to>`` and ``cast:<target>`` carry their argument).
|
|
63
|
+
|
|
64
|
+
Required typed refusals, all ``pipeline.BuildError``: ``COLUMN_MISSING`` when the
|
|
65
|
+
step names a column that is absent, ``CLEAN_TYPE`` when ``trim`` meets a non-string
|
|
66
|
+
non-null value, ``RENAME_COLLISION`` when a rename would produce duplicate column
|
|
67
|
+
names, and ``CAST_INVALID`` / ``CAST_TYPE`` from the cast. Casts must route through
|
|
68
|
+
``pipeline._cast`` rather than an engine-native parser, so cast results and cast
|
|
69
|
+
refusals stay identical across engines. Reference implementation:
|
|
70
|
+
``pipeline._apply_cleaning``.
|
|
71
|
+
"""
|
|
72
|
+
...
|
|
73
|
+
|
|
74
|
+
def join(
|
|
75
|
+
self,
|
|
76
|
+
plan: TablePlan,
|
|
77
|
+
rows_by_source: dict[str, list[dict[str, Any]]],
|
|
78
|
+
columns_by_source: dict[str, list[str]],
|
|
79
|
+
lineage_by_source: dict[str, dict[str, dict[str, Any]]],
|
|
80
|
+
) -> tuple[
|
|
81
|
+
list[dict[str, Any]],
|
|
82
|
+
list[str],
|
|
83
|
+
dict[str, dict[str, Any]],
|
|
84
|
+
dict[str, Any],
|
|
85
|
+
]:
|
|
86
|
+
"""Run ``plan.join`` and return ``(rows, columns, lineage, report)``.
|
|
87
|
+
|
|
88
|
+
One left equality join on ``plan.join.on``. Output row order is left-source order,
|
|
89
|
+
positionally: iterate the left rows and look up the right row by key. Do not rely
|
|
90
|
+
on an engine join operator's ordering. Output columns are
|
|
91
|
+
``[*left_columns, *right_non_key_columns]``; lineage merges the two sources under
|
|
92
|
+
those columns.
|
|
93
|
+
|
|
94
|
+
Required typed refusals, all ``pipeline.BuildError``: ``COLUMN_MISSING`` for an
|
|
95
|
+
absent key, ``JOIN_NULL_KEY`` for a null key value on either side,
|
|
96
|
+
``JOIN_KEY_TYPE`` when the left key does not have exactly one observed non-null
|
|
97
|
+
value type, or when the left and right key value types are not exactly equal
|
|
98
|
+
(so two empty sources refuse, even though their type sets are both empty),
|
|
99
|
+
``JOIN_COLUMN_COLLISION`` when a right non-key column name already exists on the
|
|
100
|
+
left, ``JOIN_CARDINALITY`` for non-unique right keys (or non-unique left keys under
|
|
101
|
+
``one_to_one``), ``JOIN_UNMATCHED`` for any unmatched left row, and
|
|
102
|
+
``JOIN_MULTIPLIER`` when the output/left row ratio is not exactly 1.
|
|
103
|
+
|
|
104
|
+
``report`` is the ``join-evidence.v1`` payload that is serialized into candidate
|
|
105
|
+
evidence, so its counts must be plain Python ``int`` — never an engine scalar type.
|
|
106
|
+
Reference implementation: ``pipeline._join``.
|
|
107
|
+
"""
|
|
108
|
+
...
|
|
109
|
+
|
|
110
|
+
def select(
|
|
111
|
+
self,
|
|
112
|
+
plan: TablePlan,
|
|
113
|
+
rows: list[dict[str, Any]],
|
|
114
|
+
columns: list[str],
|
|
115
|
+
lineage: dict[str, dict[str, Any]],
|
|
116
|
+
) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
|
|
117
|
+
"""Project to ``plan.select`` and return ``(rows, lineage)``.
|
|
118
|
+
|
|
119
|
+
Every returned row carries exactly the columns of ``plan.select``, in that order;
|
|
120
|
+
``lineage`` is restricted and reordered the same way. Row order is the input order.
|
|
121
|
+
Raise ``pipeline.BuildError("COLUMN_MISSING", ...)`` when ``plan.select`` names a
|
|
122
|
+
column that is not present. Reference implementation: ``pipeline._select``.
|
|
123
|
+
"""
|
|
124
|
+
...
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""The reference backend: the pure-Python compute behind the protocol.
|
|
2
|
+
|
|
3
|
+
``ReferenceBackend`` is a thin dispatch adapter that delegates ``clean`` / ``join`` /
|
|
4
|
+
``select`` to ``pipeline._apply_cleaning`` / ``pipeline._join`` / ``pipeline._select``.
|
|
5
|
+
It must never re-implement that compute: delegation is what makes its bytes provably the
|
|
6
|
+
shared layer's own, which is why the sealed goldens reproduce byte-for-byte when the build
|
|
7
|
+
seam dispatches through it.
|
|
8
|
+
|
|
9
|
+
The compute functions live in ``pipeline``, and ``pipeline`` imports the registry, so
|
|
10
|
+
importing ``pipeline`` at module top would form a
|
|
11
|
+
``pipeline -> registry -> reference -> pipeline`` cycle. Each method therefore imports
|
|
12
|
+
``pipeline`` function-locally, so ``import backends`` never triggers ``import pipeline``
|
|
13
|
+
at module-load time.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from mostlyright.data_harness.local_contracts import CleaningStep, TablePlan
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ReferenceBackend:
|
|
24
|
+
"""Pure-Python reference engine delegating to the shared ``pipeline`` compute."""
|
|
25
|
+
|
|
26
|
+
# Must stay equal to ``pipeline.REFERENCE_BACKEND_NAME``, which is the single source of
|
|
27
|
+
# truth. Deliberately a plain literal here — not a lazy ``pipeline`` import — so
|
|
28
|
+
# constructing the backend (done at ``registry`` module load) never imports
|
|
29
|
+
# ``pipeline`` and cannot form the import cycle. Drift fails closed at build time: the
|
|
30
|
+
# seam resolves ``get_backend(pipeline.REFERENCE_BACKEND_NAME)``, which raises
|
|
31
|
+
# ``BACKEND_UNKNOWN`` if this name no longer matches.
|
|
32
|
+
name: str = "reference"
|
|
33
|
+
|
|
34
|
+
def versions(self) -> dict[str, str]:
|
|
35
|
+
# Pure Python — no external engine library to version.
|
|
36
|
+
return {}
|
|
37
|
+
|
|
38
|
+
def materialize_graph_rows(
|
|
39
|
+
self,
|
|
40
|
+
columns: tuple[str, ...],
|
|
41
|
+
rows: tuple[tuple[Any, ...], ...],
|
|
42
|
+
) -> tuple[tuple[Any, ...], ...]:
|
|
43
|
+
if any(len(row) != len(columns) for row in rows):
|
|
44
|
+
raise ValueError("graph row width differs from its columns")
|
|
45
|
+
return tuple(tuple(row) for row in rows)
|
|
46
|
+
|
|
47
|
+
def clean(
|
|
48
|
+
self,
|
|
49
|
+
rows: list[dict[str, Any]],
|
|
50
|
+
columns: list[str],
|
|
51
|
+
lineage: dict[str, dict[str, Any]],
|
|
52
|
+
step: CleaningStep,
|
|
53
|
+
) -> tuple[list[dict[str, Any]], list[str], dict[str, dict[str, Any]]]:
|
|
54
|
+
from mostlyright.data_harness import pipeline
|
|
55
|
+
|
|
56
|
+
return pipeline._apply_cleaning(rows, columns, lineage, step)
|
|
57
|
+
|
|
58
|
+
def join(
|
|
59
|
+
self,
|
|
60
|
+
plan: TablePlan,
|
|
61
|
+
rows_by_source: dict[str, list[dict[str, Any]]],
|
|
62
|
+
columns_by_source: dict[str, list[str]],
|
|
63
|
+
lineage_by_source: dict[str, dict[str, dict[str, Any]]],
|
|
64
|
+
) -> tuple[
|
|
65
|
+
list[dict[str, Any]],
|
|
66
|
+
list[str],
|
|
67
|
+
dict[str, dict[str, Any]],
|
|
68
|
+
dict[str, Any],
|
|
69
|
+
]:
|
|
70
|
+
from mostlyright.data_harness import pipeline
|
|
71
|
+
|
|
72
|
+
return pipeline._join(plan, rows_by_source, columns_by_source, lineage_by_source)
|
|
73
|
+
|
|
74
|
+
def select(
|
|
75
|
+
self,
|
|
76
|
+
plan: TablePlan,
|
|
77
|
+
rows: list[dict[str, Any]],
|
|
78
|
+
columns: list[str],
|
|
79
|
+
lineage: dict[str, dict[str, Any]],
|
|
80
|
+
) -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]:
|
|
81
|
+
from mostlyright.data_harness import pipeline
|
|
82
|
+
|
|
83
|
+
return pipeline._select(plan, rows, columns, lineage)
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Explicit named backend registration and fail-closed dispatch.
|
|
2
|
+
|
|
3
|
+
The registry is a literal dict populated at import time by explicit named registration.
|
|
4
|
+
No dynamic module import, no packaging entry-point discovery, no resource-based plugin
|
|
5
|
+
lookup, and no runtime import hook of any kind is permitted anywhere under ``backends/``:
|
|
6
|
+
an engine that can be swapped in at runtime is an engine that can change sealed candidate
|
|
7
|
+
bytes without a code review. ``tests/test_backend_registry.py`` statically asserts none of
|
|
8
|
+
those dynamic-import tokens appear in this package.
|
|
9
|
+
|
|
10
|
+
``get_backend`` fails closed on an unknown name with a typed
|
|
11
|
+
``BuildError("BACKEND_UNKNOWN", ...)`` — never a silent default engine, because silently
|
|
12
|
+
falling back would produce bytes under an engine name the caller did not select.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from mostlyright.data_harness.backends.pandas_backend import PandasBackend
|
|
18
|
+
from mostlyright.data_harness.backends.polars_backend import PolarsBackend
|
|
19
|
+
from mostlyright.data_harness.backends.protocol import Backend
|
|
20
|
+
from mostlyright.data_harness.backends.reference import ReferenceBackend
|
|
21
|
+
|
|
22
|
+
# Explicit, literal registration — the only way a backend enters the table. Constructing
|
|
23
|
+
# ``PandasBackend()``/``PolarsBackend()`` here does NOT import pandas/polars: each engine
|
|
24
|
+
# import is function-local, so ``import backends`` stays import-light (the lazy-import
|
|
25
|
+
# contract, mirrored from the reference). ``get_backend`` still fails closed on an unknown
|
|
26
|
+
# name (BACKEND_UNKNOWN).
|
|
27
|
+
_REGISTRY: dict[str, Backend] = {
|
|
28
|
+
ReferenceBackend.name: ReferenceBackend(),
|
|
29
|
+
PandasBackend.name: PandasBackend(),
|
|
30
|
+
PolarsBackend.name: PolarsBackend(),
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def get_backend(name: str) -> Backend:
|
|
35
|
+
"""Return the registered backend for ``name`` or fail closed.
|
|
36
|
+
|
|
37
|
+
An unknown/unregistered name is a typed refusal
|
|
38
|
+
(``BuildError("BACKEND_UNKNOWN", ...)``), never a silent default backend.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
backend = _REGISTRY.get(name)
|
|
42
|
+
if backend is None:
|
|
43
|
+
# Lazy import keeps ``import backends`` free of a module-load dependency on
|
|
44
|
+
# ``pipeline``; ``pipeline`` imports this module, so a top-level import would
|
|
45
|
+
# cycle. Only this failure path needs ``BuildError``.
|
|
46
|
+
from mostlyright.data_harness.pipeline import BuildError
|
|
47
|
+
|
|
48
|
+
raise BuildError("BACKEND_UNKNOWN", f"no registered backend {name!r}")
|
|
49
|
+
return backend
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def registered_backends() -> frozenset[str]:
|
|
53
|
+
"""Return the frozenset of registered backend names."""
|
|
54
|
+
|
|
55
|
+
return frozenset(_REGISTRY)
|