mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1710 @@
|
|
|
1
|
+
"""Deterministic nbformat-4.5 visual sidecar for a sealed candidate.
|
|
2
|
+
|
|
3
|
+
The generated ``table.ipynb`` is useful without a kernel: it embeds a data preview and an
|
|
4
|
+
all-column analytics inspector as normal ``display_data`` outputs. The inspector's versioned rich
|
|
5
|
+
MIME payload includes logical and physical types, declared units, missingness, cardinality,
|
|
6
|
+
type-appropriate summaries and distributions, lineage, and transformations. Portable
|
|
7
|
+
``text/html`` and ``text/plain`` fallbacks keep the notebook useful in ordinary Jupyter viewers.
|
|
8
|
+
|
|
9
|
+
Runnable pandas cells remain available for local exploration. Generated code cells carry
|
|
10
|
+
nbclient's ``skip-execution`` tag, preserving the deterministic pre-filled outputs during headless
|
|
11
|
+
execution while remaining manually runnable in an interactive kernel.
|
|
12
|
+
|
|
13
|
+
The generator is deterministic: every cell ``id`` is the first 12 hex of ``sha256(cell_source)``
|
|
14
|
+
and every value is a native Python scalar, so two renders of the same sealed candidate are
|
|
15
|
+
byte-identical. The provenance block is derived the same way — no clock, no randomness.
|
|
16
|
+
|
|
17
|
+
It adds NO new dependency: the notebook is nbformat 4.5 JSON built with the stdlib ``json``, and
|
|
18
|
+
the head-N Parquet sample is read through the already-pinned ``pyarrow`` (imported lazily). It never
|
|
19
|
+
imports ``nbformat``/``nbconvert``.
|
|
20
|
+
|
|
21
|
+
The sidecar is written as a run-dir SIBLING of ``candidate/`` (the run dir stays owner-writable
|
|
22
|
+
``0o700``; only ``candidate/`` is sealed ``0o555``/``0o444``), so it never enters the sealed member
|
|
23
|
+
set, never moves a digest, and is invisible to ``verify_candidate`` (which polices only
|
|
24
|
+
``candidate/``).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import hashlib
|
|
30
|
+
import html
|
|
31
|
+
import json
|
|
32
|
+
import math
|
|
33
|
+
import os
|
|
34
|
+
import secrets
|
|
35
|
+
import stat
|
|
36
|
+
from collections import Counter
|
|
37
|
+
from copy import deepcopy
|
|
38
|
+
from decimal import Decimal
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
from typing import Any
|
|
41
|
+
|
|
42
|
+
try:
|
|
43
|
+
import fcntl
|
|
44
|
+
except ImportError: # pragma: no cover - secure candidate verification already refuses Windows
|
|
45
|
+
fcntl = None # type: ignore[assignment]
|
|
46
|
+
|
|
47
|
+
from mostlyright.data_harness import pipeline
|
|
48
|
+
from mostlyright.data_harness import recipe as recipe_contracts
|
|
49
|
+
from mostlyright.data_harness.local_contracts import UNIT_STATES
|
|
50
|
+
from mostlyright.data_harness.preparation.contracts import PHYSICAL_TYPES
|
|
51
|
+
from mostlyright.data_harness.recipe import LOGICAL_TYPES, UNITS
|
|
52
|
+
from mostlyright.data_harness.units import is_declarable_unit, legacy_aliases
|
|
53
|
+
|
|
54
|
+
# The one name this module writes the generated notebook under. It is a name other lanes have
|
|
55
|
+
# to be able to say without writing it themselves -- the hosted handoff reports the path it
|
|
56
|
+
# installed and never opens it -- so it is a constant rather than four literals.
|
|
57
|
+
TABLE_NOTEBOOK_NAME = "table.ipynb"
|
|
58
|
+
# Private compatibility import for the hosted handoff lane added before the V3 vocabulary cutover.
|
|
59
|
+
# It names the same Table sidecar and never crosses a wire or UI boundary.
|
|
60
|
+
DATASET_NOTEBOOK_NAME = TABLE_NOTEBOOK_NAME
|
|
61
|
+
_SAMPLE_ROWS = 10
|
|
62
|
+
_NBFORMAT_MAJOR = 4
|
|
63
|
+
_NBFORMAT_MINOR = 5
|
|
64
|
+
_PROVENANCE_DIGEST_PREFIX = 12
|
|
65
|
+
# ``nbclient``'s default ``skip_cells_with_tag``. Every generated code cell carries it, so the
|
|
66
|
+
# headless execute path skips them and the hand-built display outputs survive untouched.
|
|
67
|
+
_SKIP_EXECUTION_TAG = "skip-execution"
|
|
68
|
+
_COLUMN_PROFILE_MIME = "application/vnd.mostlyright.column-profile.v1+json"
|
|
69
|
+
_PROFILE_SAMPLE_ROWS = 100_000
|
|
70
|
+
_MAX_CATEGORY_BARS = 12
|
|
71
|
+
_FALLBACK_MAX_COLUMNS = 64
|
|
72
|
+
_FALLBACK_MAX_LIMITATIONS = 16
|
|
73
|
+
_FALLBACK_MAX_CATEGORIES = 16
|
|
74
|
+
_FALLBACK_TEXT_LIMIT = 512
|
|
75
|
+
|
|
76
|
+
_UNIT_DISPLAY: dict[str, tuple[str | None, str]] = {
|
|
77
|
+
"none": (None, "Not applicable"),
|
|
78
|
+
"count": ("count", "count"),
|
|
79
|
+
"ratio": ("ratio", "unitless ratio"),
|
|
80
|
+
"percent": ("%", "percent"),
|
|
81
|
+
"celsius": ("°C", "degrees Celsius"),
|
|
82
|
+
"fahrenheit": ("°F", "degrees Fahrenheit"),
|
|
83
|
+
"kelvin": ("K", "kelvin"),
|
|
84
|
+
"meter": ("m", "meters"),
|
|
85
|
+
"kilometer": ("km", "kilometers"),
|
|
86
|
+
"mile": ("mi", "miles"),
|
|
87
|
+
"microgram_per_cubic_meter": ("µg/m³", "micrograms per cubic meter"),
|
|
88
|
+
"second": ("s", "seconds"),
|
|
89
|
+
"minute": ("min", "minutes"),
|
|
90
|
+
"hour": ("h", "hours"),
|
|
91
|
+
"hectopascal": ("hPa", "hectopascals"),
|
|
92
|
+
}
|
|
93
|
+
# These are the written-out labels for the fifteen tokens the vocabulary began with, not a second
|
|
94
|
+
# vocabulary. Import-time equality holds them honest: an alias added to the pinned table without a
|
|
95
|
+
# label here is an explicit implementation failure rather than a silent gap.
|
|
96
|
+
assert set(_UNIT_DISPLAY) == set(UNITS)
|
|
97
|
+
# The same fifteen labels, reached from the codes those names stand for. ``celsius`` and ``Cel``
|
|
98
|
+
# are one unit, so the page a person reads has to say so: without this, adopting the spelling the
|
|
99
|
+
# unit grammar promotes would silently downgrade "degrees Celsius" to "Cel".
|
|
100
|
+
_CODE_DISPLAY: dict[str, tuple[str | None, str]] = {
|
|
101
|
+
code: _UNIT_DISPLAY[token] for token, code in legacy_aliases().items()
|
|
102
|
+
}
|
|
103
|
+
_UNIT_STATE_DISPLAY = {
|
|
104
|
+
"declared": "Declared physical unit",
|
|
105
|
+
"normalized": "Normalized (dimensionless)",
|
|
106
|
+
"not_applicable": "Not applicable",
|
|
107
|
+
"unknown": "Unit not recoverable",
|
|
108
|
+
}
|
|
109
|
+
assert set(_UNIT_STATE_DISPLAY) == set(UNIT_STATES)
|
|
110
|
+
assert set(PHYSICAL_TYPES) == set(LOGICAL_TYPES)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# --- nbformat 4.5 cell primitives --------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _cell_id(source: str) -> str:
|
|
117
|
+
"""First 12 hex of ``sha256(source)`` — a content-derived, stable, deterministic cell id."""
|
|
118
|
+
|
|
119
|
+
return hashlib.sha256(source.encode("utf-8")).hexdigest()[:12]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _markdown_cell(source: str) -> dict[str, Any]:
|
|
123
|
+
return {
|
|
124
|
+
"cell_type": "markdown",
|
|
125
|
+
"id": _cell_id(source),
|
|
126
|
+
"metadata": {},
|
|
127
|
+
"source": source,
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _code_cell(source: str, outputs: tuple[dict[str, Any], ...] = ()) -> dict[str, Any]:
|
|
132
|
+
"""A generated code cell, tagged ``skip-execution`` so a headless execute leaves it alone.
|
|
133
|
+
|
|
134
|
+
The tag is cell METADATA, not cell source, so the content-derived ``id`` is unaffected and the
|
|
135
|
+
render stays byte-deterministic.
|
|
136
|
+
"""
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
"cell_type": "code",
|
|
140
|
+
"execution_count": None,
|
|
141
|
+
"id": _cell_id(source),
|
|
142
|
+
"metadata": {"tags": [_SKIP_EXECUTION_TAG]},
|
|
143
|
+
"outputs": list(outputs),
|
|
144
|
+
"source": source,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _display_html(markup: str) -> dict[str, Any]:
|
|
149
|
+
return {"output_type": "display_data", "data": {"text/html": markup}, "metadata": {}}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _fallback_text(value: Any, *, limit: int = _FALLBACK_TEXT_LIMIT) -> tuple[str, bool]:
|
|
153
|
+
"""Return deterministic bounded text and whether an explicit truncation marker was added."""
|
|
154
|
+
|
|
155
|
+
rendered = "" if value is None else str(value)
|
|
156
|
+
if len(rendered) <= limit:
|
|
157
|
+
return rendered, False
|
|
158
|
+
marker = " [truncated]"
|
|
159
|
+
return rendered[: max(0, limit - len(marker))] + marker, True
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _bounded_items(value: Any, maximum: int) -> tuple[list[Any], int]:
|
|
163
|
+
items = value if isinstance(value, list) else []
|
|
164
|
+
return items[:maximum], max(0, len(items) - maximum)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _fallback_semantic_sections(payload: dict[str, Any]) -> tuple[str, list[str]]:
|
|
168
|
+
"""Render bounded deterministic dataset and column semantics for portable fallbacks."""
|
|
169
|
+
|
|
170
|
+
semantics = payload.get("semantics") if isinstance(payload.get("semantics"), dict) else {}
|
|
171
|
+
message, _ = _fallback_text(
|
|
172
|
+
semantics.get(
|
|
173
|
+
"message",
|
|
174
|
+
"This Build declares no column meanings. Nothing here is inferred from column names.",
|
|
175
|
+
)
|
|
176
|
+
)
|
|
177
|
+
summary, _ = _fallback_text(semantics.get("summary") or "Not declared")
|
|
178
|
+
grain, _ = _fallback_text(semantics.get("grain_statement") or "Not declared")
|
|
179
|
+
coverage = semantics.get("coverage") if isinstance(semantics.get("coverage"), dict) else {}
|
|
180
|
+
time_range = (
|
|
181
|
+
coverage.get("time_range") if isinstance(coverage.get("time_range"), dict) else None
|
|
182
|
+
)
|
|
183
|
+
if time_range is None:
|
|
184
|
+
time_text = "Not declared"
|
|
185
|
+
else:
|
|
186
|
+
start, _ = _fallback_text(time_range.get("start_inclusive") or "Not declared")
|
|
187
|
+
end, _ = _fallback_text(time_range.get("end_exclusive") or "Not declared")
|
|
188
|
+
time_text = f"{start} (inclusive) to {end} (exclusive)"
|
|
189
|
+
population, _ = _fallback_text(coverage.get("population") or "Not declared")
|
|
190
|
+
completeness, _ = _fallback_text(coverage.get("completeness_note") or "Not declared")
|
|
191
|
+
|
|
192
|
+
limitations, omitted_limitations = _bounded_items(
|
|
193
|
+
semantics.get("limitations"), _FALLBACK_MAX_LIMITATIONS
|
|
194
|
+
)
|
|
195
|
+
limitation_texts = [_fallback_text(item)[0] for item in limitations]
|
|
196
|
+
if not limitation_texts:
|
|
197
|
+
limitation_texts = ["None declared"]
|
|
198
|
+
|
|
199
|
+
html_parts = [
|
|
200
|
+
'<section class="mostlyright-table-semantics">',
|
|
201
|
+
"<h3>Table semantics</h3>",
|
|
202
|
+
f"<p>{html.escape(message)}</p>",
|
|
203
|
+
f"<p><strong>Summary:</strong> {html.escape(summary)}</p>",
|
|
204
|
+
f"<p><strong>One row:</strong> {html.escape(grain)}</p>",
|
|
205
|
+
f"<p><strong>Time coverage:</strong> {html.escape(time_text)}</p>",
|
|
206
|
+
f"<p><strong>Population:</strong> {html.escape(population)}</p>",
|
|
207
|
+
f"<p><strong>Completeness:</strong> {html.escape(completeness)}</p>",
|
|
208
|
+
"<h4>Known limitations</h4><ul>",
|
|
209
|
+
*(f"<li>{html.escape(item)}</li>" for item in limitation_texts),
|
|
210
|
+
]
|
|
211
|
+
plain = [
|
|
212
|
+
"Table semantics",
|
|
213
|
+
message,
|
|
214
|
+
f"Summary: {summary}",
|
|
215
|
+
f"One row: {grain}",
|
|
216
|
+
f"Time coverage: {time_text}",
|
|
217
|
+
f"Population: {population}",
|
|
218
|
+
f"Completeness: {completeness}",
|
|
219
|
+
"Known limitations:",
|
|
220
|
+
*(f"- {item}" for item in limitation_texts),
|
|
221
|
+
]
|
|
222
|
+
if omitted_limitations:
|
|
223
|
+
notice = f"[truncated: {omitted_limitations} additional limitations omitted]"
|
|
224
|
+
html_parts.append(f"<li>{html.escape(notice)}</li>")
|
|
225
|
+
plain.append(notice)
|
|
226
|
+
html_parts.extend(["</ul>", "</section>"])
|
|
227
|
+
|
|
228
|
+
columns = payload.get("columns") if isinstance(payload.get("columns"), list) else []
|
|
229
|
+
displayed_columns = columns[:_FALLBACK_MAX_COLUMNS]
|
|
230
|
+
omitted_columns = max(0, len(columns) - _FALLBACK_MAX_COLUMNS)
|
|
231
|
+
header = (
|
|
232
|
+
"<tr><th>column</th><th>label</th><th>meaning</th><th>type</th><th>unit / scaling</th>"
|
|
233
|
+
"<th>categorical definitions</th><th>evidence</th></tr>"
|
|
234
|
+
)
|
|
235
|
+
rows: list[str] = []
|
|
236
|
+
plain.extend(["", "Column semantics"])
|
|
237
|
+
for column in displayed_columns:
|
|
238
|
+
if not isinstance(column, dict):
|
|
239
|
+
continue
|
|
240
|
+
name, _ = _fallback_text(column.get("name", ""), limit=120)
|
|
241
|
+
label, _ = _fallback_text(column.get("display_label") or "Not declared", limit=120)
|
|
242
|
+
description, _ = _fallback_text(column.get("description") or "Not declared")
|
|
243
|
+
logical, _ = _fallback_text(column.get("logical_type", "unknown"), limit=64)
|
|
244
|
+
physical, _ = _fallback_text(column.get("physical_type", "unknown"), limit=64)
|
|
245
|
+
unit = column.get("unit") if isinstance(column.get("unit"), dict) else {}
|
|
246
|
+
unit_text, _ = _fallback_text(unit.get("symbol") or unit.get("label") or "not declared")
|
|
247
|
+
if unit.get("state") == "normalized":
|
|
248
|
+
normalization, _ = _fallback_text(unit.get("normalization") or "unknown", limit=64)
|
|
249
|
+
reversible = "reversible" if unit.get("reversible") else "not reversible"
|
|
250
|
+
unit_text = f"{unit_text}; {normalization}; {reversible}"
|
|
251
|
+
|
|
252
|
+
categories, omitted_categories = _bounded_items(
|
|
253
|
+
column.get("categories"), _FALLBACK_MAX_CATEGORIES
|
|
254
|
+
)
|
|
255
|
+
category_parts = []
|
|
256
|
+
for item in categories:
|
|
257
|
+
if isinstance(item, dict):
|
|
258
|
+
code, _ = _fallback_text(item.get("code"), limit=64)
|
|
259
|
+
category_label, _ = _fallback_text(item.get("label"), limit=120)
|
|
260
|
+
category_parts.append(f"{code} = {category_label}")
|
|
261
|
+
category_status = str(column.get("categories_status") or "not declared").replace("_", " ")
|
|
262
|
+
category_text = (
|
|
263
|
+
f"{category_status}: {'; '.join(category_parts)}" if category_parts else category_status
|
|
264
|
+
)
|
|
265
|
+
if omitted_categories:
|
|
266
|
+
category_text += f"; [truncated: {omitted_categories} additional definitions omitted]"
|
|
267
|
+
|
|
268
|
+
refs, omitted_refs = _bounded_items(column.get("evidence_refs"), 8)
|
|
269
|
+
evidence_parts = [
|
|
270
|
+
f"{item.get('kind')}: {item.get('ref')}" for item in refs if isinstance(item, dict)
|
|
271
|
+
]
|
|
272
|
+
evidence_text = "; ".join(evidence_parts) or "No evidence reference declared"
|
|
273
|
+
if omitted_refs:
|
|
274
|
+
evidence_text += f"; [truncated: {omitted_refs} additional references omitted]"
|
|
275
|
+
|
|
276
|
+
rows.append(
|
|
277
|
+
"<tr>"
|
|
278
|
+
f"<td>{html.escape(name)}</td><td>{html.escape(label)}</td>"
|
|
279
|
+
f"<td>{html.escape(description)}</td>"
|
|
280
|
+
f"<td>{html.escape(logical)} / {html.escape(physical)}</td>"
|
|
281
|
+
f"<td>{html.escape(unit_text)}</td>"
|
|
282
|
+
f"<td>{html.escape(category_text)}</td>"
|
|
283
|
+
f"<td>{html.escape(evidence_text)}</td></tr>"
|
|
284
|
+
)
|
|
285
|
+
plain.extend(
|
|
286
|
+
[
|
|
287
|
+
f"- {name} — {label}",
|
|
288
|
+
f" Meaning: {description}",
|
|
289
|
+
f" Type: {logical} / {physical}",
|
|
290
|
+
f" Unit / scaling: {unit_text}",
|
|
291
|
+
f" Categorical definitions: {category_text}",
|
|
292
|
+
f" Evidence: {evidence_text}",
|
|
293
|
+
]
|
|
294
|
+
)
|
|
295
|
+
if omitted_columns:
|
|
296
|
+
notice = f"[truncated: {omitted_columns} additional columns omitted]"
|
|
297
|
+
rows.append(f'<tr><td colspan="7">{html.escape(notice)}</td></tr>')
|
|
298
|
+
plain.append(notice)
|
|
299
|
+
html_parts.extend(
|
|
300
|
+
[
|
|
301
|
+
'<section class="mostlyright-column-semantics">',
|
|
302
|
+
"<h3>Column semantics</h3>",
|
|
303
|
+
'<table border="1" class="dataframe"><thead>',
|
|
304
|
+
header,
|
|
305
|
+
f"</thead><tbody>{''.join(rows)}</tbody></table>",
|
|
306
|
+
"</section>",
|
|
307
|
+
]
|
|
308
|
+
)
|
|
309
|
+
return "".join(html_parts), plain
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _display_column_profiles(
|
|
313
|
+
payload: dict[str, Any], *, fallback_payload: dict[str, Any] | None = None
|
|
314
|
+
) -> dict[str, Any]:
|
|
315
|
+
"""Portable custom output with bounded, complete semantic fallbacks."""
|
|
316
|
+
|
|
317
|
+
raw_columns = payload.get("columns", [])
|
|
318
|
+
columns = raw_columns[:_FALLBACK_MAX_COLUMNS] if isinstance(raw_columns, list) else []
|
|
319
|
+
semantic_html, plain = _fallback_semantic_sections(fallback_payload or payload)
|
|
320
|
+
header = (
|
|
321
|
+
"<tr><th>column</th><th>label</th><th>logical</th><th>physical</th><th>unit state</th>"
|
|
322
|
+
"<th>missing</th><th>distinct</th></tr>"
|
|
323
|
+
)
|
|
324
|
+
rows: list[str] = []
|
|
325
|
+
plain.extend(["", "Column inspector"])
|
|
326
|
+
for column in columns if isinstance(columns, list) else []:
|
|
327
|
+
if not isinstance(column, dict):
|
|
328
|
+
continue
|
|
329
|
+
unit = column.get("unit") if isinstance(column.get("unit"), dict) else {}
|
|
330
|
+
unit_text = unit.get("symbol") or unit.get("label") or "not declared"
|
|
331
|
+
name = str(column.get("name", ""))
|
|
332
|
+
label = str(column.get("display_label") or "Not declared")
|
|
333
|
+
logical = str(column.get("logical_type", "unknown"))
|
|
334
|
+
physical = str(column.get("physical_type", "unknown"))
|
|
335
|
+
missing = int(column.get("missing_count", 0) or 0)
|
|
336
|
+
distinct = int(column.get("distinct_count", 0) or 0)
|
|
337
|
+
rows.append(
|
|
338
|
+
"<tr>"
|
|
339
|
+
f"<td>{html.escape(name)}</td><td>{html.escape(label)}</td>"
|
|
340
|
+
f"<td>{html.escape(logical)}</td>"
|
|
341
|
+
f"<td>{html.escape(physical)}</td><td>{html.escape(str(unit_text))}</td>"
|
|
342
|
+
f"<td>{missing}</td><td>{distinct}</td></tr>"
|
|
343
|
+
)
|
|
344
|
+
plain.append(
|
|
345
|
+
f"- {name} ({label}): {logical} / {physical}; unit {unit_text}; "
|
|
346
|
+
f"{missing} missing; {distinct} distinct"
|
|
347
|
+
)
|
|
348
|
+
omitted_inspector_columns = max(
|
|
349
|
+
0, len(raw_columns) - _FALLBACK_MAX_COLUMNS if isinstance(raw_columns, list) else 0
|
|
350
|
+
)
|
|
351
|
+
if omitted_inspector_columns:
|
|
352
|
+
notice = f"[truncated: {omitted_inspector_columns} additional profiles omitted]"
|
|
353
|
+
rows.append(f'<tr><td colspan="7">{html.escape(notice)}</td></tr>')
|
|
354
|
+
plain.append(notice)
|
|
355
|
+
fallback = (
|
|
356
|
+
'<div class="mostlyright-column-profile">'
|
|
357
|
+
f"{semantic_html}"
|
|
358
|
+
"<p><strong>Column inspector</strong> — open this notebook in Mostly Right for charts "
|
|
359
|
+
"and interactive column selection.</p>"
|
|
360
|
+
'<table border="1" class="dataframe"><thead>'
|
|
361
|
+
f"{header}</thead><tbody>{''.join(rows)}</tbody></table></div>"
|
|
362
|
+
)
|
|
363
|
+
return {
|
|
364
|
+
"output_type": "display_data",
|
|
365
|
+
"data": {
|
|
366
|
+
_COLUMN_PROFILE_MIME: payload,
|
|
367
|
+
"text/html": fallback,
|
|
368
|
+
"text/plain": "\n".join(plain),
|
|
369
|
+
},
|
|
370
|
+
"metadata": {},
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
# --- hand-built, kernel-free display outputs ---------------------------------------------------
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def _cell_text(value: Any) -> str:
|
|
378
|
+
return "" if value is None else str(value)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def _html_table(rows: list[dict[str, Any]], columns: list[str]) -> dict[str, Any]:
|
|
382
|
+
"""A hand-built HTML ``<table>`` of the head-N sample rows (every cell HTML-escaped)."""
|
|
383
|
+
|
|
384
|
+
# Match pandas' structural index signal: a blank leading header and a row-header ``th``. The
|
|
385
|
+
# renderer can then distinguish the index from real data columns without guessing from values.
|
|
386
|
+
header = "<th></th>" + "".join(f"<th>{html.escape(str(column))}</th>" for column in columns)
|
|
387
|
+
body_rows: list[str] = []
|
|
388
|
+
for index, row in enumerate(rows):
|
|
389
|
+
cells = "".join(
|
|
390
|
+
f"<td>{html.escape(_cell_text(row.get(column)))}</td>" for column in columns
|
|
391
|
+
)
|
|
392
|
+
body_rows.append(f"<tr><th>{index}</th>{cells}</tr>")
|
|
393
|
+
if not body_rows:
|
|
394
|
+
body_rows.append(
|
|
395
|
+
f'<tr><th></th><td colspan="{max(1, len(columns))}"><em>no rows</em></td></tr>'
|
|
396
|
+
)
|
|
397
|
+
markup = (
|
|
398
|
+
'<table border="1" class="dataframe">'
|
|
399
|
+
f"<thead><tr>{header}</tr></thead>"
|
|
400
|
+
f"<tbody>{''.join(body_rows)}</tbody>"
|
|
401
|
+
"</table>"
|
|
402
|
+
)
|
|
403
|
+
return _display_html(markup)
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
# --- markdown evidence sections ----------------------------------------------------------------
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _esc(value: Any) -> str:
|
|
410
|
+
"""HTML-escape an evidence-derived value before interpolating it into a markdown cell.
|
|
411
|
+
|
|
412
|
+
These cells are rendered to HTML by the viewer's ``HTMLExporter``, which passes raw HTML in
|
|
413
|
+
markdown through live — so every evidence-derived value (question, profiled types, lineage ops,
|
|
414
|
+
quality rows, join facts, source origin/rights) is escaped here, exactly as the sample-row table
|
|
415
|
+
and SVG labels already escape. Escaping is safe in every markdown context (code span, table
|
|
416
|
+
cell, or bold flow): an escaped ``<`` never re-opens as a tag even if the value breaks a
|
|
417
|
+
backtick.
|
|
418
|
+
"""
|
|
419
|
+
|
|
420
|
+
return html.escape("" if value is None else str(value))
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def _header_markdown(question: str, row_count: int, columns: int) -> str:
|
|
424
|
+
title = question.strip().rstrip("?") or "Research dataset"
|
|
425
|
+
return f"# {_esc(title)}\n\n{row_count:,} rows · {columns} columns."
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _overview_markdown(semantics: dict[str, Any]) -> str:
|
|
429
|
+
lines = [
|
|
430
|
+
"## Dataset contract",
|
|
431
|
+
"",
|
|
432
|
+
_esc(semantics.get("message", "Semantic status unavailable.")),
|
|
433
|
+
]
|
|
434
|
+
summary = semantics.get("summary")
|
|
435
|
+
grain = semantics.get("grain_statement")
|
|
436
|
+
coverage = semantics.get("coverage")
|
|
437
|
+
if isinstance(summary, str):
|
|
438
|
+
lines.extend(["", "### Definition", "", _esc(summary)])
|
|
439
|
+
else:
|
|
440
|
+
lines.extend(["", "### Definition", "", "No dataset-level summary was declared."])
|
|
441
|
+
if isinstance(grain, str):
|
|
442
|
+
lines.append(f"**Row grain:** {_esc(grain)}")
|
|
443
|
+
else:
|
|
444
|
+
lines.append("**Row grain:** Not declared.")
|
|
445
|
+
lines.extend(["", "### Coverage", ""])
|
|
446
|
+
if isinstance(coverage, dict):
|
|
447
|
+
time_range = coverage.get("time_range")
|
|
448
|
+
if isinstance(time_range, dict):
|
|
449
|
+
lines.append(
|
|
450
|
+
"**Time:** "
|
|
451
|
+
f"{_esc(time_range.get('start_inclusive'))} (inclusive) to "
|
|
452
|
+
f"{_esc(time_range.get('end_exclusive'))} (exclusive)."
|
|
453
|
+
)
|
|
454
|
+
else:
|
|
455
|
+
lines.append("**Time:** Not declared.")
|
|
456
|
+
lines.append(f"**Population:** {_esc(coverage.get('population'))}")
|
|
457
|
+
lines.append(f"**Completeness:** {_esc(coverage.get('completeness_note'))}")
|
|
458
|
+
limitations = semantics.get("limitations")
|
|
459
|
+
lines.extend(["", "### Known limitations", ""])
|
|
460
|
+
if isinstance(limitations, list) and limitations:
|
|
461
|
+
lines.extend(f"- {_esc(item)}" for item in limitations)
|
|
462
|
+
elif isinstance(limitations, list):
|
|
463
|
+
lines.append("No limitations were declared.")
|
|
464
|
+
else:
|
|
465
|
+
lines.append("No limitations statement was declared.")
|
|
466
|
+
return "\n".join(lines)
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def _dictionary_markdown(
|
|
470
|
+
column_names: list[str],
|
|
471
|
+
semantics_by_name: dict[str, dict[str, Any]],
|
|
472
|
+
logical_by_name: dict[str, str],
|
|
473
|
+
) -> str:
|
|
474
|
+
headings = ("Column", "Label", "Meaning", "Type", "Unit / scaling", "Values", "Evidence")
|
|
475
|
+
rows: list[str] = []
|
|
476
|
+
|
|
477
|
+
def table_value(value: Any) -> str:
|
|
478
|
+
return (
|
|
479
|
+
" ".join(("" if value is None else str(value)).split())
|
|
480
|
+
.replace("\\", r"\\")
|
|
481
|
+
.replace("|", r"\|")
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
for name in column_names:
|
|
485
|
+
record = semantics_by_name.get(name)
|
|
486
|
+
unit = _unit_metadata(record)
|
|
487
|
+
unit_text = unit["label"]
|
|
488
|
+
if unit.get("state") == "normalized":
|
|
489
|
+
reversible = "reversible" if unit.get("reversible") else "not reversible"
|
|
490
|
+
unit_text = f"{unit_text}: {unit.get('normalization')} ({reversible})"
|
|
491
|
+
categories = record.get("categories") if isinstance(record, dict) else None
|
|
492
|
+
category_status = (
|
|
493
|
+
record.get("categories_status", "unknown") if isinstance(record, dict) else "unknown"
|
|
494
|
+
)
|
|
495
|
+
category_text = str(category_status).replace("_", " ")
|
|
496
|
+
if isinstance(categories, list) and categories:
|
|
497
|
+
definitions = "; ".join(
|
|
498
|
+
f"{item.get('code')} = {item.get('label')}"
|
|
499
|
+
for item in categories
|
|
500
|
+
if isinstance(item, dict)
|
|
501
|
+
)
|
|
502
|
+
category_text = f"{category_text}: {definitions}"
|
|
503
|
+
refs = record.get("evidence_refs") if isinstance(record, dict) else None
|
|
504
|
+
evidence = "; ".join(
|
|
505
|
+
f"{item.get('kind')}: {item.get('ref')}"
|
|
506
|
+
for item in refs or []
|
|
507
|
+
if isinstance(item, dict)
|
|
508
|
+
)
|
|
509
|
+
values = (
|
|
510
|
+
name,
|
|
511
|
+
record.get("display_label", "Not declared")
|
|
512
|
+
if isinstance(record, dict)
|
|
513
|
+
else "Not declared",
|
|
514
|
+
record.get("description", "Not declared")
|
|
515
|
+
if isinstance(record, dict)
|
|
516
|
+
else "Not declared",
|
|
517
|
+
logical_by_name.get(name, "unknown"),
|
|
518
|
+
unit_text,
|
|
519
|
+
category_text,
|
|
520
|
+
evidence or "No evidence reference declared",
|
|
521
|
+
)
|
|
522
|
+
rows.append("| " + " | ".join(table_value(value) for value in values) + " |")
|
|
523
|
+
header = "| " + " | ".join(headings) + " |"
|
|
524
|
+
rule = "| " + " | ".join("---" for _heading in headings) + " |"
|
|
525
|
+
return "\n".join(("### Data dictionary", "", header, rule, *rows))
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _quality_markdown(quality: dict[str, Any]) -> str:
|
|
529
|
+
checks = quality.get("checks", []) if isinstance(quality, dict) else []
|
|
530
|
+
passed = sum(bool(check.get("passed")) for check in checks if isinstance(check, dict))
|
|
531
|
+
failed = [check for check in checks if isinstance(check, dict) and not check.get("passed")]
|
|
532
|
+
lines = [
|
|
533
|
+
"## Validation",
|
|
534
|
+
"",
|
|
535
|
+
"### Build checks",
|
|
536
|
+
"",
|
|
537
|
+
f"{passed} of {len(checks)} checks passed.",
|
|
538
|
+
]
|
|
539
|
+
for check in failed:
|
|
540
|
+
lines.append(f"- {_esc(check.get('check_id', 'Check'))}: needs attention")
|
|
541
|
+
return "\n".join(lines)
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _join_markdown(join: dict[str, Any]) -> str:
|
|
545
|
+
keys = ", ".join(f"`{_esc(key)}`" for key in join.get("keys", [])) or "—"
|
|
546
|
+
lines = [
|
|
547
|
+
"### Join",
|
|
548
|
+
"",
|
|
549
|
+
f"- **cardinality:** {_esc(join.get('cardinality', '—'))}",
|
|
550
|
+
f"- **keys:** {keys}",
|
|
551
|
+
f"- **left:** `{_esc(join.get('left_source', '—'))}` "
|
|
552
|
+
f"({_esc(join.get('left_rows', '—'))} rows)",
|
|
553
|
+
f"- **right:** `{_esc(join.get('right_source', '—'))}` "
|
|
554
|
+
f"({_esc(join.get('right_rows', '—'))} rows)",
|
|
555
|
+
f"- **output rows:** {_esc(join.get('output_rows', '—'))}",
|
|
556
|
+
f"- **unmatched left rows:** {_esc(join.get('unmatched_left_rows', '—'))}",
|
|
557
|
+
]
|
|
558
|
+
return "\n".join(lines)
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _operations_markdown(evidence: dict[str, Any]) -> str:
|
|
562
|
+
nodes = evidence.get("nodes", []) if isinstance(evidence, dict) else []
|
|
563
|
+
lines = ["## How this dataset was built", "", "### Deterministic operations", ""]
|
|
564
|
+
for node in nodes if isinstance(nodes, list) else []:
|
|
565
|
+
if not isinstance(node, dict):
|
|
566
|
+
continue
|
|
567
|
+
inputs = ", ".join(f"`{_esc(item)}`" for item in node.get("inputs", [])) or "source"
|
|
568
|
+
lines.append(
|
|
569
|
+
f"- **{_esc(node.get('node_id', 'node'))}** — "
|
|
570
|
+
f"`{_esc(node.get('operation', 'operation'))}` from {inputs}; "
|
|
571
|
+
f"{_esc(node.get('input_rows', '—'))} → {_esc(node.get('output_rows', '—'))} rows"
|
|
572
|
+
)
|
|
573
|
+
parameters = node.get("parameters")
|
|
574
|
+
if isinstance(parameters, dict):
|
|
575
|
+
rendered = json.dumps(
|
|
576
|
+
parameters,
|
|
577
|
+
ensure_ascii=False,
|
|
578
|
+
sort_keys=True,
|
|
579
|
+
separators=(",", ":"),
|
|
580
|
+
)
|
|
581
|
+
markdown_safe = rendered.replace("`", "\\u0060")
|
|
582
|
+
lines.append(f" - parameters: `{markdown_safe}`")
|
|
583
|
+
inspectable = evidence.get("schema_version") == pipeline.GRAPH_OPERATION_EVIDENCE_VERSION
|
|
584
|
+
lines.extend(
|
|
585
|
+
[
|
|
586
|
+
"",
|
|
587
|
+
(
|
|
588
|
+
"Exact parameters, ordering, lineage, engine identity, and logical digests are "
|
|
589
|
+
"sealed in `evidence/operations.json`."
|
|
590
|
+
if inspectable
|
|
591
|
+
else "Parameter digests, ordering, lineage, engine identity, and logical digests "
|
|
592
|
+
"are sealed in `evidence/operations.json`."
|
|
593
|
+
),
|
|
594
|
+
]
|
|
595
|
+
)
|
|
596
|
+
return "\n".join(lines)
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
def _sources_markdown(sources: tuple[dict[str, Any], ...]) -> str:
|
|
600
|
+
lines = ["## Provenance", "", "### Sources", ""]
|
|
601
|
+
for source in sources:
|
|
602
|
+
source = source if isinstance(source, dict) else {}
|
|
603
|
+
rights = source.get("rights", {})
|
|
604
|
+
status = rights.get("status", "unspecified") if isinstance(rights, dict) else "unspecified"
|
|
605
|
+
origin = source.get("origin", "")
|
|
606
|
+
lines.append(
|
|
607
|
+
f"- **{_esc(source.get('source_id', 'source'))}** — {_esc(origin) or 'local source'} "
|
|
608
|
+
f"({_esc(status).replace('_', ' ')})"
|
|
609
|
+
)
|
|
610
|
+
return "\n".join(lines)
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
def _provenance_metadata(inspection: Any) -> dict[str, str]:
|
|
614
|
+
"""Build ``metadata.mostlyright.provenance`` from the rendered evidence members.
|
|
615
|
+
|
|
616
|
+
Four string fields in renderer order — ``source``, ``join``, ``revision``, ``cache`` — each
|
|
617
|
+
derived from the already-read build reports (``evidence/sources.json``, ``evidence/join.json``,
|
|
618
|
+
the verified manifest identity). No clock and no randomness: two renders of the same candidate
|
|
619
|
+
produce the same block, exactly like the cells. Values are raw text, not HTML-escaped — a
|
|
620
|
+
renderer escapes metadata at interpolation, so escaping here would double-encode.
|
|
621
|
+
"""
|
|
622
|
+
|
|
623
|
+
source_ids = [
|
|
624
|
+
str(source.get("source_id", ""))
|
|
625
|
+
for source in inspection.sources
|
|
626
|
+
if isinstance(source, dict) and source.get("source_id")
|
|
627
|
+
]
|
|
628
|
+
join = inspection.join if isinstance(inspection.join, dict) else {}
|
|
629
|
+
if join.get("schema_version") in {
|
|
630
|
+
pipeline.GRAPH_OPERATION_EVIDENCE_V1,
|
|
631
|
+
pipeline.GRAPH_OPERATION_EVIDENCE_VERSION,
|
|
632
|
+
}:
|
|
633
|
+
nodes = join.get("nodes", [])
|
|
634
|
+
join_text = f"{len(nodes) if isinstance(nodes, list) else 0} graph operations"
|
|
635
|
+
else:
|
|
636
|
+
keys = [str(key) for key in (join.get("keys") or [])]
|
|
637
|
+
cardinality = str(join.get("cardinality") or "")
|
|
638
|
+
if cardinality and keys:
|
|
639
|
+
join_text = f"{cardinality} on {', '.join(keys)}"
|
|
640
|
+
else:
|
|
641
|
+
join_text = cardinality or "none"
|
|
642
|
+
snapshots = len(source_ids)
|
|
643
|
+
noun = "snapshot" if snapshots == 1 else "snapshots"
|
|
644
|
+
digest = inspection.candidate_digest[:_PROVENANCE_DIGEST_PREFIX]
|
|
645
|
+
return {
|
|
646
|
+
"source": ", ".join(source_ids) or "none",
|
|
647
|
+
"join": join_text,
|
|
648
|
+
# Digest first: a renderer dims the parenthetical tail, and the identity is the digest —
|
|
649
|
+
# the manifest version is the format constant.
|
|
650
|
+
"revision": f"{digest} ({inspection.manifest_version})",
|
|
651
|
+
"cache": f"{snapshots} raw {noun} in raw/",
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
# --- all-column analytics ---------------------------------------------------------------------
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def _number(value: int | float | Decimal) -> str:
|
|
659
|
+
"""Format a number without first narrowing an integer through binary64."""
|
|
660
|
+
|
|
661
|
+
if isinstance(value, int):
|
|
662
|
+
return f"{value:,}"
|
|
663
|
+
if isinstance(value, Decimal):
|
|
664
|
+
if not value.is_finite():
|
|
665
|
+
return "Not finite"
|
|
666
|
+
# Keep exact fixed-point rendering across the complete int64 range, while avoiding a
|
|
667
|
+
# hundreds-of-characters label if a binary64-derived Decimal reaches an extreme exponent.
|
|
668
|
+
if value and (value.adjusted() >= 20 or value.adjusted() <= -5):
|
|
669
|
+
return _trim_scientific(format(value, ".4g"))
|
|
670
|
+
if value == value.to_integral_value():
|
|
671
|
+
return f"{int(value):,}"
|
|
672
|
+
return format(value, ",f")
|
|
673
|
+
if not math.isfinite(value):
|
|
674
|
+
return "Not finite"
|
|
675
|
+
# ``float.is_integer`` is also true for 1e308. Expanding that value through ``int`` creates a
|
|
676
|
+
# 309-character chart label, so reserve exact-looking grouped integers for ordinary magnitudes.
|
|
677
|
+
if value and (abs(value) >= 1e20 or abs(value) < 1e-4):
|
|
678
|
+
return _trim_scientific(f"{value:.4g}")
|
|
679
|
+
if math.isfinite(value) and value.is_integer():
|
|
680
|
+
return f"{int(value):,}"
|
|
681
|
+
return f"{value:,.4g}"
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
def _trim_scientific(value: str) -> str:
|
|
685
|
+
"""Normalize a compact general-format number without changing its significant digits."""
|
|
686
|
+
|
|
687
|
+
rendered = value.lower()
|
|
688
|
+
if "e" not in rendered:
|
|
689
|
+
return rendered
|
|
690
|
+
mantissa, exponent = rendered.split("e", 1)
|
|
691
|
+
mantissa = mantissa.rstrip("0").rstrip(".")
|
|
692
|
+
return f"{mantissa}e{exponent}"
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def _unit_metadata(semantics: dict[str, Any] | None) -> dict[str, Any]:
|
|
696
|
+
if semantics is None:
|
|
697
|
+
return {
|
|
698
|
+
"value": None,
|
|
699
|
+
"symbol": None,
|
|
700
|
+
"label": "Not declared",
|
|
701
|
+
"status": "unknown",
|
|
702
|
+
"state": "unknown",
|
|
703
|
+
"normalization": None,
|
|
704
|
+
"reversible": None,
|
|
705
|
+
}
|
|
706
|
+
raw = str(semantics.get("unit", "none"))
|
|
707
|
+
semantic_type = str(semantics.get("semantic_type", ""))
|
|
708
|
+
state = semantics.get("unit_state")
|
|
709
|
+
if state not in UNIT_STATES:
|
|
710
|
+
state = (
|
|
711
|
+
"declared"
|
|
712
|
+
if raw != "none"
|
|
713
|
+
else ("unknown" if semantic_type == "measure" else "not_applicable")
|
|
714
|
+
)
|
|
715
|
+
if state == "normalized":
|
|
716
|
+
return {
|
|
717
|
+
"value": raw,
|
|
718
|
+
"symbol": None,
|
|
719
|
+
"label": _UNIT_STATE_DISPLAY[state],
|
|
720
|
+
"status": state,
|
|
721
|
+
"state": state,
|
|
722
|
+
"normalization": semantics.get("normalization"),
|
|
723
|
+
"reversible": semantics.get("reversible"),
|
|
724
|
+
}
|
|
725
|
+
if state == "not_applicable":
|
|
726
|
+
return {
|
|
727
|
+
"value": raw,
|
|
728
|
+
"symbol": None,
|
|
729
|
+
"label": _UNIT_STATE_DISPLAY[state],
|
|
730
|
+
"status": "not_applicable",
|
|
731
|
+
"state": state,
|
|
732
|
+
"normalization": None,
|
|
733
|
+
"reversible": None,
|
|
734
|
+
}
|
|
735
|
+
if state == "unknown":
|
|
736
|
+
return {
|
|
737
|
+
"value": raw,
|
|
738
|
+
"symbol": None,
|
|
739
|
+
"label": _UNIT_STATE_DISPLAY[state],
|
|
740
|
+
"status": "unknown",
|
|
741
|
+
"state": state,
|
|
742
|
+
"normalization": None,
|
|
743
|
+
"reversible": None,
|
|
744
|
+
}
|
|
745
|
+
# A declared unit that is neither one of the written-out names nor a code the grammar resolves
|
|
746
|
+
# cannot have got past Recipe validation. Keep this branch fail-visible for a caller building a
|
|
747
|
+
# fixture without going through that validation.
|
|
748
|
+
if raw not in _UNIT_DISPLAY and raw not in _CODE_DISPLAY and not is_declarable_unit(raw):
|
|
749
|
+
return {
|
|
750
|
+
"value": raw,
|
|
751
|
+
"symbol": None,
|
|
752
|
+
"label": "Unknown unit",
|
|
753
|
+
"status": "unknown",
|
|
754
|
+
"state": "unknown",
|
|
755
|
+
"normalization": None,
|
|
756
|
+
"reversible": None,
|
|
757
|
+
}
|
|
758
|
+
symbol, label = _UNIT_DISPLAY.get(raw) or _CODE_DISPLAY.get(raw) or (raw, raw)
|
|
759
|
+
return {
|
|
760
|
+
"value": raw,
|
|
761
|
+
"symbol": symbol,
|
|
762
|
+
"label": label,
|
|
763
|
+
"status": "declared",
|
|
764
|
+
"state": "declared",
|
|
765
|
+
"normalization": None,
|
|
766
|
+
"reversible": None,
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
|
|
770
|
+
def _value_text(value: Any) -> str:
|
|
771
|
+
if value is None:
|
|
772
|
+
return "null"
|
|
773
|
+
if isinstance(value, float):
|
|
774
|
+
if math.isnan(value):
|
|
775
|
+
return "NaN"
|
|
776
|
+
# ``repr`` is the shortest round-trippable spelling. Category labels therefore cannot
|
|
777
|
+
# merge near-equal binary64 values merely because the chart uses compact typography.
|
|
778
|
+
if value == 0.0:
|
|
779
|
+
return "0.0"
|
|
780
|
+
return repr(value)
|
|
781
|
+
return str(value)
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
def _analysis_note(analysis_rows: int, total_rows: int) -> str:
|
|
785
|
+
if analysis_rows >= total_rows:
|
|
786
|
+
return f"{analysis_rows:,} observations · all rows"
|
|
787
|
+
return f"{analysis_rows:,} observations · first {analysis_rows:,} of {total_rows:,} rows"
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def _histogram(
|
|
791
|
+
values: list[int | float],
|
|
792
|
+
*,
|
|
793
|
+
unit_symbol: str | None,
|
|
794
|
+
analysis_rows: int,
|
|
795
|
+
total_rows: int,
|
|
796
|
+
) -> dict[str, Any]:
|
|
797
|
+
if not values:
|
|
798
|
+
return {"title": "Distribution", "note": "No non-missing values", "items": []}
|
|
799
|
+
minimum, maximum = min(values), max(values)
|
|
800
|
+
suffix = f" {unit_symbol}" if unit_symbol else ""
|
|
801
|
+
if minimum == maximum:
|
|
802
|
+
return {
|
|
803
|
+
"title": "Distribution",
|
|
804
|
+
"note": _analysis_note(analysis_rows, total_rows),
|
|
805
|
+
"items": [{"label": f"{_number(minimum)}{suffix}", "count": len(values)}],
|
|
806
|
+
}
|
|
807
|
+
bins = min(8, max(4, round(math.sqrt(len(values)))))
|
|
808
|
+
counts = [0] * bins
|
|
809
|
+
items = []
|
|
810
|
+
if all(isinstance(value, int) for value in values):
|
|
811
|
+
# Integer arithmetic throughout: neither +/-2**53 nor int64 extrema are rounded through a
|
|
812
|
+
# float. Inclusive integer buckets are gap-free and cover both extrema exactly.
|
|
813
|
+
integer_minimum = int(minimum)
|
|
814
|
+
integer_maximum = int(maximum)
|
|
815
|
+
span = integer_maximum - integer_minimum + 1
|
|
816
|
+
bins = min(bins, span)
|
|
817
|
+
counts = [0] * bins
|
|
818
|
+
for value in values:
|
|
819
|
+
index = min(bins - 1, ((int(value) - integer_minimum) * bins) // span)
|
|
820
|
+
counts[index] += 1
|
|
821
|
+
for index, count in enumerate(counts):
|
|
822
|
+
low = integer_minimum + ((span * index + bins - 1) // bins)
|
|
823
|
+
high = integer_minimum + ((span * (index + 1) + bins - 1) // bins) - 1
|
|
824
|
+
items.append({"label": f"{_number(low)} to {_number(high)}{suffix}", "count": count})
|
|
825
|
+
else:
|
|
826
|
+
float_minimum = float(minimum)
|
|
827
|
+
float_maximum = float(maximum)
|
|
828
|
+
crosses_zero = float_minimum < 0.0 < float_maximum
|
|
829
|
+
if crosses_zero:
|
|
830
|
+
# Scaling first keeps the normalized span in [1, 2]. Directly subtracting valid
|
|
831
|
+
# binary64 extrema (for example -1e308 and 1e308) would overflow to infinity.
|
|
832
|
+
scale = max(-float_minimum, float_maximum)
|
|
833
|
+
scaled_minimum = float_minimum / scale
|
|
834
|
+
scaled_maximum = float_maximum / scale
|
|
835
|
+
scaled_span = scaled_maximum - scaled_minimum
|
|
836
|
+
|
|
837
|
+
def position(value: int | float) -> float:
|
|
838
|
+
return ((float(value) / scale) - scaled_minimum) / scaled_span
|
|
839
|
+
|
|
840
|
+
def boundary(fraction: float) -> float:
|
|
841
|
+
# A convex combination never exceeds either finite endpoint. Unlike
|
|
842
|
+
# ``minimum + fraction * (maximum - minimum)``, it needs no overflowing span.
|
|
843
|
+
return float_minimum * (1.0 - fraction) + float_maximum * fraction
|
|
844
|
+
|
|
845
|
+
else:
|
|
846
|
+
span = float_maximum - float_minimum
|
|
847
|
+
|
|
848
|
+
def position(value: int | float) -> float:
|
|
849
|
+
return (float(value) - float_minimum) / span
|
|
850
|
+
|
|
851
|
+
def boundary(fraction: float) -> float:
|
|
852
|
+
return float_minimum + span * fraction
|
|
853
|
+
|
|
854
|
+
for value in values:
|
|
855
|
+
index = min(bins - 1, max(0, int(position(value) * bins)))
|
|
856
|
+
counts[index] += 1
|
|
857
|
+
for index, count in enumerate(counts):
|
|
858
|
+
low = boundary(index / bins)
|
|
859
|
+
high = boundary((index + 1) / bins)
|
|
860
|
+
items.append({"label": f"{_number(low)} to {_number(high)}{suffix}", "count": count})
|
|
861
|
+
return {
|
|
862
|
+
"title": "Distribution",
|
|
863
|
+
"note": _analysis_note(analysis_rows, total_rows),
|
|
864
|
+
"items": items,
|
|
865
|
+
}
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
def _is_missing(value: Any) -> bool:
|
|
869
|
+
"""Notebook policy: null and every NaN payload are missing; signed zero is observed."""
|
|
870
|
+
|
|
871
|
+
return value is None or (isinstance(value, float) and math.isnan(value))
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def _distinct_key(value: Any) -> tuple[str, Any]:
|
|
875
|
+
"""Typed exact identity for cardinality/counting, separate from display formatting.
|
|
876
|
+
|
|
877
|
+
All NaNs are excluded by ``_is_missing``. IEEE signed zeros intentionally form one value,
|
|
878
|
+
matching Arrow/Python equality semantics for an analytical distinct count.
|
|
879
|
+
"""
|
|
880
|
+
|
|
881
|
+
if _is_missing(value): # pragma: no cover - guarded by callers, retained as an invariant
|
|
882
|
+
raise ValueError("missing values do not have a distinct key")
|
|
883
|
+
if isinstance(value, bool):
|
|
884
|
+
return ("boolean", value)
|
|
885
|
+
if isinstance(value, int):
|
|
886
|
+
return ("int64", value)
|
|
887
|
+
if isinstance(value, float):
|
|
888
|
+
return ("float64", 0.0 if value == 0.0 else value)
|
|
889
|
+
return (type(value).__qualname__, value)
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
def _value_counts(
|
|
893
|
+
values: list[Any],
|
|
894
|
+
*,
|
|
895
|
+
title: str = "Value counts",
|
|
896
|
+
analysis_rows: int,
|
|
897
|
+
total_rows: int,
|
|
898
|
+
) -> dict[str, Any]:
|
|
899
|
+
keyed = Counter(_distinct_key(value) for value in values if not _is_missing(value))
|
|
900
|
+
counts = [(_value_text(key[1]), count) for key, count in keyed.items()]
|
|
901
|
+
ordered = sorted(counts, key=lambda item: (-item[1], item[0]))
|
|
902
|
+
visible = ordered[:_MAX_CATEGORY_BARS]
|
|
903
|
+
hidden = max(0, len(ordered) - len(visible))
|
|
904
|
+
note = _analysis_note(analysis_rows, total_rows)
|
|
905
|
+
if hidden:
|
|
906
|
+
note += f" · top {_MAX_CATEGORY_BARS} of {len(ordered):,} values"
|
|
907
|
+
return {
|
|
908
|
+
"title": title,
|
|
909
|
+
"note": note,
|
|
910
|
+
"items": [{"label": label, "count": count} for label, count in visible],
|
|
911
|
+
}
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
def _column_summary(
|
|
915
|
+
values: list[Any],
|
|
916
|
+
logical: str,
|
|
917
|
+
unit: dict[str, Any],
|
|
918
|
+
*,
|
|
919
|
+
analysis_rows: int,
|
|
920
|
+
total_rows: int,
|
|
921
|
+
) -> tuple[dict[str, str], dict[str, Any]]:
|
|
922
|
+
non_null = [value for value in values if not _is_missing(value)]
|
|
923
|
+
if not non_null:
|
|
924
|
+
return (
|
|
925
|
+
{"label": "Observed", "value": "0", "detail": "No non-missing values"},
|
|
926
|
+
{"title": "Distribution", "note": "No non-missing values", "items": []},
|
|
927
|
+
)
|
|
928
|
+
symbol = unit.get("symbol") if isinstance(unit.get("symbol"), str) else None
|
|
929
|
+
suffix = f" {symbol}" if symbol else ""
|
|
930
|
+
if logical in {"int64", "float64"}:
|
|
931
|
+
numeric = [
|
|
932
|
+
value
|
|
933
|
+
for value in non_null
|
|
934
|
+
if isinstance(value, (int, float))
|
|
935
|
+
and not isinstance(value, bool)
|
|
936
|
+
and (isinstance(value, int) or math.isfinite(value))
|
|
937
|
+
]
|
|
938
|
+
if numeric:
|
|
939
|
+
if logical == "int64" and all(isinstance(value, int) for value in numeric):
|
|
940
|
+
ordered = sorted(numeric)
|
|
941
|
+
middle = len(ordered) // 2
|
|
942
|
+
if len(ordered) % 2:
|
|
943
|
+
median: int | float | Decimal = ordered[middle]
|
|
944
|
+
else:
|
|
945
|
+
total = ordered[middle - 1] + ordered[middle]
|
|
946
|
+
median = total // 2 if total % 2 == 0 else Decimal(total) / Decimal(2)
|
|
947
|
+
else:
|
|
948
|
+
ordered_floats = sorted(float(value) for value in numeric)
|
|
949
|
+
middle = len(ordered_floats) // 2
|
|
950
|
+
if len(ordered_floats) % 2:
|
|
951
|
+
median = ordered_floats[middle]
|
|
952
|
+
else:
|
|
953
|
+
low = ordered_floats[middle - 1]
|
|
954
|
+
high = ordered_floats[middle]
|
|
955
|
+
if low < 0.0 < high:
|
|
956
|
+
# Opposite signs make the sum safe even when the direct span overflows.
|
|
957
|
+
median = (low + high) / 2.0
|
|
958
|
+
else:
|
|
959
|
+
# Same-sign subtraction is finite; this also preserves repeated extrema.
|
|
960
|
+
median = low + (high - low) / 2.0
|
|
961
|
+
summary = {
|
|
962
|
+
"label": "Median",
|
|
963
|
+
"value": f"{_number(median)}{suffix}",
|
|
964
|
+
"detail": f"{_number(min(numeric))} → {_number(max(numeric))}{suffix}",
|
|
965
|
+
}
|
|
966
|
+
return summary, _histogram(
|
|
967
|
+
numeric,
|
|
968
|
+
unit_symbol=symbol,
|
|
969
|
+
analysis_rows=analysis_rows,
|
|
970
|
+
total_rows=total_rows,
|
|
971
|
+
)
|
|
972
|
+
if logical in {"date", "timestamp_utc"}:
|
|
973
|
+
rendered = sorted(_value_text(value) for value in non_null)
|
|
974
|
+
summary = {
|
|
975
|
+
"label": "Range",
|
|
976
|
+
"value": rendered[0],
|
|
977
|
+
"detail": f"through {rendered[-1]}",
|
|
978
|
+
}
|
|
979
|
+
return summary, _value_counts(
|
|
980
|
+
non_null,
|
|
981
|
+
title="Observations over time",
|
|
982
|
+
analysis_rows=analysis_rows,
|
|
983
|
+
total_rows=total_rows,
|
|
984
|
+
)
|
|
985
|
+
keyed_counts = Counter(_distinct_key(value) for value in non_null)
|
|
986
|
+
mode_key, count = sorted(
|
|
987
|
+
keyed_counts.items(), key=lambda item: (-item[1], _value_text(item[0][1]))
|
|
988
|
+
)[0]
|
|
989
|
+
mode = _value_text(mode_key[1])
|
|
990
|
+
summary = {
|
|
991
|
+
"label": "Most common",
|
|
992
|
+
"value": mode,
|
|
993
|
+
"detail": f"{count:,} of {len(non_null):,} observed",
|
|
994
|
+
}
|
|
995
|
+
title = "Boolean counts" if logical == "boolean" else "Value counts"
|
|
996
|
+
return summary, _value_counts(
|
|
997
|
+
non_null,
|
|
998
|
+
title=title,
|
|
999
|
+
analysis_rows=analysis_rows,
|
|
1000
|
+
total_rows=total_rows,
|
|
1001
|
+
)
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
def _semantics_context(
|
|
1005
|
+
candidate_dir: Path,
|
|
1006
|
+
members: set[str],
|
|
1007
|
+
column_names: list[str],
|
|
1008
|
+
*,
|
|
1009
|
+
member_bytes: dict[str, bytes] | None = None,
|
|
1010
|
+
verified_recipe: dict[str, Any] | None = None,
|
|
1011
|
+
) -> tuple[dict[str, Any], dict[str, dict[str, Any]]]:
|
|
1012
|
+
"""Derive semantic authority only from verified bindings or sealed local declarations."""
|
|
1013
|
+
|
|
1014
|
+
document: dict[str, Any] | None = None
|
|
1015
|
+
records: list[Any] = []
|
|
1016
|
+
if "recipe.json" in members:
|
|
1017
|
+
authority = "approved_recipe" if isinstance(verified_recipe, dict) else "none"
|
|
1018
|
+
raw_document = verified_recipe.get("table_semantics") if verified_recipe else None
|
|
1019
|
+
if isinstance(raw_document, dict):
|
|
1020
|
+
document = raw_document
|
|
1021
|
+
records = raw_document.get("columns", [])
|
|
1022
|
+
elif verified_recipe and isinstance(verified_recipe.get("output_semantics"), list):
|
|
1023
|
+
records = verified_recipe["output_semantics"]
|
|
1024
|
+
elif "semantics.json" in members:
|
|
1025
|
+
authority = "declared_local_plan"
|
|
1026
|
+
raw_document = _read_member_json(
|
|
1027
|
+
candidate_dir,
|
|
1028
|
+
"semantics.json",
|
|
1029
|
+
members,
|
|
1030
|
+
member_bytes=member_bytes,
|
|
1031
|
+
)
|
|
1032
|
+
if isinstance(raw_document, dict):
|
|
1033
|
+
document = raw_document
|
|
1034
|
+
records = raw_document.get("columns", [])
|
|
1035
|
+
else:
|
|
1036
|
+
authority = "none"
|
|
1037
|
+
|
|
1038
|
+
by_name = {
|
|
1039
|
+
str(item.get("name")): item
|
|
1040
|
+
for item in records
|
|
1041
|
+
if isinstance(item, dict) and isinstance(item.get("name"), str)
|
|
1042
|
+
}
|
|
1043
|
+
missing = sum(
|
|
1044
|
+
1
|
|
1045
|
+
for name in column_names
|
|
1046
|
+
if not isinstance(by_name.get(name, {}).get("display_label"), str)
|
|
1047
|
+
or not by_name[name].get("display_label")
|
|
1048
|
+
or not isinstance(by_name[name].get("description"), str)
|
|
1049
|
+
or not by_name[name].get("description")
|
|
1050
|
+
or _unit_metadata(by_name[name]).get("state") == "unknown"
|
|
1051
|
+
)
|
|
1052
|
+
complete = document is not None and missing == 0
|
|
1053
|
+
completeness = "absent" if authority == "none" else "complete" if complete else "partial"
|
|
1054
|
+
total = len(column_names)
|
|
1055
|
+
if completeness == "absent":
|
|
1056
|
+
message = (
|
|
1057
|
+
"This Build declares no column meanings. Nothing here is inferred from column names."
|
|
1058
|
+
)
|
|
1059
|
+
elif authority == "approved_recipe" and complete:
|
|
1060
|
+
message = "Column meanings are declared in an approved Recipe and sealed with this Build."
|
|
1061
|
+
elif authority == "approved_recipe":
|
|
1062
|
+
message = (
|
|
1063
|
+
"Some column meanings are declared in the approved Recipe. "
|
|
1064
|
+
f"{missing} of {total} columns have no declared meaning."
|
|
1065
|
+
)
|
|
1066
|
+
elif complete:
|
|
1067
|
+
message = (
|
|
1068
|
+
"Column meanings were declared with this local plan and sealed with this Build. "
|
|
1069
|
+
"They are not human-approved."
|
|
1070
|
+
)
|
|
1071
|
+
else:
|
|
1072
|
+
message = (
|
|
1073
|
+
"Some column meanings were declared with this local plan. "
|
|
1074
|
+
f"{missing} of {total} columns have no declared meaning. They are not human-approved."
|
|
1075
|
+
)
|
|
1076
|
+
context = {
|
|
1077
|
+
"schema_version": "mostlyright.column-semantics.v1",
|
|
1078
|
+
"authority": authority,
|
|
1079
|
+
"completeness": completeness,
|
|
1080
|
+
"message": message,
|
|
1081
|
+
"missing_column_count": missing if authority != "none" else total,
|
|
1082
|
+
"column_count": total,
|
|
1083
|
+
"render_scope": "full",
|
|
1084
|
+
}
|
|
1085
|
+
if document is not None:
|
|
1086
|
+
context.update(
|
|
1087
|
+
{
|
|
1088
|
+
"summary": document.get("summary"),
|
|
1089
|
+
"grain_statement": document.get("grain_statement"),
|
|
1090
|
+
"coverage": document.get("coverage"),
|
|
1091
|
+
"limitations": document.get("limitations"),
|
|
1092
|
+
}
|
|
1093
|
+
)
|
|
1094
|
+
return context, by_name
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
def _fit_column_profile_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
|
1098
|
+
"""Reduce optional semantic detail deterministically until the portable MIME is bounded."""
|
|
1099
|
+
|
|
1100
|
+
from mostlyright.data_harness.nbrender.outputs_data import is_valid_column_profiles_payload
|
|
1101
|
+
|
|
1102
|
+
if is_valid_column_profiles_payload(payload):
|
|
1103
|
+
return payload
|
|
1104
|
+
reduced = deepcopy(payload)
|
|
1105
|
+
semantics = reduced.get("semantics")
|
|
1106
|
+
if isinstance(semantics, dict):
|
|
1107
|
+
semantics["render_scope"] = "reduced"
|
|
1108
|
+
for key in ("coverage", "limitations"):
|
|
1109
|
+
semantics.pop(key, None)
|
|
1110
|
+
for column in reduced.get("columns", []):
|
|
1111
|
+
if isinstance(column, dict):
|
|
1112
|
+
for key in ("description", "categories", "evidence_refs"):
|
|
1113
|
+
column.pop(key, None)
|
|
1114
|
+
if is_valid_column_profiles_payload(reduced):
|
|
1115
|
+
return reduced
|
|
1116
|
+
minimal = deepcopy(reduced)
|
|
1117
|
+
semantics = minimal.get("semantics")
|
|
1118
|
+
if isinstance(semantics, dict):
|
|
1119
|
+
semantics["render_scope"] = "minimal"
|
|
1120
|
+
for key in ("summary", "grain_statement"):
|
|
1121
|
+
semantics.pop(key, None)
|
|
1122
|
+
for column in minimal.get("columns", []):
|
|
1123
|
+
if isinstance(column, dict):
|
|
1124
|
+
for key in ("display_label", "categories_status"):
|
|
1125
|
+
column.pop(key, None)
|
|
1126
|
+
if is_valid_column_profiles_payload(minimal):
|
|
1127
|
+
return minimal
|
|
1128
|
+
legacy = deepcopy(minimal)
|
|
1129
|
+
for column in legacy.get("columns", []):
|
|
1130
|
+
if isinstance(column, dict):
|
|
1131
|
+
column["unit"] = {
|
|
1132
|
+
key: column.get("unit", {}).get(key)
|
|
1133
|
+
for key in ("value", "symbol", "label", "status")
|
|
1134
|
+
}
|
|
1135
|
+
if is_valid_column_profiles_payload(legacy):
|
|
1136
|
+
return legacy
|
|
1137
|
+
|
|
1138
|
+
# A verified string column may legitimately contain a very large value. Summary and chart
|
|
1139
|
+
# labels derived from ten such columns can exceed the portable 1 MiB rich-MIME ceiling even
|
|
1140
|
+
# after semantic detail is reduced. Keep the analytical shape and semantic authority, but
|
|
1141
|
+
# deterministically bound presentation strings before considering refusal. Complete values
|
|
1142
|
+
# never belonged in this MIME contract; the portable fallbacks describe the dataset rather
|
|
1143
|
+
# than repeating profiled cell contents.
|
|
1144
|
+
bounded_rich = deepcopy(legacy)
|
|
1145
|
+
for column in bounded_rich.get("columns", []):
|
|
1146
|
+
if not isinstance(column, dict):
|
|
1147
|
+
continue
|
|
1148
|
+
for field in ("summary", "source"):
|
|
1149
|
+
value = column.get(field)
|
|
1150
|
+
if isinstance(value, dict):
|
|
1151
|
+
for key, item in tuple(value.items()):
|
|
1152
|
+
if isinstance(item, str):
|
|
1153
|
+
value[key] = _bounded_rich_label(item)
|
|
1154
|
+
chart = column.get("chart")
|
|
1155
|
+
if isinstance(chart, dict):
|
|
1156
|
+
for key in ("title", "note"):
|
|
1157
|
+
if isinstance(chart.get(key), str):
|
|
1158
|
+
chart[key] = _bounded_rich_label(chart[key])
|
|
1159
|
+
items = chart.get("items")
|
|
1160
|
+
if isinstance(items, list):
|
|
1161
|
+
for item in items:
|
|
1162
|
+
if isinstance(item, dict) and isinstance(item.get("label"), str):
|
|
1163
|
+
item["label"] = _bounded_rich_label(item["label"])
|
|
1164
|
+
unit = column.get("unit")
|
|
1165
|
+
if isinstance(unit, dict):
|
|
1166
|
+
for key, item in tuple(unit.items()):
|
|
1167
|
+
if isinstance(item, str):
|
|
1168
|
+
unit[key] = _bounded_rich_label(item)
|
|
1169
|
+
operations = column.get("transformations")
|
|
1170
|
+
if isinstance(operations, list):
|
|
1171
|
+
column["transformations"] = [
|
|
1172
|
+
_bounded_rich_label(item) if isinstance(item, str) else item for item in operations
|
|
1173
|
+
]
|
|
1174
|
+
if not is_valid_column_profiles_payload(bounded_rich):
|
|
1175
|
+
raise ValueError("generated column-profile MIME exceeds its bounded contract")
|
|
1176
|
+
return bounded_rich
|
|
1177
|
+
|
|
1178
|
+
|
|
1179
|
+
def _bounded_rich_label(value: str, *, limit: int = 128) -> str:
|
|
1180
|
+
"""Bound one derived rich-MIME label with an explicit deterministic marker."""
|
|
1181
|
+
|
|
1182
|
+
if len(value) <= limit:
|
|
1183
|
+
return value
|
|
1184
|
+
marker = " [truncated]"
|
|
1185
|
+
return value[: limit - len(marker)] + marker
|
|
1186
|
+
|
|
1187
|
+
|
|
1188
|
+
def _column_profiles(
|
|
1189
|
+
candidate_dir: Path,
|
|
1190
|
+
table: Any,
|
|
1191
|
+
inspection: Any,
|
|
1192
|
+
profile: dict[str, Any],
|
|
1193
|
+
members: set[str],
|
|
1194
|
+
*,
|
|
1195
|
+
member_bytes: dict[str, bytes] | None = None,
|
|
1196
|
+
verified_recipe: dict[str, Any] | None = None,
|
|
1197
|
+
fit_payload: bool = True,
|
|
1198
|
+
) -> dict[str, Any]:
|
|
1199
|
+
"""Compute bounded visual analytics over the verified Parquet and bind declared units."""
|
|
1200
|
+
|
|
1201
|
+
semantics_context, semantics_by_name = _semantics_context(
|
|
1202
|
+
candidate_dir,
|
|
1203
|
+
members,
|
|
1204
|
+
list(table.column_names),
|
|
1205
|
+
member_bytes=member_bytes,
|
|
1206
|
+
verified_recipe=verified_recipe,
|
|
1207
|
+
)
|
|
1208
|
+
logical_by_name = {
|
|
1209
|
+
str(item.get("name")): str(item.get("logical_type", "unknown"))
|
|
1210
|
+
for item in profile.get("logical_schema", [])
|
|
1211
|
+
if isinstance(item, dict)
|
|
1212
|
+
}
|
|
1213
|
+
type_by_name = (
|
|
1214
|
+
{str(name): str(value) for name, value in profile.get("types", {}).items()}
|
|
1215
|
+
if isinstance(profile.get("types"), dict)
|
|
1216
|
+
else {}
|
|
1217
|
+
)
|
|
1218
|
+
nulls_by_name = (
|
|
1219
|
+
profile.get("null_counts", {}) if isinstance(profile.get("null_counts"), dict) else {}
|
|
1220
|
+
)
|
|
1221
|
+
rows = int(profile.get("row_count", table.num_rows) or 0)
|
|
1222
|
+
columns: list[dict[str, Any]] = []
|
|
1223
|
+
for name in table.column_names:
|
|
1224
|
+
semantics = semantics_by_name.get(name)
|
|
1225
|
+
logical = str(
|
|
1226
|
+
logical_by_name.get(name)
|
|
1227
|
+
or (semantics or {}).get("logical_type")
|
|
1228
|
+
or type_by_name.get(name)
|
|
1229
|
+
or "unknown"
|
|
1230
|
+
)
|
|
1231
|
+
physical = PHYSICAL_TYPES.get(logical, str(table.schema.field(name).type))
|
|
1232
|
+
unit = _unit_metadata(semantics)
|
|
1233
|
+
array = table[name]
|
|
1234
|
+
sample_size = min(table.num_rows, _PROFILE_SAMPLE_ROWS)
|
|
1235
|
+
sample = array.slice(0, sample_size).to_pylist()
|
|
1236
|
+
null_count = int(nulls_by_name.get(name, array.null_count) or 0)
|
|
1237
|
+
nan_count = 0
|
|
1238
|
+
if logical == "float64":
|
|
1239
|
+
import pyarrow.compute as pc
|
|
1240
|
+
|
|
1241
|
+
# Arrow's sealed profile counts nulls. The notebook's visual missingness policy also
|
|
1242
|
+
# treats every NaN payload as missing, so calculate that second exact count over the
|
|
1243
|
+
# same verified Arrow snapshot and state both parts in the payload.
|
|
1244
|
+
nan_count = int(pc.sum(pc.is_nan(array)).as_py() or 0)
|
|
1245
|
+
missing = null_count + nan_count
|
|
1246
|
+
non_missing_sample = [value for value in sample if not _is_missing(value)]
|
|
1247
|
+
distinct_sample = len({_distinct_key(value) for value in non_missing_sample})
|
|
1248
|
+
# Exact for small/medium columns; high-cardinality columns are clearly labelled sampled.
|
|
1249
|
+
if table.num_rows <= _PROFILE_SAMPLE_ROWS:
|
|
1250
|
+
distinct = distinct_sample
|
|
1251
|
+
distinct_status = "exact"
|
|
1252
|
+
else:
|
|
1253
|
+
distinct = distinct_sample
|
|
1254
|
+
distinct_status = f"sampled from first {sample_size:,} rows"
|
|
1255
|
+
summary, chart = _column_summary(
|
|
1256
|
+
sample,
|
|
1257
|
+
logical,
|
|
1258
|
+
unit,
|
|
1259
|
+
analysis_rows=sample_size,
|
|
1260
|
+
total_rows=table.num_rows,
|
|
1261
|
+
)
|
|
1262
|
+
detail = inspection.lineage.get(name, {}) if isinstance(inspection.lineage, dict) else {}
|
|
1263
|
+
detail = detail if isinstance(detail, dict) else {}
|
|
1264
|
+
column = {
|
|
1265
|
+
"name": name,
|
|
1266
|
+
"logical_type": logical,
|
|
1267
|
+
"physical_type": physical,
|
|
1268
|
+
"semantic_type": (semantics or {}).get("semantic_type"),
|
|
1269
|
+
"unit": unit,
|
|
1270
|
+
"row_count": rows,
|
|
1271
|
+
"null_count": null_count,
|
|
1272
|
+
"nan_count": nan_count,
|
|
1273
|
+
"missing_count": missing,
|
|
1274
|
+
"missing_rate": (missing / rows) if rows else 0.0,
|
|
1275
|
+
"missing_policy": "null_or_nan",
|
|
1276
|
+
"distinct_count": distinct,
|
|
1277
|
+
"distinct_rate": (
|
|
1278
|
+
distinct / max(1, rows if distinct_status == "exact" else sample_size)
|
|
1279
|
+
),
|
|
1280
|
+
"distinct_status": distinct_status,
|
|
1281
|
+
"distinct_policy": "typed_exact; null_and_nan_excluded; signed_zero_collapsed",
|
|
1282
|
+
"sample_size": sample_size,
|
|
1283
|
+
"summary": summary,
|
|
1284
|
+
"chart": chart,
|
|
1285
|
+
"source": {
|
|
1286
|
+
"source_id": detail.get("source_id"),
|
|
1287
|
+
"column": detail.get("source_column", name),
|
|
1288
|
+
},
|
|
1289
|
+
"transformations": list(detail.get("operations", []))
|
|
1290
|
+
if isinstance(detail.get("operations"), list)
|
|
1291
|
+
else [],
|
|
1292
|
+
}
|
|
1293
|
+
if semantics is not None:
|
|
1294
|
+
column.update(
|
|
1295
|
+
{
|
|
1296
|
+
"display_label": semantics.get("display_label"),
|
|
1297
|
+
"description": semantics.get("description"),
|
|
1298
|
+
"categories": semantics.get("categories"),
|
|
1299
|
+
"categories_status": semantics.get("categories_status"),
|
|
1300
|
+
"evidence_refs": semantics.get("evidence_refs"),
|
|
1301
|
+
}
|
|
1302
|
+
)
|
|
1303
|
+
columns.append(column)
|
|
1304
|
+
payload = {
|
|
1305
|
+
"schema_version": "mostlyright.column-profile.v1",
|
|
1306
|
+
"table_digest": inspection.table_sha256,
|
|
1307
|
+
"row_count": rows,
|
|
1308
|
+
"column_count": len(columns),
|
|
1309
|
+
"analysis_scope": "all rows"
|
|
1310
|
+
if table.num_rows <= _PROFILE_SAMPLE_ROWS
|
|
1311
|
+
else f"first {_PROFILE_SAMPLE_ROWS:,} rows for charts",
|
|
1312
|
+
"columns": columns,
|
|
1313
|
+
"semantics": semantics_context,
|
|
1314
|
+
}
|
|
1315
|
+
return _fit_column_profile_payload(payload) if fit_payload else payload
|
|
1316
|
+
|
|
1317
|
+
|
|
1318
|
+
# --- reading the sealed candidate --------------------------------------------------------------
|
|
1319
|
+
|
|
1320
|
+
|
|
1321
|
+
def _read_member_json(
|
|
1322
|
+
candidate_dir: Path,
|
|
1323
|
+
relative: str,
|
|
1324
|
+
members: set[str],
|
|
1325
|
+
*,
|
|
1326
|
+
member_bytes: dict[str, bytes] | None = None,
|
|
1327
|
+
) -> Any | None:
|
|
1328
|
+
"""Parse ``relative`` from ``candidate_dir`` only when it is a recorded, present member."""
|
|
1329
|
+
|
|
1330
|
+
if relative not in members:
|
|
1331
|
+
return None
|
|
1332
|
+
try:
|
|
1333
|
+
if member_bytes is not None:
|
|
1334
|
+
raw = member_bytes.get(relative)
|
|
1335
|
+
return json.loads(raw.decode("utf-8")) if raw is not None else None
|
|
1336
|
+
path = candidate_dir / relative
|
|
1337
|
+
if not path.exists():
|
|
1338
|
+
return None
|
|
1339
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
1340
|
+
except (OSError, ValueError):
|
|
1341
|
+
return None
|
|
1342
|
+
|
|
1343
|
+
|
|
1344
|
+
def _read_dataset(
|
|
1345
|
+
candidate_dir: Path,
|
|
1346
|
+
limit: int,
|
|
1347
|
+
*,
|
|
1348
|
+
member_bytes: dict[str, bytes] | None = None,
|
|
1349
|
+
) -> tuple[Any, list[dict[str, Any]], list[str]]:
|
|
1350
|
+
"""Read the verified Parquet once, returning the table and a native head sample."""
|
|
1351
|
+
|
|
1352
|
+
import pyarrow.parquet as pq # lazy — the already-pinned base dependency
|
|
1353
|
+
|
|
1354
|
+
if member_bytes is not None:
|
|
1355
|
+
import pyarrow as pa
|
|
1356
|
+
|
|
1357
|
+
table = pq.read_table(pa.BufferReader(member_bytes["data/table.parquet"]))
|
|
1358
|
+
else:
|
|
1359
|
+
table = pq.read_table(candidate_dir / "data" / "table.parquet")
|
|
1360
|
+
sample = table.slice(0, limit).to_pylist()
|
|
1361
|
+
return table, sample, list(table.column_names)
|
|
1362
|
+
|
|
1363
|
+
|
|
1364
|
+
# --- notebook assembly -------------------------------------------------------------------------
|
|
1365
|
+
|
|
1366
|
+
|
|
1367
|
+
def build_notebook(candidate_dir: Path) -> dict[str, Any]:
|
|
1368
|
+
"""Render a sealed candidate from one retained replay-verified snapshot."""
|
|
1369
|
+
|
|
1370
|
+
candidate_dir = Path(candidate_dir)
|
|
1371
|
+
with pipeline.open_verified_snapshot(candidate_dir.parent) as verified_snapshot:
|
|
1372
|
+
return _build_notebook_from_snapshot(candidate_dir, verified_snapshot)
|
|
1373
|
+
|
|
1374
|
+
|
|
1375
|
+
def _build_notebook_from_snapshot(
|
|
1376
|
+
candidate_dir: Path,
|
|
1377
|
+
verified_snapshot: pipeline.VerifiedSnapshot,
|
|
1378
|
+
) -> dict[str, Any]:
|
|
1379
|
+
"""Derive notebook bytes without reopening any candidate member pathname."""
|
|
1380
|
+
|
|
1381
|
+
member_bytes = dict(verified_snapshot.members)
|
|
1382
|
+
inspection = verified_snapshot.inspection()
|
|
1383
|
+
members = set(inspection.member_paths)
|
|
1384
|
+
profile = (
|
|
1385
|
+
_read_member_json(
|
|
1386
|
+
candidate_dir,
|
|
1387
|
+
"evidence/profile.json",
|
|
1388
|
+
members,
|
|
1389
|
+
member_bytes=member_bytes,
|
|
1390
|
+
)
|
|
1391
|
+
or {}
|
|
1392
|
+
)
|
|
1393
|
+
table = verified_snapshot.parquet_file().read()
|
|
1394
|
+
sample_rows = table.slice(0, _SAMPLE_ROWS).to_pylist()
|
|
1395
|
+
sample_columns = list(table.column_names)
|
|
1396
|
+
verified_recipe: dict[str, Any] | None = None
|
|
1397
|
+
if "recipe.json" in members:
|
|
1398
|
+
try:
|
|
1399
|
+
bindings = recipe_contracts.verify_recipe_authority_snapshot(verified_snapshot)
|
|
1400
|
+
except recipe_contracts.RecipeError:
|
|
1401
|
+
# Generic Build verification proves bytes, not human approval. A malformed or
|
|
1402
|
+
# unbound Recipe candidate remains inspectable, but it must not gain approved copy.
|
|
1403
|
+
verified_recipe = None
|
|
1404
|
+
else:
|
|
1405
|
+
verified_recipe = bindings["recipe"].to_dict()
|
|
1406
|
+
|
|
1407
|
+
full_column_profiles = _column_profiles(
|
|
1408
|
+
candidate_dir,
|
|
1409
|
+
table,
|
|
1410
|
+
inspection,
|
|
1411
|
+
profile,
|
|
1412
|
+
members,
|
|
1413
|
+
member_bytes=member_bytes,
|
|
1414
|
+
verified_recipe=verified_recipe,
|
|
1415
|
+
fit_payload=False,
|
|
1416
|
+
)
|
|
1417
|
+
column_profiles = _fit_column_profile_payload(full_column_profiles)
|
|
1418
|
+
semantics_context, semantics_by_name = _semantics_context(
|
|
1419
|
+
candidate_dir,
|
|
1420
|
+
members,
|
|
1421
|
+
sample_columns,
|
|
1422
|
+
member_bytes=member_bytes,
|
|
1423
|
+
verified_recipe=verified_recipe,
|
|
1424
|
+
)
|
|
1425
|
+
logical_by_name = {
|
|
1426
|
+
str(item.get("name")): str(item.get("logical_type", "unknown"))
|
|
1427
|
+
for item in profile.get("logical_schema", [])
|
|
1428
|
+
if isinstance(item, dict)
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
cells: list[dict[str, Any]] = [
|
|
1432
|
+
_markdown_cell(
|
|
1433
|
+
_header_markdown(
|
|
1434
|
+
inspection.question,
|
|
1435
|
+
inspection.row_count,
|
|
1436
|
+
len(inspection.columns),
|
|
1437
|
+
)
|
|
1438
|
+
),
|
|
1439
|
+
_markdown_cell(_overview_markdown(semantics_context)),
|
|
1440
|
+
_markdown_cell(
|
|
1441
|
+
"## Data\n\n### Preview\n\nThe first rows are embedded so the result is readable "
|
|
1442
|
+
"without starting Python. Run the cell to load the complete Parquet file."
|
|
1443
|
+
),
|
|
1444
|
+
_code_cell(
|
|
1445
|
+
"import pandas as pd\n"
|
|
1446
|
+
# The sidecar is a run-dir SIBLING of candidate/, so every relative path a runnable
|
|
1447
|
+
# cell uses must go through candidate/ to resolve from the notebook's own directory.
|
|
1448
|
+
"df = pd.read_parquet('candidate/data/table.parquet')\n"
|
|
1449
|
+
f"df.head({_SAMPLE_ROWS}) # first rows",
|
|
1450
|
+
(_html_table(sample_rows, sample_columns),),
|
|
1451
|
+
),
|
|
1452
|
+
_markdown_cell(_dictionary_markdown(sample_columns, semantics_by_name, logical_by_name)),
|
|
1453
|
+
_markdown_cell(
|
|
1454
|
+
"### Column profiles\n\nSelect a column to inspect its type, unit or scaling state, "
|
|
1455
|
+
"missingness, "
|
|
1456
|
+
"cardinality, distribution, source, and transformations. Charts are computed from "
|
|
1457
|
+
"the verified Parquet; semantic claims come only from sealed Build members."
|
|
1458
|
+
),
|
|
1459
|
+
_code_cell(
|
|
1460
|
+
"# visual profile for every column\ndf.describe(include='all')",
|
|
1461
|
+
(
|
|
1462
|
+
_display_column_profiles(
|
|
1463
|
+
column_profiles,
|
|
1464
|
+
fallback_payload=full_column_profiles,
|
|
1465
|
+
),
|
|
1466
|
+
),
|
|
1467
|
+
),
|
|
1468
|
+
_markdown_cell(_quality_markdown(inspection.quality)),
|
|
1469
|
+
]
|
|
1470
|
+
|
|
1471
|
+
if isinstance(inspection.join, dict) and inspection.join:
|
|
1472
|
+
if inspection.join.get("schema_version") in {
|
|
1473
|
+
pipeline.GRAPH_OPERATION_EVIDENCE_V1,
|
|
1474
|
+
pipeline.GRAPH_OPERATION_EVIDENCE_VERSION,
|
|
1475
|
+
}:
|
|
1476
|
+
cells.append(_markdown_cell(_operations_markdown(inspection.join)))
|
|
1477
|
+
else:
|
|
1478
|
+
cells.append(_markdown_cell(_join_markdown(inspection.join)))
|
|
1479
|
+
|
|
1480
|
+
cells.append(_markdown_cell(_sources_markdown(inspection.sources)))
|
|
1481
|
+
|
|
1482
|
+
cells.extend(
|
|
1483
|
+
[
|
|
1484
|
+
_markdown_cell(
|
|
1485
|
+
"## Analysis\n\nThe following cells are optional local queries. They can be "
|
|
1486
|
+
"edited without changing the packaged Parquet data."
|
|
1487
|
+
),
|
|
1488
|
+
_code_cell("df.describe(include='all')"),
|
|
1489
|
+
_code_cell("df.dtypes"),
|
|
1490
|
+
_code_cell("df[df.columns[0]].value_counts().head(15)"),
|
|
1491
|
+
]
|
|
1492
|
+
)
|
|
1493
|
+
|
|
1494
|
+
notebook = {
|
|
1495
|
+
"cells": cells,
|
|
1496
|
+
"metadata": {
|
|
1497
|
+
"mostlyright": {
|
|
1498
|
+
"provenance": _provenance_metadata(inspection),
|
|
1499
|
+
"column_profile_mime": _COLUMN_PROFILE_MIME,
|
|
1500
|
+
}
|
|
1501
|
+
},
|
|
1502
|
+
"nbformat": _NBFORMAT_MAJOR,
|
|
1503
|
+
"nbformat_minor": _NBFORMAT_MINOR,
|
|
1504
|
+
}
|
|
1505
|
+
verified_snapshot.validate()
|
|
1506
|
+
return notebook
|
|
1507
|
+
|
|
1508
|
+
|
|
1509
|
+
def render_table_notebook(
|
|
1510
|
+
run_dir: Path,
|
|
1511
|
+
*,
|
|
1512
|
+
replace_existing: bool = True,
|
|
1513
|
+
_run_fd: int | None = None,
|
|
1514
|
+
_ancestor_validator: Any | None = None,
|
|
1515
|
+
) -> Path:
|
|
1516
|
+
"""Build the notebook for the candidate under ``run_dir`` and write ``table.ipynb`` beside it.
|
|
1517
|
+
|
|
1518
|
+
Resolves the candidate root via ``_candidate_run_target`` (which follows the workspace
|
|
1519
|
+
``result/`` indirection) and writes ``table.ipynb`` (0o644) as a SIBLING of ``candidate/`` —
|
|
1520
|
+
never inside it. The sidecar is atomically replaced through the verified Build directory's
|
|
1521
|
+
retained descriptor, without following an existing output link or reopening its path.
|
|
1522
|
+
|
|
1523
|
+
``replace_existing`` separates the command a person ran from the viewer's unattended recovery.
|
|
1524
|
+
``mr-data notebook`` was asked for a fresh notebook and replaces whatever is there. The watcher
|
|
1525
|
+
may only fill an absence, and it checks that absence long before this write: pass ``False`` and
|
|
1526
|
+
a sidecar that appeared in between raises :class:`FileExistsError` instead of being replaced.
|
|
1527
|
+
"""
|
|
1528
|
+
|
|
1529
|
+
# Imported inside the function to break an import cycle: ``cli`` imports this module at module
|
|
1530
|
+
# level for the notebook sidecar, so this module cannot import ``cli`` at module level.
|
|
1531
|
+
from mostlyright.data_harness.cli import _candidate_run_target
|
|
1532
|
+
|
|
1533
|
+
run_dir = Path(run_dir).absolute()
|
|
1534
|
+
target = run_dir if _run_fd is not None else _candidate_run_target(run_dir)
|
|
1535
|
+
handle = None if _run_fd is not None else pipeline._open_candidate_run_handle(target)
|
|
1536
|
+
active_run_fd = _run_fd if _run_fd is not None else handle.run_fd
|
|
1537
|
+
sidecar_lock_fd = -1
|
|
1538
|
+
sidecar_lock_identity: tuple[int, int, int] | None = None
|
|
1539
|
+
|
|
1540
|
+
def validate() -> None:
|
|
1541
|
+
if handle is not None:
|
|
1542
|
+
pipeline._validate_candidate_run_handle(handle)
|
|
1543
|
+
if sidecar_lock_fd >= 0 and sidecar_lock_identity is not None:
|
|
1544
|
+
_validate_table_sidecar_lock(
|
|
1545
|
+
active_run_fd,
|
|
1546
|
+
sidecar_lock_fd,
|
|
1547
|
+
sidecar_lock_identity,
|
|
1548
|
+
)
|
|
1549
|
+
if _ancestor_validator is not None:
|
|
1550
|
+
_ancestor_validator()
|
|
1551
|
+
|
|
1552
|
+
try:
|
|
1553
|
+
if handle is not None:
|
|
1554
|
+
sidecar_lock_fd, sidecar_lock_identity = _acquire_table_sidecar_lock(active_run_fd)
|
|
1555
|
+
with pipeline.open_verified_snapshot(
|
|
1556
|
+
target,
|
|
1557
|
+
_run_fd=active_run_fd,
|
|
1558
|
+
_ancestor_validator=validate,
|
|
1559
|
+
) as verified_snapshot:
|
|
1560
|
+
notebook = _build_notebook_from_snapshot(target / "candidate", verified_snapshot)
|
|
1561
|
+
validate()
|
|
1562
|
+
raw = (json.dumps(notebook, indent=1, ensure_ascii=False) + "\n").encode("utf-8")
|
|
1563
|
+
_atomic_write_table_notebook(active_run_fd, raw, replace_existing=replace_existing)
|
|
1564
|
+
validate()
|
|
1565
|
+
finally:
|
|
1566
|
+
try:
|
|
1567
|
+
if sidecar_lock_fd >= 0:
|
|
1568
|
+
if fcntl is not None:
|
|
1569
|
+
fcntl.flock(sidecar_lock_fd, fcntl.LOCK_UN)
|
|
1570
|
+
os.close(sidecar_lock_fd)
|
|
1571
|
+
finally:
|
|
1572
|
+
if handle is not None:
|
|
1573
|
+
handle.close()
|
|
1574
|
+
return target / TABLE_NOTEBOOK_NAME
|
|
1575
|
+
|
|
1576
|
+
|
|
1577
|
+
def _acquire_table_sidecar_lock(run_fd: int) -> tuple[int, tuple[int, int, int]]:
|
|
1578
|
+
"""Exclusively serialize every sidecar writer on the Build's structural activity file."""
|
|
1579
|
+
|
|
1580
|
+
if fcntl is None:
|
|
1581
|
+
raise pipeline.BuildError(
|
|
1582
|
+
"CANDIDATE_PLATFORM_UNSUPPORTED",
|
|
1583
|
+
"secure table notebook locking is unavailable on this platform",
|
|
1584
|
+
)
|
|
1585
|
+
lock_fd = -1
|
|
1586
|
+
try:
|
|
1587
|
+
named = os.stat(
|
|
1588
|
+
pipeline.RUN_BUILD_ACTIVITY_LOCK,
|
|
1589
|
+
dir_fd=run_fd,
|
|
1590
|
+
follow_symlinks=False,
|
|
1591
|
+
)
|
|
1592
|
+
if not stat.S_ISREG(named.st_mode) or named.st_nlink != 1:
|
|
1593
|
+
raise pipeline.BuildError(
|
|
1594
|
+
"CANDIDATE_MEMBER_INVALID",
|
|
1595
|
+
"run build activity lock is not a stable regular file",
|
|
1596
|
+
)
|
|
1597
|
+
lock_fd = os.open(
|
|
1598
|
+
pipeline.RUN_BUILD_ACTIVITY_LOCK,
|
|
1599
|
+
os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0),
|
|
1600
|
+
dir_fd=run_fd,
|
|
1601
|
+
)
|
|
1602
|
+
opened = os.fstat(lock_fd)
|
|
1603
|
+
identity = pipeline._entry_identity(opened)
|
|
1604
|
+
if (
|
|
1605
|
+
not stat.S_ISREG(opened.st_mode)
|
|
1606
|
+
or opened.st_nlink != 1
|
|
1607
|
+
or identity != pipeline._entry_identity(named)
|
|
1608
|
+
):
|
|
1609
|
+
raise pipeline.BuildError(
|
|
1610
|
+
"CANDIDATE_MEMBER_INVALID",
|
|
1611
|
+
"run build activity lock changed identity",
|
|
1612
|
+
)
|
|
1613
|
+
fcntl.flock(lock_fd, fcntl.LOCK_EX)
|
|
1614
|
+
_validate_table_sidecar_lock(run_fd, lock_fd, identity)
|
|
1615
|
+
return lock_fd, identity
|
|
1616
|
+
except BaseException:
|
|
1617
|
+
if lock_fd >= 0:
|
|
1618
|
+
os.close(lock_fd)
|
|
1619
|
+
raise
|
|
1620
|
+
|
|
1621
|
+
|
|
1622
|
+
def _validate_table_sidecar_lock(
|
|
1623
|
+
run_fd: int,
|
|
1624
|
+
lock_fd: int,
|
|
1625
|
+
identity: tuple[int, int, int],
|
|
1626
|
+
) -> None:
|
|
1627
|
+
"""Require the held activity lock to remain the exact named run child."""
|
|
1628
|
+
|
|
1629
|
+
opened = os.fstat(lock_fd)
|
|
1630
|
+
named = os.stat(
|
|
1631
|
+
pipeline.RUN_BUILD_ACTIVITY_LOCK,
|
|
1632
|
+
dir_fd=run_fd,
|
|
1633
|
+
follow_symlinks=False,
|
|
1634
|
+
)
|
|
1635
|
+
if (
|
|
1636
|
+
not stat.S_ISREG(opened.st_mode)
|
|
1637
|
+
or opened.st_nlink != 1
|
|
1638
|
+
or not stat.S_ISREG(named.st_mode)
|
|
1639
|
+
or named.st_nlink != 1
|
|
1640
|
+
or pipeline._entry_identity(opened) != identity
|
|
1641
|
+
or pipeline._entry_identity(named) != identity
|
|
1642
|
+
):
|
|
1643
|
+
raise pipeline.BuildError(
|
|
1644
|
+
"CANDIDATE_MEMBER_INVALID",
|
|
1645
|
+
"run build activity lock changed identity",
|
|
1646
|
+
)
|
|
1647
|
+
|
|
1648
|
+
|
|
1649
|
+
def _atomic_write_table_notebook(run_fd: int, raw: bytes, *, replace_existing: bool = True) -> None:
|
|
1650
|
+
"""Install the unsealed sidecar inside one descriptor-pinned Build directory.
|
|
1651
|
+
|
|
1652
|
+
Linking a complete temporary file is an atomic no-replace installation, so a caller that may
|
|
1653
|
+
only fill an absence never overwrites a sidecar that arrived while it was rendering.
|
|
1654
|
+
"""
|
|
1655
|
+
|
|
1656
|
+
temporary = ""
|
|
1657
|
+
descriptor = -1
|
|
1658
|
+
try:
|
|
1659
|
+
for _attempt in range(100):
|
|
1660
|
+
temporary = f".{TABLE_NOTEBOOK_NAME}.{secrets.token_hex(12)}"
|
|
1661
|
+
try:
|
|
1662
|
+
descriptor = os.open(
|
|
1663
|
+
temporary,
|
|
1664
|
+
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_CLOEXEC | os.O_NOFOLLOW,
|
|
1665
|
+
0o600,
|
|
1666
|
+
dir_fd=run_fd,
|
|
1667
|
+
)
|
|
1668
|
+
break
|
|
1669
|
+
except FileExistsError:
|
|
1670
|
+
continue
|
|
1671
|
+
else:
|
|
1672
|
+
raise FileExistsError("could not allocate a table notebook temporary file")
|
|
1673
|
+
|
|
1674
|
+
offset = 0
|
|
1675
|
+
while offset < len(raw):
|
|
1676
|
+
written = os.write(descriptor, raw[offset:])
|
|
1677
|
+
if written < 1:
|
|
1678
|
+
raise OSError("table notebook write did not make progress")
|
|
1679
|
+
offset += written
|
|
1680
|
+
os.fsync(descriptor)
|
|
1681
|
+
os.fchmod(descriptor, 0o644)
|
|
1682
|
+
os.fsync(descriptor)
|
|
1683
|
+
os.close(descriptor)
|
|
1684
|
+
descriptor = -1
|
|
1685
|
+
if replace_existing:
|
|
1686
|
+
os.replace(
|
|
1687
|
+
temporary,
|
|
1688
|
+
TABLE_NOTEBOOK_NAME,
|
|
1689
|
+
src_dir_fd=run_fd,
|
|
1690
|
+
dst_dir_fd=run_fd,
|
|
1691
|
+
)
|
|
1692
|
+
else:
|
|
1693
|
+
os.link(
|
|
1694
|
+
temporary,
|
|
1695
|
+
TABLE_NOTEBOOK_NAME,
|
|
1696
|
+
src_dir_fd=run_fd,
|
|
1697
|
+
dst_dir_fd=run_fd,
|
|
1698
|
+
follow_symlinks=False,
|
|
1699
|
+
)
|
|
1700
|
+
os.unlink(temporary, dir_fd=run_fd)
|
|
1701
|
+
temporary = ""
|
|
1702
|
+
os.fsync(run_fd)
|
|
1703
|
+
finally:
|
|
1704
|
+
if descriptor >= 0:
|
|
1705
|
+
os.close(descriptor)
|
|
1706
|
+
if temporary:
|
|
1707
|
+
try:
|
|
1708
|
+
os.unlink(temporary, dir_fd=run_fd)
|
|
1709
|
+
except FileNotFoundError:
|
|
1710
|
+
pass
|