mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,485 @@
|
|
|
1
|
+
"""Declaratively project bounded JSON or NDJSON documents into canonical tabular CSV.
|
|
2
|
+
|
|
3
|
+
The Reader deliberately implements selection rather than inference. A recipe identifies the
|
|
4
|
+
record collection, any arrays to expand, and every output column with RFC 6901 JSON Pointers.
|
|
5
|
+
The same implementation therefore handles a top-level array, an API response envelope, and
|
|
6
|
+
nested observations without source-specific code or executable expressions.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import math
|
|
13
|
+
import re
|
|
14
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from mostlyright.data_harness.formats import (
|
|
19
|
+
FORMAT_MEDIA_TYPES,
|
|
20
|
+
FORMAT_SUFFIXES,
|
|
21
|
+
READER_CONTRACT_VERSION,
|
|
22
|
+
READER_OUTPUT_FORMATS,
|
|
23
|
+
)
|
|
24
|
+
from mostlyright.data_harness.readers.contracts import (
|
|
25
|
+
ReaderBudgets,
|
|
26
|
+
ReaderError,
|
|
27
|
+
ReaderPin,
|
|
28
|
+
ReaderResult,
|
|
29
|
+
bulk_default_budgets,
|
|
30
|
+
)
|
|
31
|
+
from mostlyright.data_harness.readers.tabular import (
|
|
32
|
+
FALLBACK_STEM,
|
|
33
|
+
encode_canonical_csv,
|
|
34
|
+
sealed_filename,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__all__ = ["JsonTabularReader"]
|
|
38
|
+
|
|
39
|
+
(_OUTPUT_FORMAT,) = READER_OUTPUT_FORMATS
|
|
40
|
+
_OUTPUT_MEDIA_TYPE = sorted(FORMAT_MEDIA_TYPES[_OUTPUT_FORMAT])[0]
|
|
41
|
+
_OUTPUT_SUFFIX = sorted(FORMAT_SUFFIXES[_OUTPUT_FORMAT])[0]
|
|
42
|
+
|
|
43
|
+
_DOCUMENT_FORMATS = ("json", "ndjson")
|
|
44
|
+
_OPTION_KEYS = frozenset({"document_format", "records_pointer", "expand", "columns"})
|
|
45
|
+
_COLUMN_KEYS = frozenset({"name", "pointer", "required"})
|
|
46
|
+
_SAFE_COLUMN = re.compile(r"^[^\x00-\x1f\x7f]{1,256}$")
|
|
47
|
+
_MAX_POINTER_BYTES = 1_024
|
|
48
|
+
_MAX_EXPANSIONS = 16
|
|
49
|
+
_MAX_JSON_DEPTH = 64
|
|
50
|
+
_MISSING = object()
|
|
51
|
+
|
|
52
|
+
# JSON expansion retains one small immutable view per output row and expansion level. Keeping the
|
|
53
|
+
# family ceiling below the shared million-row default makes even the worst 16-level chain bounded
|
|
54
|
+
# well inside the 512 MiB clean-room boundary, while ordinary non-expanded tables remain large.
|
|
55
|
+
_JSON_DEFAULT_BUDGETS = bulk_default_budgets()
|
|
56
|
+
_MAX_OVERLAY_RECORDS = 200_000
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _refuse_unknown(mapping: Mapping[str, Any], admitted: frozenset[str], subject: str) -> None:
|
|
60
|
+
unknown = sorted(str(key) for key in mapping if key not in admitted)
|
|
61
|
+
if unknown:
|
|
62
|
+
raise ReaderError(
|
|
63
|
+
"READER_OPTIONS",
|
|
64
|
+
subject,
|
|
65
|
+
f"names no such setting: {', '.join(unknown)}; admitted settings are "
|
|
66
|
+
f"{', '.join(sorted(admitted))}",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _pointer(value: Any, subject: str) -> tuple[str, ...]:
|
|
71
|
+
if not isinstance(value, str):
|
|
72
|
+
raise ReaderError(
|
|
73
|
+
"READER_OPTIONS", subject, "must be an RFC 6901 JSON Pointer of at most 1024 bytes"
|
|
74
|
+
)
|
|
75
|
+
try:
|
|
76
|
+
encoded = value.encode("utf-8")
|
|
77
|
+
except UnicodeEncodeError:
|
|
78
|
+
raise ReaderError(
|
|
79
|
+
"READER_OPTIONS", subject, "must contain valid Unicode scalar values"
|
|
80
|
+
) from None
|
|
81
|
+
if len(encoded) > _MAX_POINTER_BYTES:
|
|
82
|
+
raise ReaderError(
|
|
83
|
+
"READER_OPTIONS", subject, "must be an RFC 6901 JSON Pointer of at most 1024 bytes"
|
|
84
|
+
)
|
|
85
|
+
if value == "":
|
|
86
|
+
return ()
|
|
87
|
+
if not value.startswith("/"):
|
|
88
|
+
raise ReaderError("READER_OPTIONS", subject, "must be empty or start with '/'")
|
|
89
|
+
tokens: list[str] = []
|
|
90
|
+
for raw in value[1:].split("/"):
|
|
91
|
+
index = 0
|
|
92
|
+
decoded: list[str] = []
|
|
93
|
+
while index < len(raw):
|
|
94
|
+
if raw[index] != "~":
|
|
95
|
+
decoded.append(raw[index])
|
|
96
|
+
index += 1
|
|
97
|
+
continue
|
|
98
|
+
if index + 1 >= len(raw) or raw[index + 1] not in "01":
|
|
99
|
+
raise ReaderError(
|
|
100
|
+
"READER_OPTIONS", subject, "contains an invalid RFC 6901 '~' escape"
|
|
101
|
+
)
|
|
102
|
+
decoded.append("~" if raw[index + 1] == "0" else "/")
|
|
103
|
+
index += 2
|
|
104
|
+
tokens.append("".join(decoded))
|
|
105
|
+
return tuple(tokens)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _validated_options(options: Any, budgets: ReaderBudgets) -> dict[str, Any]:
|
|
109
|
+
subject = "reader.json.tabular.decode_options"
|
|
110
|
+
if not isinstance(options, Mapping):
|
|
111
|
+
raise ReaderError("READER_OPTIONS", subject, "must be an object")
|
|
112
|
+
_refuse_unknown(options, _OPTION_KEYS, subject)
|
|
113
|
+
|
|
114
|
+
document_format = options.get("document_format", "json")
|
|
115
|
+
if document_format not in _DOCUMENT_FORMATS:
|
|
116
|
+
raise ReaderError(
|
|
117
|
+
"READER_OPTIONS",
|
|
118
|
+
f"{subject}.document_format",
|
|
119
|
+
f"must be one of {', '.join(_DOCUMENT_FORMATS)}",
|
|
120
|
+
)
|
|
121
|
+
records_pointer = options.get("records_pointer", "")
|
|
122
|
+
_pointer(records_pointer, f"{subject}.records_pointer")
|
|
123
|
+
|
|
124
|
+
expand = options.get("expand", [])
|
|
125
|
+
if not isinstance(expand, list) or len(expand) > _MAX_EXPANSIONS:
|
|
126
|
+
raise ReaderError(
|
|
127
|
+
"READER_OPTIONS",
|
|
128
|
+
f"{subject}.expand",
|
|
129
|
+
f"must be an array of at most {_MAX_EXPANSIONS} JSON Pointers",
|
|
130
|
+
)
|
|
131
|
+
for index, item in enumerate(expand):
|
|
132
|
+
_pointer(item, f"{subject}.expand[{index}]")
|
|
133
|
+
|
|
134
|
+
columns = options.get("columns")
|
|
135
|
+
if not isinstance(columns, list) or not columns or len(columns) > budgets.max_columns:
|
|
136
|
+
raise ReaderError(
|
|
137
|
+
"READER_OPTIONS",
|
|
138
|
+
f"{subject}.columns",
|
|
139
|
+
"must be a non-empty array within the column budget",
|
|
140
|
+
)
|
|
141
|
+
admitted_columns: list[dict[str, Any]] = []
|
|
142
|
+
names: list[str] = []
|
|
143
|
+
for index, column in enumerate(columns):
|
|
144
|
+
item_subject = f"{subject}.columns[{index}]"
|
|
145
|
+
if not isinstance(column, Mapping):
|
|
146
|
+
raise ReaderError("READER_OPTIONS", item_subject, "must be an object")
|
|
147
|
+
_refuse_unknown(column, _COLUMN_KEYS, item_subject)
|
|
148
|
+
name = column.get("name")
|
|
149
|
+
if not isinstance(name, str) or _SAFE_COLUMN.fullmatch(name) is None:
|
|
150
|
+
raise ReaderError(
|
|
151
|
+
"READER_OPTIONS",
|
|
152
|
+
f"{item_subject}.name",
|
|
153
|
+
"must be bounded text without control characters",
|
|
154
|
+
)
|
|
155
|
+
pointer = column.get("pointer")
|
|
156
|
+
_pointer(pointer, f"{item_subject}.pointer")
|
|
157
|
+
required = column.get("required", True)
|
|
158
|
+
if not isinstance(required, bool):
|
|
159
|
+
raise ReaderError("READER_OPTIONS", f"{item_subject}.required", "must be boolean")
|
|
160
|
+
names.append(name)
|
|
161
|
+
admitted_columns.append({"name": name, "pointer": pointer, "required": required})
|
|
162
|
+
if len(set(names)) != len(names):
|
|
163
|
+
raise ReaderError("READER_OPTIONS", f"{subject}.columns", "column names must be unique")
|
|
164
|
+
|
|
165
|
+
return {
|
|
166
|
+
"document_format": document_format,
|
|
167
|
+
"records_pointer": records_pointer,
|
|
168
|
+
"expand": list(expand),
|
|
169
|
+
"columns": admitted_columns,
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
174
|
+
value: dict[str, Any] = {}
|
|
175
|
+
for key, item in pairs:
|
|
176
|
+
if key in value:
|
|
177
|
+
raise ReaderError(
|
|
178
|
+
"READER_ADMISSION",
|
|
179
|
+
"reader.json.tabular.content",
|
|
180
|
+
f"JSON object repeats key {key!r}",
|
|
181
|
+
)
|
|
182
|
+
value[key] = item
|
|
183
|
+
return value
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _decode_documents(content: bytes, document_format: str, budgets: ReaderBudgets) -> list[Any]:
|
|
187
|
+
if not isinstance(content, (bytes, bytearray)):
|
|
188
|
+
raise ReaderError("READER_ADMISSION", "reader.json.tabular.content", "must be exact bytes")
|
|
189
|
+
if not content or len(content) > budgets.max_input_bytes:
|
|
190
|
+
raise ReaderError(
|
|
191
|
+
"READER_BUDGET",
|
|
192
|
+
"reader.json.tabular.content",
|
|
193
|
+
"is empty or exceeds the input byte budget",
|
|
194
|
+
)
|
|
195
|
+
try:
|
|
196
|
+
text = bytes(content).removeprefix(b"\xef\xbb\xbf").decode("utf-8", errors="strict")
|
|
197
|
+
except UnicodeDecodeError:
|
|
198
|
+
raise ReaderError(
|
|
199
|
+
"READER_ADMISSION", "reader.json.tabular.content", "must be strict UTF-8"
|
|
200
|
+
) from None
|
|
201
|
+
|
|
202
|
+
decoder = json.JSONDecoder(
|
|
203
|
+
object_pairs_hook=_unique_object,
|
|
204
|
+
parse_constant=lambda value: (_ for _ in ()).throw(
|
|
205
|
+
ReaderError(
|
|
206
|
+
"READER_ADMISSION",
|
|
207
|
+
"reader.json.tabular.content",
|
|
208
|
+
f"non-finite JSON number {value!r} is not admitted",
|
|
209
|
+
)
|
|
210
|
+
),
|
|
211
|
+
)
|
|
212
|
+
try:
|
|
213
|
+
if document_format == "json":
|
|
214
|
+
documents = [decoder.decode(text)]
|
|
215
|
+
else:
|
|
216
|
+
documents = [decoder.decode(line) for line in text.splitlines() if line.strip()]
|
|
217
|
+
except ReaderError:
|
|
218
|
+
raise
|
|
219
|
+
except (json.JSONDecodeError, ValueError):
|
|
220
|
+
raise ReaderError(
|
|
221
|
+
"READER_DECODE", "reader.json.tabular.content", "JSON syntax is malformed"
|
|
222
|
+
) from None
|
|
223
|
+
except RecursionError:
|
|
224
|
+
raise ReaderError(
|
|
225
|
+
"READER_BUDGET", "reader.json.tabular.content", "JSON nesting exceeds the depth limit"
|
|
226
|
+
) from None
|
|
227
|
+
if not documents:
|
|
228
|
+
raise ReaderError(
|
|
229
|
+
"READER_ADMISSION", "reader.json.tabular.content", "contains no JSON documents"
|
|
230
|
+
)
|
|
231
|
+
return documents
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _validate_tree(value: Any) -> None:
|
|
235
|
+
pending: list[tuple[Any, int]] = [(value, 0)]
|
|
236
|
+
while pending:
|
|
237
|
+
item, depth = pending.pop()
|
|
238
|
+
if depth > _MAX_JSON_DEPTH:
|
|
239
|
+
raise ReaderError(
|
|
240
|
+
"READER_BUDGET",
|
|
241
|
+
"reader.json.tabular.content",
|
|
242
|
+
f"JSON nesting exceeds the fixed {_MAX_JSON_DEPTH}-level limit",
|
|
243
|
+
)
|
|
244
|
+
if isinstance(item, dict):
|
|
245
|
+
for key in item:
|
|
246
|
+
try:
|
|
247
|
+
key.encode("utf-8")
|
|
248
|
+
except UnicodeEncodeError:
|
|
249
|
+
raise ReaderError(
|
|
250
|
+
"READER_ADMISSION",
|
|
251
|
+
"reader.json.tabular.content",
|
|
252
|
+
"contains an object key that is not valid Unicode scalar text",
|
|
253
|
+
) from None
|
|
254
|
+
pending.extend((child, depth + 1) for child in item.values())
|
|
255
|
+
elif isinstance(item, list):
|
|
256
|
+
pending.extend((child, depth + 1) for child in item)
|
|
257
|
+
elif isinstance(item, float) and not math.isfinite(item):
|
|
258
|
+
raise ReaderError(
|
|
259
|
+
"READER_ADMISSION",
|
|
260
|
+
"reader.json.tabular.content",
|
|
261
|
+
"contains a number outside finite binary64 range",
|
|
262
|
+
)
|
|
263
|
+
elif isinstance(item, str):
|
|
264
|
+
try:
|
|
265
|
+
item.encode("utf-8")
|
|
266
|
+
except UnicodeEncodeError:
|
|
267
|
+
raise ReaderError(
|
|
268
|
+
"READER_ADMISSION",
|
|
269
|
+
"reader.json.tabular.content",
|
|
270
|
+
"contains a string that is not valid Unicode scalar text",
|
|
271
|
+
) from None
|
|
272
|
+
elif item is not None and not isinstance(item, str | bool | int | float):
|
|
273
|
+
raise ReaderError(
|
|
274
|
+
"READER_ADMISSION",
|
|
275
|
+
"reader.json.tabular.content",
|
|
276
|
+
f"contains unsupported {type(item).__name__} value",
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _array_index(token: str, subject: str) -> int:
|
|
281
|
+
if (
|
|
282
|
+
token == "-"
|
|
283
|
+
or not token.isascii()
|
|
284
|
+
or not token.isdigit()
|
|
285
|
+
or (len(token) > 1 and token.startswith("0"))
|
|
286
|
+
):
|
|
287
|
+
raise ReaderError("READER_ADMISSION", subject, "does not name an RFC 6901 array index")
|
|
288
|
+
return int(token)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
@dataclass(frozen=True, slots=True)
|
|
292
|
+
class _ExpandedRecord:
|
|
293
|
+
"""One constant-size overlay; parents and wide source containers are never copied."""
|
|
294
|
+
|
|
295
|
+
parent: Any
|
|
296
|
+
pointer: str
|
|
297
|
+
replacement: Any
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _overlay(value: Any, pointer: str) -> tuple[Any, tuple[str, ...]]:
|
|
301
|
+
tokens = _pointer(pointer, "reader.json.tabular.pointer")
|
|
302
|
+
current = value
|
|
303
|
+
while isinstance(current, _ExpandedRecord):
|
|
304
|
+
replacement_tokens = _pointer(current.pointer, "reader.json.tabular.expand")
|
|
305
|
+
if tokens[: len(replacement_tokens)] == replacement_tokens:
|
|
306
|
+
return current.replacement, tokens[len(replacement_tokens) :]
|
|
307
|
+
current = current.parent
|
|
308
|
+
return current, tokens
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _resolve(value: Any, pointer: str, subject: str, *, missing: Any = _MISSING) -> Any:
|
|
312
|
+
current, tokens = _overlay(value, pointer)
|
|
313
|
+
for token in tokens:
|
|
314
|
+
if isinstance(current, dict):
|
|
315
|
+
if token not in current:
|
|
316
|
+
if missing is not _MISSING:
|
|
317
|
+
return missing
|
|
318
|
+
raise ReaderError("READER_ADMISSION", subject, f"does not exist at {pointer!r}")
|
|
319
|
+
current = current[token]
|
|
320
|
+
elif isinstance(current, list):
|
|
321
|
+
index = _array_index(token, subject)
|
|
322
|
+
if index >= len(current):
|
|
323
|
+
if missing is not _MISSING:
|
|
324
|
+
return missing
|
|
325
|
+
raise ReaderError("READER_ADMISSION", subject, f"does not exist at {pointer!r}")
|
|
326
|
+
current = current[index]
|
|
327
|
+
else:
|
|
328
|
+
raise ReaderError("READER_ADMISSION", subject, f"crosses a scalar at {token!r}")
|
|
329
|
+
return current
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _selected_records(documents: Sequence[Any], pointer: str, budgets: ReaderBudgets) -> list[Any]:
|
|
333
|
+
selected: list[Any] = []
|
|
334
|
+
for index, document in enumerate(documents):
|
|
335
|
+
_validate_tree(document)
|
|
336
|
+
value = _resolve(
|
|
337
|
+
document, pointer, f"reader.json.tabular.documents[{index}].records_pointer"
|
|
338
|
+
)
|
|
339
|
+
if isinstance(value, list):
|
|
340
|
+
if len(selected) + len(value) > budgets.max_rows:
|
|
341
|
+
raise ReaderError(
|
|
342
|
+
"READER_BUDGET",
|
|
343
|
+
f"reader.json.tabular.documents[{index}].records_pointer",
|
|
344
|
+
"selection exceeds the row budget",
|
|
345
|
+
)
|
|
346
|
+
selected.extend(value)
|
|
347
|
+
elif isinstance(value, dict):
|
|
348
|
+
if len(selected) >= budgets.max_rows:
|
|
349
|
+
raise ReaderError(
|
|
350
|
+
"READER_BUDGET",
|
|
351
|
+
f"reader.json.tabular.documents[{index}].records_pointer",
|
|
352
|
+
"selection exceeds the row budget",
|
|
353
|
+
)
|
|
354
|
+
selected.append(value)
|
|
355
|
+
else:
|
|
356
|
+
raise ReaderError(
|
|
357
|
+
"READER_ADMISSION",
|
|
358
|
+
f"reader.json.tabular.documents[{index}].records_pointer",
|
|
359
|
+
"must select an object or an array of records",
|
|
360
|
+
)
|
|
361
|
+
return selected
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _expanded_records(
|
|
365
|
+
records: list[Any], pointers: Sequence[str], budgets: ReaderBudgets
|
|
366
|
+
) -> list[Any]:
|
|
367
|
+
current = records
|
|
368
|
+
overlay_records = 0
|
|
369
|
+
for expansion_index, pointer in enumerate(pointers):
|
|
370
|
+
expanded: list[Any] = []
|
|
371
|
+
for record_index, record in enumerate(current):
|
|
372
|
+
subject = f"reader.json.tabular.expand[{expansion_index}].records[{record_index}]"
|
|
373
|
+
values = _resolve(record, pointer, subject)
|
|
374
|
+
if not isinstance(values, list):
|
|
375
|
+
raise ReaderError("READER_ADMISSION", subject, "must select an array")
|
|
376
|
+
if len(expanded) + len(values) > budgets.max_rows:
|
|
377
|
+
raise ReaderError("READER_BUDGET", subject, "expansion exceeds the row budget")
|
|
378
|
+
if overlay_records + len(values) > _MAX_OVERLAY_RECORDS:
|
|
379
|
+
raise ReaderError(
|
|
380
|
+
"READER_BUDGET", subject, "expansion exceeds the cumulative overlay budget"
|
|
381
|
+
)
|
|
382
|
+
expanded.extend(_ExpandedRecord(record, pointer, item) for item in values)
|
|
383
|
+
overlay_records += len(values)
|
|
384
|
+
current = expanded
|
|
385
|
+
if len(current) > budgets.max_rows:
|
|
386
|
+
raise ReaderError("READER_BUDGET", "reader.json.tabular.records", "exceeds the row budget")
|
|
387
|
+
return current
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
def _project(records: Sequence[Any], columns: Sequence[Mapping[str, Any]]) -> Iterator[list[Any]]:
|
|
391
|
+
for row_index, record in enumerate(records):
|
|
392
|
+
row: list[Any] = []
|
|
393
|
+
for column in columns:
|
|
394
|
+
subject = f"reader.json.tabular.rows[{row_index}].{column['name']}"
|
|
395
|
+
value = _resolve(
|
|
396
|
+
record,
|
|
397
|
+
str(column["pointer"]),
|
|
398
|
+
subject,
|
|
399
|
+
missing=_MISSING if bool(column["required"]) else None,
|
|
400
|
+
)
|
|
401
|
+
if isinstance(value, dict | list):
|
|
402
|
+
raise ReaderError(
|
|
403
|
+
"READER_ADMISSION",
|
|
404
|
+
subject,
|
|
405
|
+
"selects an object or array; expand it or select a scalar child",
|
|
406
|
+
)
|
|
407
|
+
row.append(value)
|
|
408
|
+
yield row
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
@dataclass(frozen=True)
|
|
412
|
+
class JsonTabularReader:
|
|
413
|
+
"""A generic, versioned JSON/NDJSON-to-table Reader."""
|
|
414
|
+
|
|
415
|
+
family_id: str = "json.tabular"
|
|
416
|
+
family_version: str = "1.0.0"
|
|
417
|
+
contract_version: str = READER_CONTRACT_VERSION
|
|
418
|
+
output_format: str = _OUTPUT_FORMAT
|
|
419
|
+
accepted_media_types: tuple[str, ...] = ("application/json", "application/x-ndjson")
|
|
420
|
+
default_budgets: ReaderBudgets = field(default_factory=lambda: _JSON_DEFAULT_BUDGETS)
|
|
421
|
+
|
|
422
|
+
def validate_options(self, options: Mapping[str, Any]) -> Mapping[str, Any]:
|
|
423
|
+
return _validated_options(options, self.default_budgets)
|
|
424
|
+
|
|
425
|
+
def decode(self, content: bytes, pin: ReaderPin, budgets: ReaderBudgets) -> ReaderResult:
|
|
426
|
+
options = _validated_options(pin.decode_options, budgets)
|
|
427
|
+
documents = _decode_documents(content, str(options["document_format"]), budgets)
|
|
428
|
+
records = _selected_records(documents, str(options["records_pointer"]), budgets)
|
|
429
|
+
records = _expanded_records(records, list(options["expand"]), budgets)
|
|
430
|
+
columns = list(options["columns"])
|
|
431
|
+
if len(records) * len(columns) > budgets.max_declared_cells:
|
|
432
|
+
raise ReaderError(
|
|
433
|
+
"READER_BUDGET", "reader.json.tabular.records", "exceeds the cell budget"
|
|
434
|
+
)
|
|
435
|
+
names = tuple(str(column["name"]) for column in columns)
|
|
436
|
+
return ReaderResult(
|
|
437
|
+
content=encode_canonical_csv(names, _project(records, columns), budgets=budgets),
|
|
438
|
+
data_format=_OUTPUT_FORMAT,
|
|
439
|
+
media_type=_OUTPUT_MEDIA_TYPE,
|
|
440
|
+
filename=sealed_filename(FALLBACK_STEM, suffix=_OUTPUT_SUFFIX),
|
|
441
|
+
row_count=len(records),
|
|
442
|
+
column_names=names,
|
|
443
|
+
declared_cell_count=len(records) * len(names),
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
@dataclass(frozen=True)
|
|
448
|
+
class JsonTabularReaderV1_1(JsonTabularReader):
|
|
449
|
+
"""``json.tabular@1.1.0`` admits the weak labels, and decodes nothing new.
|
|
450
|
+
|
|
451
|
+
A publisher that serves a JSON document under ``text/plain`` is not describing a different
|
|
452
|
+
document; it is declining to describe the document at all.
|
|
453
|
+
``acquisition/http.py`` already writes that rule down for the byte-stream spelling -- a weak
|
|
454
|
+
label "says only that a server declined to say anything", and the control is the structural
|
|
455
|
+
check behind the sandbox boundary, never the header. Version ``1.0.0`` admitted the two
|
|
456
|
+
labels that name the format outright, which refused whole archives that were never unfit:
|
|
457
|
+
a large share of public agency JSON and NDJSON endpoints answer under ``text/plain``.
|
|
458
|
+
|
|
459
|
+
This is an admission change and not a decode change. ``decode`` is inherited untouched, so
|
|
460
|
+
the bytes a recipe seals under this coordinate are the bytes ``1.0.0`` would have sealed;
|
|
461
|
+
what moves is only which responses reach the decoder. It is a new coordinate for the reason
|
|
462
|
+
``delimited_text@1.1.0`` and ``archive.zip@1.2.0`` are: ``1.0.0`` stays closed to the weak
|
|
463
|
+
labels, so an already-approved recipe cannot silently begin admitting responses its review
|
|
464
|
+
never saw.
|
|
465
|
+
|
|
466
|
+
The label is the only thing widened. ``text/html`` remains outside every allowlist, which is
|
|
467
|
+
what keeps the bot-wall refusal intact, and the decoder's own refusals -- a document that is
|
|
468
|
+
not valid JSON or NDJSON, a pointer that selects nothing, a column that resolves to an object
|
|
469
|
+
or array, and every budget bound -- are unchanged and remain the whole admission. Archive and
|
|
470
|
+
executable bytes arriving under these labels are refused as invalid JSON rather than by
|
|
471
|
+
prefix: the magic table is enforced in ``acquisition/parsing`` and in the container families,
|
|
472
|
+
and this family imports neither.
|
|
473
|
+
|
|
474
|
+
``application/octet-stream`` is deliberately not added here. It is the weak label an object
|
|
475
|
+
store uses for a stored document, and admitting it is a separate widening with its own
|
|
476
|
+
evidence, taken the way ``archive.zip`` took its two: one label at a time, each its own
|
|
477
|
+
coordinate.
|
|
478
|
+
"""
|
|
479
|
+
|
|
480
|
+
family_version: str = "1.1.0"
|
|
481
|
+
accepted_media_types: tuple[str, ...] = (
|
|
482
|
+
"application/json",
|
|
483
|
+
"application/x-ndjson",
|
|
484
|
+
"text/plain",
|
|
485
|
+
)
|