mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,798 @@
|
|
|
1
|
+
"""Strict metadata-only parser for the official Data.gov Catalog API v4 search response."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Any
|
|
9
|
+
from urllib.parse import parse_qs, urlencode, urlsplit
|
|
10
|
+
|
|
11
|
+
from mostlyright.data_harness.acquisition.http import RetrievalLimits
|
|
12
|
+
from mostlyright.data_harness.acquisition.url_policy import MAX_URL_LENGTH
|
|
13
|
+
from mostlyright.data_harness.canonical import (
|
|
14
|
+
CanonicalJSONError,
|
|
15
|
+
canonical_json_bytes,
|
|
16
|
+
canonical_sha256,
|
|
17
|
+
sha256_bytes,
|
|
18
|
+
)
|
|
19
|
+
from mostlyright.data_harness.sources.catalog.harvest.protocol import (
|
|
20
|
+
DATAGOV_V4_HARVESTER_ID,
|
|
21
|
+
DATAGOV_V4_HARVESTER_VERSION,
|
|
22
|
+
CatalogHarvestError,
|
|
23
|
+
HarvestCursor,
|
|
24
|
+
HarvesterDescriptor,
|
|
25
|
+
HarvestPage,
|
|
26
|
+
HarvestResponseEvidence,
|
|
27
|
+
validate_datagov_cursor_text,
|
|
28
|
+
)
|
|
29
|
+
from mostlyright.data_harness.sources.contracts import EvidenceReference, _CanonicalContract
|
|
30
|
+
|
|
31
|
+
DATAGOV_V4_ENDPOINT = "https://api.gsa.gov/technology/datagov/v4/search"
|
|
32
|
+
DATAGOV_V4_RECORD_VERSION = "harness-datagov-v4-normalized-record.v2"
|
|
33
|
+
DATAGOV_V4_MAX_RESPONSE_BYTES = 16 * 1024 * 1024
|
|
34
|
+
DATAGOV_V4_MAX_RECORD_BYTES = 2 * 1024 * 1024
|
|
35
|
+
DATAGOV_V4_MAX_IDENTIFIER_CHARS = 512
|
|
36
|
+
DATAGOV_V4_MAX_CURSOR = 2_048
|
|
37
|
+
DATAGOV_V4_MAX_JSON_DEPTH = 256
|
|
38
|
+
DATAGOV_V4_MAX_NUMBER_TOKEN = 128
|
|
39
|
+
DATAGOV_V4_LIMITS = RetrievalLimits(
|
|
40
|
+
max_response_bytes=DATAGOV_V4_MAX_RESPONSE_BYTES,
|
|
41
|
+
max_aggregate_response_bytes=DATAGOV_V4_MAX_RESPONSE_BYTES,
|
|
42
|
+
max_redirects=0,
|
|
43
|
+
max_requests=1,
|
|
44
|
+
allowed_media_types=("application/json",),
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
_TOP_LEVEL_FIELDS = frozenset(
|
|
48
|
+
{
|
|
49
|
+
"dcat",
|
|
50
|
+
"description",
|
|
51
|
+
"distribution_titles",
|
|
52
|
+
"harvest_record",
|
|
53
|
+
"harvest_record_raw",
|
|
54
|
+
"harvest_record_transformed",
|
|
55
|
+
"has_download",
|
|
56
|
+
"has_spatial",
|
|
57
|
+
"identifier",
|
|
58
|
+
"keyword",
|
|
59
|
+
"last_harvested_date",
|
|
60
|
+
"organization",
|
|
61
|
+
"popularity",
|
|
62
|
+
"publisher",
|
|
63
|
+
"slug",
|
|
64
|
+
"spatial_centroid",
|
|
65
|
+
"spatial_shape",
|
|
66
|
+
"theme",
|
|
67
|
+
"title",
|
|
68
|
+
}
|
|
69
|
+
)
|
|
70
|
+
# Catalog-ingestion observations the provider rewrites on every re-harvest even when no metadata
|
|
71
|
+
# changed. They stay in raw and normalized evidence but never enter the semantic digest, because a
|
|
72
|
+
# digest that moved with a nightly harvest burst would make two adjacent equal pass maps
|
|
73
|
+
# unsatisfiable. `_score`/`_sort` are per-response ranking values and are not allowlisted at all.
|
|
74
|
+
DATAGOV_V4_VOLATILE_FIELDS = frozenset({"last_harvested_date", "popularity"})
|
|
75
|
+
_VOLATILE_PATHS = frozenset(
|
|
76
|
+
path for name in DATAGOV_V4_VOLATILE_FIELDS for path in (f"$.{name}", f"$.{name}[*]")
|
|
77
|
+
)
|
|
78
|
+
_DCAT_FIELDS = frozenset(
|
|
79
|
+
{
|
|
80
|
+
"@type",
|
|
81
|
+
"accessLevel",
|
|
82
|
+
"accrualPeriodicity",
|
|
83
|
+
"agencyDataSeriesURL",
|
|
84
|
+
"agencyProgramURL",
|
|
85
|
+
"analysisUnit",
|
|
86
|
+
"bureauCode",
|
|
87
|
+
"categoryDesignation",
|
|
88
|
+
"collectionInstrument",
|
|
89
|
+
"contactPoint",
|
|
90
|
+
"dataQuality",
|
|
91
|
+
"describedBy",
|
|
92
|
+
"describedByType",
|
|
93
|
+
"description",
|
|
94
|
+
"distribution",
|
|
95
|
+
"identifier",
|
|
96
|
+
"isPartOf",
|
|
97
|
+
"issued",
|
|
98
|
+
"keyword",
|
|
99
|
+
"landingPage",
|
|
100
|
+
"language",
|
|
101
|
+
"license",
|
|
102
|
+
"modified",
|
|
103
|
+
"phone",
|
|
104
|
+
"programCode",
|
|
105
|
+
"publisher",
|
|
106
|
+
"references",
|
|
107
|
+
"rights",
|
|
108
|
+
"spatial",
|
|
109
|
+
"temporal",
|
|
110
|
+
"theme",
|
|
111
|
+
"title",
|
|
112
|
+
}
|
|
113
|
+
)
|
|
114
|
+
_ORGANIZATION_FIELDS = frozenset(
|
|
115
|
+
{"id", "name", "slug", "organization_type", "aliases", "logo", "description"}
|
|
116
|
+
)
|
|
117
|
+
_DISTRIBUTION_FIELDS = frozenset(
|
|
118
|
+
{
|
|
119
|
+
"@type",
|
|
120
|
+
"accessURL",
|
|
121
|
+
"describedBy",
|
|
122
|
+
"describedByType",
|
|
123
|
+
"description",
|
|
124
|
+
"downloadURL",
|
|
125
|
+
"format",
|
|
126
|
+
"mediaType",
|
|
127
|
+
"title",
|
|
128
|
+
}
|
|
129
|
+
)
|
|
130
|
+
_PROVIDER_PATH = re.compile(r'^\$(?:\.[A-Za-z_][A-Za-z0-9_-]*|\[\*\]|\["@type"\])+$')
|
|
131
|
+
_PLAIN_PROVIDER_KEY = re.compile(r"^[A-Za-z_][A-Za-z0-9_-]*$")
|
|
132
|
+
_JSON_NUMBER = re.compile(r"-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class _ForeignJsonError(ValueError):
|
|
136
|
+
pass
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@dataclass(frozen=True)
|
|
140
|
+
class _ForeignNumber:
|
|
141
|
+
lexeme: str
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@dataclass(frozen=True)
|
|
145
|
+
class DatagovProjectedField(_CanonicalContract):
|
|
146
|
+
path: str
|
|
147
|
+
value: Any
|
|
148
|
+
encoding: str = "canonical_json"
|
|
149
|
+
|
|
150
|
+
def __post_init__(self) -> None:
|
|
151
|
+
if (
|
|
152
|
+
not isinstance(self.path, str)
|
|
153
|
+
or not _PROVIDER_PATH.fullmatch(self.path)
|
|
154
|
+
or not _allowed_provider_path(self.path, self.value, self.encoding)
|
|
155
|
+
):
|
|
156
|
+
raise CatalogHarvestError("DATAGOV_PATH", "record.fields.path", "path is invalid")
|
|
157
|
+
if self.encoding not in {"canonical_json", "typed_json_ast.v1"}:
|
|
158
|
+
raise CatalogHarvestError("DATAGOV_VALUE", "record.fields.encoding", "is invalid")
|
|
159
|
+
try:
|
|
160
|
+
canonical_json_bytes(self.value)
|
|
161
|
+
except Exception as error:
|
|
162
|
+
raise CatalogHarvestError(
|
|
163
|
+
"DATAGOV_VALUE", "record.fields.value", "value is not strict JSON"
|
|
164
|
+
) from error
|
|
165
|
+
if self.encoding == "typed_json_ast.v1":
|
|
166
|
+
_validate_typed_json_ast(self.value)
|
|
167
|
+
|
|
168
|
+
def to_dict(self) -> dict[str, Any]:
|
|
169
|
+
result = {"path": self.path, "value": self.value}
|
|
170
|
+
if self.encoding != "canonical_json":
|
|
171
|
+
result["encoding"] = self.encoding
|
|
172
|
+
return result
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@dataclass(frozen=True)
|
|
176
|
+
class DatagovV4Record(_CanonicalContract):
|
|
177
|
+
record_id: str
|
|
178
|
+
fields: tuple[DatagovProjectedField, ...]
|
|
179
|
+
raw_response_sha256: str
|
|
180
|
+
observations: tuple[DatagovProjectedField, ...] = ()
|
|
181
|
+
protocol: str = "datagov_v4"
|
|
182
|
+
schema_version: str = DATAGOV_V4_RECORD_VERSION
|
|
183
|
+
|
|
184
|
+
def __post_init__(self) -> None:
|
|
185
|
+
if self.protocol != "datagov_v4" or self.schema_version != DATAGOV_V4_RECORD_VERSION:
|
|
186
|
+
raise CatalogHarvestError("VERSION", "record", "unsupported v4 record")
|
|
187
|
+
if not isinstance(self.record_id, str) or not self.record_id:
|
|
188
|
+
raise CatalogHarvestError(
|
|
189
|
+
"DATAGOV_IDENTIFIER", "record.record_id", "identifier must be nonempty"
|
|
190
|
+
)
|
|
191
|
+
try:
|
|
192
|
+
self.record_id.encode("utf-8", errors="strict")
|
|
193
|
+
except UnicodeEncodeError:
|
|
194
|
+
raise CatalogHarvestError(
|
|
195
|
+
"DATAGOV_IDENTIFIER", "record.record_id", "identifier must be valid Unicode"
|
|
196
|
+
) from None
|
|
197
|
+
if len(self.record_id) > DATAGOV_V4_MAX_IDENTIFIER_CHARS:
|
|
198
|
+
raise CatalogHarvestError(
|
|
199
|
+
"DATAGOV_IDENTIFIER",
|
|
200
|
+
"record.record_id",
|
|
201
|
+
"identifier exceeds the shared provider identity bound",
|
|
202
|
+
)
|
|
203
|
+
for character in self.record_id:
|
|
204
|
+
codepoint = ord(character)
|
|
205
|
+
if (
|
|
206
|
+
codepoint < 0x20
|
|
207
|
+
or 0x7F <= codepoint <= 0x9F
|
|
208
|
+
or (character.isspace() and character != " ")
|
|
209
|
+
):
|
|
210
|
+
raise CatalogHarvestError(
|
|
211
|
+
"DATAGOV_IDENTIFIER",
|
|
212
|
+
"record.record_id",
|
|
213
|
+
"identifier contains unsafe provider identity text",
|
|
214
|
+
)
|
|
215
|
+
if not isinstance(self.fields, tuple) or not self.fields:
|
|
216
|
+
raise CatalogHarvestError("DATAGOV_FIELDS", "record.fields", "fields are required")
|
|
217
|
+
if any(not isinstance(field, DatagovProjectedField) for field in self.fields):
|
|
218
|
+
raise CatalogHarvestError(
|
|
219
|
+
"DATAGOV_FIELDS", "record.fields", "fields must use the strict contract"
|
|
220
|
+
)
|
|
221
|
+
paths = tuple(field.path for field in self.fields)
|
|
222
|
+
if paths != tuple(sorted(paths)) or len(paths) != len(set(paths)):
|
|
223
|
+
raise CatalogHarvestError(
|
|
224
|
+
"DATAGOV_FIELDS", "record.fields", "field paths must be sorted and unique"
|
|
225
|
+
)
|
|
226
|
+
if any(path in _VOLATILE_PATHS for path in paths):
|
|
227
|
+
raise CatalogHarvestError(
|
|
228
|
+
"DATAGOV_FIELDS",
|
|
229
|
+
"record.fields",
|
|
230
|
+
"volatile observation fields may not carry a semantic field path",
|
|
231
|
+
)
|
|
232
|
+
if not isinstance(self.observations, tuple) or any(
|
|
233
|
+
not isinstance(field, DatagovProjectedField) for field in self.observations
|
|
234
|
+
):
|
|
235
|
+
raise CatalogHarvestError(
|
|
236
|
+
"DATAGOV_OBSERVATIONS",
|
|
237
|
+
"record.observations",
|
|
238
|
+
"observations must use the strict contract",
|
|
239
|
+
)
|
|
240
|
+
observed_paths = tuple(field.path for field in self.observations)
|
|
241
|
+
if (
|
|
242
|
+
observed_paths != tuple(sorted(observed_paths))
|
|
243
|
+
or len(observed_paths) != len(set(observed_paths))
|
|
244
|
+
or any(path not in _VOLATILE_PATHS for path in observed_paths)
|
|
245
|
+
):
|
|
246
|
+
raise CatalogHarvestError(
|
|
247
|
+
"DATAGOV_OBSERVATIONS",
|
|
248
|
+
"record.observations",
|
|
249
|
+
"observation paths must be sorted, unique, and volatile",
|
|
250
|
+
)
|
|
251
|
+
identifier_fields = tuple(field for field in self.fields if field.path == "$.identifier")
|
|
252
|
+
if (
|
|
253
|
+
len(identifier_fields) != 1
|
|
254
|
+
or identifier_fields[0].encoding != "canonical_json"
|
|
255
|
+
or identifier_fields[0].value != self.record_id
|
|
256
|
+
):
|
|
257
|
+
raise CatalogHarvestError(
|
|
258
|
+
"DATAGOV_IDENTIFIER",
|
|
259
|
+
"record.fields",
|
|
260
|
+
"exactly one canonical $.identifier must equal record_id",
|
|
261
|
+
)
|
|
262
|
+
if len(self.raw_response_sha256) != 64 or any(
|
|
263
|
+
character not in "0123456789abcdef" for character in self.raw_response_sha256
|
|
264
|
+
):
|
|
265
|
+
raise CatalogHarvestError("DIGEST", "record.raw_response_sha256", "must be SHA-256")
|
|
266
|
+
try:
|
|
267
|
+
semantic = self._semantic_dict()
|
|
268
|
+
observed = self._observed_dict()
|
|
269
|
+
normalized_size = len(canonical_json_bytes(observed))
|
|
270
|
+
normalized_record = dict(observed)
|
|
271
|
+
normalized_record["raw_response_sha256"] = self.raw_response_sha256
|
|
272
|
+
normalized_record["normalized_sha256"] = canonical_sha256(semantic)
|
|
273
|
+
canonical_json_bytes(
|
|
274
|
+
{
|
|
275
|
+
"schema_version": "harness-catalog-fill-shard.v1",
|
|
276
|
+
"kind": "normalized",
|
|
277
|
+
"payload": {
|
|
278
|
+
"schema_version": "harness-datagov-v4-normalized-record-segment.v1",
|
|
279
|
+
"raw_response_sha256": self.raw_response_sha256,
|
|
280
|
+
"first_record": 0,
|
|
281
|
+
"records": [normalized_record],
|
|
282
|
+
},
|
|
283
|
+
}
|
|
284
|
+
)
|
|
285
|
+
except CanonicalJSONError:
|
|
286
|
+
raise CatalogHarvestError(
|
|
287
|
+
"DATAGOV_RECORD_LIMIT",
|
|
288
|
+
"record",
|
|
289
|
+
"normalized record exceeds its canonical member contract",
|
|
290
|
+
) from None
|
|
291
|
+
if normalized_size > DATAGOV_V4_MAX_RECORD_BYTES:
|
|
292
|
+
raise CatalogHarvestError(
|
|
293
|
+
"DATAGOV_RECORD_LIMIT", "record", "normalized record exceeds its byte bound"
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
def _semantic_dict(self) -> dict[str, Any]:
|
|
297
|
+
"""Exactly the normalized allowlisted metadata fields the pass map converges on."""
|
|
298
|
+
|
|
299
|
+
return {
|
|
300
|
+
"schema_version": self.schema_version,
|
|
301
|
+
"protocol": self.protocol,
|
|
302
|
+
"record_id": self.record_id,
|
|
303
|
+
"fields": [field.to_dict() for field in self.fields],
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
def _observed_dict(self) -> dict[str, Any]:
|
|
307
|
+
"""The semantic record plus retained, non-semantic volatile observation values."""
|
|
308
|
+
|
|
309
|
+
result = self._semantic_dict()
|
|
310
|
+
if self.observations:
|
|
311
|
+
result["observations"] = [field.to_dict() for field in self.observations]
|
|
312
|
+
return result
|
|
313
|
+
|
|
314
|
+
@property
|
|
315
|
+
def normalized_sha256(self) -> str:
|
|
316
|
+
return canonical_sha256(self._semantic_dict())
|
|
317
|
+
|
|
318
|
+
@property
|
|
319
|
+
def digest(self) -> str:
|
|
320
|
+
return self.normalized_sha256
|
|
321
|
+
|
|
322
|
+
def to_dict(self) -> dict[str, Any]:
|
|
323
|
+
result = self._observed_dict()
|
|
324
|
+
result["raw_response_sha256"] = self.raw_response_sha256
|
|
325
|
+
result["normalized_sha256"] = self.normalized_sha256
|
|
326
|
+
return result
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class DatagovV4Harvester:
|
|
330
|
+
descriptor = HarvesterDescriptor(
|
|
331
|
+
protocol="datagov_v4",
|
|
332
|
+
harvester_id=DATAGOV_V4_HARVESTER_ID,
|
|
333
|
+
harvester_version=DATAGOV_V4_HARVESTER_VERSION,
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
def parse(
|
|
337
|
+
self,
|
|
338
|
+
payload: bytes,
|
|
339
|
+
*,
|
|
340
|
+
uri: str,
|
|
341
|
+
observed_at: str,
|
|
342
|
+
evidence: EvidenceReference,
|
|
343
|
+
) -> tuple[DatagovV4Record, ...]:
|
|
344
|
+
return self.parse_page(
|
|
345
|
+
payload,
|
|
346
|
+
uri=uri,
|
|
347
|
+
observed_at=observed_at,
|
|
348
|
+
evidence=evidence,
|
|
349
|
+
response_evidence=None,
|
|
350
|
+
cursor=None,
|
|
351
|
+
).records
|
|
352
|
+
|
|
353
|
+
def parse_page(
|
|
354
|
+
self,
|
|
355
|
+
payload: bytes,
|
|
356
|
+
*,
|
|
357
|
+
uri: str,
|
|
358
|
+
observed_at: str,
|
|
359
|
+
evidence: EvidenceReference,
|
|
360
|
+
response_evidence: HarvestResponseEvidence | None,
|
|
361
|
+
cursor: HarvestCursor | None,
|
|
362
|
+
) -> HarvestPage:
|
|
363
|
+
if not isinstance(payload, bytes) or len(payload) > DATAGOV_V4_MAX_RESPONSE_BYTES:
|
|
364
|
+
raise CatalogHarvestError(
|
|
365
|
+
"DATAGOV_RESPONSE_LIMIT", "harvest.response", "v4 response exceeds 16 MiB"
|
|
366
|
+
)
|
|
367
|
+
if cursor is not None and cursor.protocol != "datagov_v4":
|
|
368
|
+
raise CatalogHarvestError("HARVEST_CURSOR", "cursor", "protocol mismatch")
|
|
369
|
+
try:
|
|
370
|
+
_preflight_json(payload)
|
|
371
|
+
document = json.loads(
|
|
372
|
+
payload.decode("utf-8", errors="strict"),
|
|
373
|
+
object_pairs_hook=_unique_object,
|
|
374
|
+
parse_float=_bounded_foreign_number,
|
|
375
|
+
parse_int=_bounded_foreign_integer,
|
|
376
|
+
parse_constant=_reject_nonfinite,
|
|
377
|
+
)
|
|
378
|
+
except (
|
|
379
|
+
UnicodeDecodeError,
|
|
380
|
+
json.JSONDecodeError,
|
|
381
|
+
_ForeignJsonError,
|
|
382
|
+
RecursionError,
|
|
383
|
+
ValueError,
|
|
384
|
+
OverflowError,
|
|
385
|
+
):
|
|
386
|
+
raise CatalogHarvestError(
|
|
387
|
+
"DATAGOV_JSON", "harvest.response", "v4 response must be strict UTF-8 JSON"
|
|
388
|
+
) from None
|
|
389
|
+
if not isinstance(document, dict) or set(document) - {"results", "sort", "after"}:
|
|
390
|
+
raise CatalogHarvestError(
|
|
391
|
+
"DATAGOV_ENVELOPE", "harvest.response", "v4 response envelope is invalid"
|
|
392
|
+
)
|
|
393
|
+
if document.get("sort") != _requested_sort(uri) or not isinstance(
|
|
394
|
+
document.get("results"), list
|
|
395
|
+
):
|
|
396
|
+
raise CatalogHarvestError(
|
|
397
|
+
"DATAGOV_ENVELOPE", "harvest.response", "v4 sort/results are invalid"
|
|
398
|
+
)
|
|
399
|
+
requested_page_size = _requested_page_size(uri)
|
|
400
|
+
if len(document["results"]) > requested_page_size:
|
|
401
|
+
raise CatalogHarvestError(
|
|
402
|
+
"DATAGOV_PAGE_SIZE",
|
|
403
|
+
"harvest.response.results",
|
|
404
|
+
"v4 response contains more results than the exact request admitted",
|
|
405
|
+
)
|
|
406
|
+
after = document.get("after")
|
|
407
|
+
if after is not None and (
|
|
408
|
+
not isinstance(after, str) or not after or len(after) > DATAGOV_V4_MAX_CURSOR
|
|
409
|
+
):
|
|
410
|
+
raise CatalogHarvestError(
|
|
411
|
+
"HARVEST_CURSOR", "harvest.response.after", "v4 cursor is invalid"
|
|
412
|
+
)
|
|
413
|
+
if after is not None:
|
|
414
|
+
_require_safe_cursor(after)
|
|
415
|
+
raw_sha256 = sha256_bytes(payload)
|
|
416
|
+
records: list[DatagovV4Record] = []
|
|
417
|
+
skipped: list[str] = []
|
|
418
|
+
for index, item in enumerate(document["results"]):
|
|
419
|
+
if not isinstance(item, dict):
|
|
420
|
+
skipped.append(f"index:{index}")
|
|
421
|
+
continue
|
|
422
|
+
identifier = item.get("identifier")
|
|
423
|
+
if not isinstance(identifier, str) or not identifier:
|
|
424
|
+
skipped.append(f"index:{index}")
|
|
425
|
+
continue
|
|
426
|
+
try:
|
|
427
|
+
fields, observations = _project_record(item)
|
|
428
|
+
records.append(
|
|
429
|
+
DatagovV4Record(
|
|
430
|
+
record_id=identifier,
|
|
431
|
+
fields=fields,
|
|
432
|
+
observations=observations,
|
|
433
|
+
raw_response_sha256=raw_sha256,
|
|
434
|
+
)
|
|
435
|
+
)
|
|
436
|
+
except CatalogHarvestError:
|
|
437
|
+
skipped.append(f"sha256:{sha256_bytes(identifier.encode('utf-8', 'replace'))}")
|
|
438
|
+
next_cursor = (
|
|
439
|
+
None
|
|
440
|
+
if after is None
|
|
441
|
+
else HarvestCursor(protocol="datagov_v4", kind="after", value=after)
|
|
442
|
+
)
|
|
443
|
+
return HarvestPage(
|
|
444
|
+
protocol="datagov_v4",
|
|
445
|
+
records=tuple(records),
|
|
446
|
+
next_cursor=next_cursor,
|
|
447
|
+
provider_count=None,
|
|
448
|
+
count_basis="not_reported",
|
|
449
|
+
skipped_record_ids=tuple(sorted(skipped)),
|
|
450
|
+
evidence=evidence,
|
|
451
|
+
response_evidence=response_evidence,
|
|
452
|
+
)
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def _project_record(
|
|
456
|
+
item: dict[str, Any],
|
|
457
|
+
) -> tuple[tuple[DatagovProjectedField, ...], tuple[DatagovProjectedField, ...]]:
|
|
458
|
+
"""Project the allowlisted union, then split semantic fields from volatile observations."""
|
|
459
|
+
|
|
460
|
+
projected: list[DatagovProjectedField] = []
|
|
461
|
+
base = "$"
|
|
462
|
+
for name in sorted(set(item) & _TOP_LEVEL_FIELDS):
|
|
463
|
+
value = item[name]
|
|
464
|
+
path = f"{base}.{name}"
|
|
465
|
+
if name == "dcat":
|
|
466
|
+
if not isinstance(value, dict):
|
|
467
|
+
raise CatalogHarvestError("DATAGOV_VALUE", path, "dcat must be a closed object")
|
|
468
|
+
for child in sorted(set(value) & _DCAT_FIELDS):
|
|
469
|
+
_flatten_allowed(
|
|
470
|
+
projected,
|
|
471
|
+
_provider_child_path(path, child),
|
|
472
|
+
value[child],
|
|
473
|
+
child_allowlist=_DISTRIBUTION_FIELDS if child == "distribution" else None,
|
|
474
|
+
allow_structured_value=child != "distribution",
|
|
475
|
+
)
|
|
476
|
+
elif name == "organization":
|
|
477
|
+
if not isinstance(value, dict):
|
|
478
|
+
raise CatalogHarvestError(
|
|
479
|
+
"DATAGOV_VALUE", path, "organization must be a closed object"
|
|
480
|
+
)
|
|
481
|
+
for child in sorted(set(value) & _ORGANIZATION_FIELDS):
|
|
482
|
+
_flatten_allowed(
|
|
483
|
+
projected,
|
|
484
|
+
_provider_child_path(path, child),
|
|
485
|
+
value[child],
|
|
486
|
+
allow_structured_value=True,
|
|
487
|
+
)
|
|
488
|
+
else:
|
|
489
|
+
_flatten_allowed(projected, path, value, allow_structured_value=True)
|
|
490
|
+
ordered = sorted(projected, key=lambda field: field.path)
|
|
491
|
+
return (
|
|
492
|
+
tuple(field for field in ordered if field.path not in _VOLATILE_PATHS),
|
|
493
|
+
tuple(field for field in ordered if field.path in _VOLATILE_PATHS),
|
|
494
|
+
)
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def _flatten_allowed(
|
|
498
|
+
target: list[DatagovProjectedField],
|
|
499
|
+
path: str,
|
|
500
|
+
value: Any,
|
|
501
|
+
*,
|
|
502
|
+
child_allowlist: frozenset[str] | None = None,
|
|
503
|
+
allow_structured_value: bool = False,
|
|
504
|
+
) -> None:
|
|
505
|
+
if isinstance(value, dict):
|
|
506
|
+
if child_allowlist is None:
|
|
507
|
+
if not allow_structured_value:
|
|
508
|
+
raise CatalogHarvestError(
|
|
509
|
+
"DATAGOV_VALUE", path, "nested objects require a closed field schema"
|
|
510
|
+
)
|
|
511
|
+
target.append(_projected_field(path, value))
|
|
512
|
+
return
|
|
513
|
+
keys = set(value) & child_allowlist
|
|
514
|
+
if not keys:
|
|
515
|
+
target.append(DatagovProjectedField(path, {}))
|
|
516
|
+
return
|
|
517
|
+
for key in sorted(keys):
|
|
518
|
+
_flatten_allowed(target, _provider_child_path(path, key), value[key])
|
|
519
|
+
return
|
|
520
|
+
if isinstance(value, list):
|
|
521
|
+
if not value:
|
|
522
|
+
target.append(DatagovProjectedField(f"{path}[*]", []))
|
|
523
|
+
return
|
|
524
|
+
if child_allowlist is not None:
|
|
525
|
+
if not all(isinstance(item, dict) for item in value):
|
|
526
|
+
raise CatalogHarvestError(
|
|
527
|
+
"DATAGOV_VALUE", path, "closed object arrays may not mix member types"
|
|
528
|
+
)
|
|
529
|
+
keys = set().union(*(set(item) for item in value)) & child_allowlist
|
|
530
|
+
for key in sorted(keys):
|
|
531
|
+
values = [item[key] for item in value if key in item]
|
|
532
|
+
stable_value: Any = values[0] if len(values) == 1 else values
|
|
533
|
+
target.append(
|
|
534
|
+
_projected_field(_provider_child_path(f"{path}[*]", key), stable_value)
|
|
535
|
+
)
|
|
536
|
+
return
|
|
537
|
+
if not allow_structured_value and any(isinstance(item, (dict, list)) for item in value):
|
|
538
|
+
raise CatalogHarvestError(
|
|
539
|
+
"DATAGOV_VALUE", path, "nested arrays or objects require a closed field schema"
|
|
540
|
+
)
|
|
541
|
+
target.append(_projected_field(f"{path}[*]", value))
|
|
542
|
+
return
|
|
543
|
+
target.append(_projected_field(path, value))
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
547
|
+
result: dict[str, Any] = {}
|
|
548
|
+
for key, value in pairs:
|
|
549
|
+
if key in result:
|
|
550
|
+
raise _ForeignJsonError("duplicate object key")
|
|
551
|
+
result[key] = value
|
|
552
|
+
return result
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _provider_child_path(path: str, key: str) -> str:
|
|
556
|
+
if key == "@type":
|
|
557
|
+
return f'{path}["@type"]'
|
|
558
|
+
if not _PLAIN_PROVIDER_KEY.fullmatch(key):
|
|
559
|
+
raise _ForeignJsonError("allowlisted provider key has no canonical path spelling")
|
|
560
|
+
return f"{path}.{key}"
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _allowed_provider_path(path: str, value: Any, encoding: str) -> bool:
|
|
564
|
+
if path == "$.dcat.distribution":
|
|
565
|
+
return encoding == "canonical_json" and value == {}
|
|
566
|
+
if path == "$.dcat.distribution[*]":
|
|
567
|
+
return encoding == "canonical_json" and value == []
|
|
568
|
+
for name in _TOP_LEVEL_FIELDS - {"dcat", "organization"}:
|
|
569
|
+
if path in {f"$.{name}", f"$.{name}[*]"}:
|
|
570
|
+
return True
|
|
571
|
+
for name in _ORGANIZATION_FIELDS:
|
|
572
|
+
if path in {f"$.organization.{name}", f"$.organization.{name}[*]"}:
|
|
573
|
+
return True
|
|
574
|
+
for name in _DCAT_FIELDS - {"distribution"}:
|
|
575
|
+
base = _provider_child_path("$.dcat", name)
|
|
576
|
+
if path in {base, f"{base}[*]"}:
|
|
577
|
+
return True
|
|
578
|
+
for name in _DISTRIBUTION_FIELDS:
|
|
579
|
+
if path == _provider_child_path("$.dcat.distribution[*]", name):
|
|
580
|
+
return True
|
|
581
|
+
return False
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
def _validate_typed_json_ast(value: Any) -> None:
|
|
585
|
+
pending: list[tuple[Any, int]] = [(value, 1)]
|
|
586
|
+
visited = 0
|
|
587
|
+
while pending:
|
|
588
|
+
current, depth = pending.pop()
|
|
589
|
+
visited += 1
|
|
590
|
+
if visited > 1_000_000 or depth > DATAGOV_V4_MAX_JSON_DEPTH:
|
|
591
|
+
raise CatalogHarvestError(
|
|
592
|
+
"DATAGOV_VALUE", "record.fields.value", "typed JSON AST exceeds its bound"
|
|
593
|
+
)
|
|
594
|
+
if not isinstance(current, dict) or not isinstance(current.get("type"), str):
|
|
595
|
+
raise CatalogHarvestError(
|
|
596
|
+
"DATAGOV_VALUE", "record.fields.value", "typed JSON AST node is invalid"
|
|
597
|
+
)
|
|
598
|
+
kind = current["type"]
|
|
599
|
+
if kind == "null":
|
|
600
|
+
valid = set(current) == {"type"}
|
|
601
|
+
elif kind == "boolean":
|
|
602
|
+
valid = set(current) == {"type", "value"} and type(current["value"]) is bool
|
|
603
|
+
elif kind == "integer":
|
|
604
|
+
valid = set(current) == {"type", "value"} and type(current["value"]) is int
|
|
605
|
+
elif kind == "number":
|
|
606
|
+
lexeme = current.get("lexeme")
|
|
607
|
+
valid = (
|
|
608
|
+
set(current) == {"type", "lexeme"}
|
|
609
|
+
and isinstance(lexeme, str)
|
|
610
|
+
and len(lexeme) <= DATAGOV_V4_MAX_NUMBER_TOKEN
|
|
611
|
+
and _JSON_NUMBER.fullmatch(lexeme) is not None
|
|
612
|
+
)
|
|
613
|
+
elif kind == "string":
|
|
614
|
+
valid = set(current) == {"type", "value"} and isinstance(current["value"], str)
|
|
615
|
+
elif kind == "array":
|
|
616
|
+
items = current.get("items")
|
|
617
|
+
valid = set(current) == {"type", "items"} and isinstance(items, list)
|
|
618
|
+
if valid:
|
|
619
|
+
pending.extend((item, depth + 1) for item in reversed(items))
|
|
620
|
+
elif kind == "object":
|
|
621
|
+
entries = current.get("entries")
|
|
622
|
+
valid = set(current) == {"type", "entries"} and isinstance(entries, list)
|
|
623
|
+
if valid:
|
|
624
|
+
keys: list[str] = []
|
|
625
|
+
for entry in entries:
|
|
626
|
+
if (
|
|
627
|
+
not isinstance(entry, dict)
|
|
628
|
+
or set(entry) != {"key", "value"}
|
|
629
|
+
or not isinstance(entry["key"], str)
|
|
630
|
+
):
|
|
631
|
+
valid = False
|
|
632
|
+
break
|
|
633
|
+
keys.append(entry["key"])
|
|
634
|
+
pending.append((entry["value"], depth + 1))
|
|
635
|
+
valid = valid and keys == sorted(set(keys))
|
|
636
|
+
else:
|
|
637
|
+
valid = False
|
|
638
|
+
if not valid:
|
|
639
|
+
raise CatalogHarvestError(
|
|
640
|
+
"DATAGOV_VALUE", "record.fields.value", "typed JSON AST node is invalid"
|
|
641
|
+
)
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def _bounded_foreign_integer(value: str) -> int | _ForeignNumber:
|
|
645
|
+
if len(value) > DATAGOV_V4_MAX_NUMBER_TOKEN:
|
|
646
|
+
raise _ForeignJsonError("numeric token exceeds bound")
|
|
647
|
+
parsed = int(value)
|
|
648
|
+
return parsed if -(1 << 63) <= parsed <= (1 << 63) - 1 else _ForeignNumber(value)
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
def _reject_nonfinite(_value: str) -> None:
|
|
652
|
+
raise _ForeignJsonError("non-finite number")
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def _bounded_foreign_number(value: str) -> _ForeignNumber:
|
|
656
|
+
if len(value) > DATAGOV_V4_MAX_NUMBER_TOKEN:
|
|
657
|
+
raise _ForeignJsonError("numeric token exceeds bound")
|
|
658
|
+
return _ForeignNumber(value)
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def _projected_field(path: str, value: Any) -> DatagovProjectedField:
|
|
662
|
+
try:
|
|
663
|
+
if not _contains_foreign_number(value):
|
|
664
|
+
return DatagovProjectedField(path, value)
|
|
665
|
+
return DatagovProjectedField(path, _typed_json_ast(value), encoding="typed_json_ast.v1")
|
|
666
|
+
except (_ForeignJsonError, RecursionError):
|
|
667
|
+
raise CatalogHarvestError(
|
|
668
|
+
"DATAGOV_VALUE", path, "provider value exceeds the bounded traversal contract"
|
|
669
|
+
) from None
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
def _contains_foreign_number(value: Any) -> bool:
|
|
673
|
+
pending = [value]
|
|
674
|
+
visited = 0
|
|
675
|
+
while pending:
|
|
676
|
+
current = pending.pop()
|
|
677
|
+
visited += 1
|
|
678
|
+
if visited > 1_000_000:
|
|
679
|
+
raise _ForeignJsonError("foreign JSON value exceeds traversal bound")
|
|
680
|
+
if isinstance(current, _ForeignNumber):
|
|
681
|
+
return True
|
|
682
|
+
if isinstance(current, list):
|
|
683
|
+
pending.extend(current)
|
|
684
|
+
elif isinstance(current, dict):
|
|
685
|
+
pending.extend(current.values())
|
|
686
|
+
return False
|
|
687
|
+
|
|
688
|
+
|
|
689
|
+
def _typed_json_ast(value: Any) -> dict[str, Any]:
|
|
690
|
+
if value is None:
|
|
691
|
+
return {"type": "null"}
|
|
692
|
+
if type(value) is bool:
|
|
693
|
+
return {"type": "boolean", "value": value}
|
|
694
|
+
if type(value) is int:
|
|
695
|
+
return {"type": "integer", "value": value}
|
|
696
|
+
if isinstance(value, _ForeignNumber):
|
|
697
|
+
return {"type": "number", "lexeme": value.lexeme}
|
|
698
|
+
if isinstance(value, str):
|
|
699
|
+
return {"type": "string", "value": value}
|
|
700
|
+
if isinstance(value, list):
|
|
701
|
+
return {"type": "array", "items": [_typed_json_ast(item) for item in value]}
|
|
702
|
+
if isinstance(value, dict):
|
|
703
|
+
return {
|
|
704
|
+
"type": "object",
|
|
705
|
+
"entries": [
|
|
706
|
+
{"key": key, "value": _typed_json_ast(value[key])} for key in sorted(value)
|
|
707
|
+
],
|
|
708
|
+
}
|
|
709
|
+
raise _ForeignJsonError("foreign JSON value type is invalid")
|
|
710
|
+
|
|
711
|
+
|
|
712
|
+
def _preflight_json(payload: bytes) -> None:
|
|
713
|
+
"""Bound nesting and numeric lexemes before CPython's decoder touches them."""
|
|
714
|
+
|
|
715
|
+
depth = 0
|
|
716
|
+
in_string = False
|
|
717
|
+
escaped = False
|
|
718
|
+
index = 0
|
|
719
|
+
while index < len(payload):
|
|
720
|
+
byte = payload[index]
|
|
721
|
+
if in_string:
|
|
722
|
+
if escaped:
|
|
723
|
+
escaped = False
|
|
724
|
+
elif byte == 0x5C:
|
|
725
|
+
escaped = True
|
|
726
|
+
elif byte == 0x22:
|
|
727
|
+
in_string = False
|
|
728
|
+
index += 1
|
|
729
|
+
continue
|
|
730
|
+
if byte == 0x22:
|
|
731
|
+
in_string = True
|
|
732
|
+
elif byte in {0x5B, 0x7B}:
|
|
733
|
+
depth += 1
|
|
734
|
+
if depth > DATAGOV_V4_MAX_JSON_DEPTH:
|
|
735
|
+
raise _ForeignJsonError("JSON nesting exceeds bound")
|
|
736
|
+
elif byte in {0x5D, 0x7D}:
|
|
737
|
+
depth -= 1
|
|
738
|
+
if depth < 0:
|
|
739
|
+
raise _ForeignJsonError("JSON nesting is invalid")
|
|
740
|
+
elif byte == 0x2D or 0x30 <= byte <= 0x39:
|
|
741
|
+
end = index + 1
|
|
742
|
+
while end < len(payload) and payload[end] in b"0123456789eE+.-":
|
|
743
|
+
end += 1
|
|
744
|
+
if end - index > DATAGOV_V4_MAX_NUMBER_TOKEN:
|
|
745
|
+
raise _ForeignJsonError("numeric token exceeds bound")
|
|
746
|
+
index = end
|
|
747
|
+
continue
|
|
748
|
+
index += 1
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
#: The sorts the v4 cursor contract admits, longest first so the URL bound stays the strictest.
|
|
752
|
+
DATAGOV_V4_SORT_ORDERS = ("last_harvested_date", "relevance")
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _requested_sort(uri: str) -> str:
|
|
756
|
+
query = parse_qs(urlsplit(uri).query, keep_blank_values=True)
|
|
757
|
+
values = query.get("sort")
|
|
758
|
+
if values is None:
|
|
759
|
+
return DATAGOV_V4_SORT_ORDERS[0]
|
|
760
|
+
if len(values) != 1 or values[0] not in DATAGOV_V4_SORT_ORDERS:
|
|
761
|
+
raise CatalogHarvestError(
|
|
762
|
+
"DATAGOV_SORT", "harvest.request.sort", "sort is not one this contract can page"
|
|
763
|
+
)
|
|
764
|
+
return values[0]
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
def _requested_page_size(uri: str) -> int:
|
|
768
|
+
query = parse_qs(urlsplit(uri).query, keep_blank_values=True)
|
|
769
|
+
values = query.get("per_page")
|
|
770
|
+
if values is None:
|
|
771
|
+
return 1_000
|
|
772
|
+
if len(values) != 1 or not re.fullmatch(r"[1-9][0-9]{0,3}", values[0]):
|
|
773
|
+
raise CatalogHarvestError(
|
|
774
|
+
"DATAGOV_PAGE_SIZE", "harvest.request.per_page", "per_page is invalid"
|
|
775
|
+
)
|
|
776
|
+
page_size = int(values[0])
|
|
777
|
+
if page_size > 1_000:
|
|
778
|
+
raise CatalogHarvestError(
|
|
779
|
+
"DATAGOV_PAGE_SIZE", "harvest.request.per_page", "per_page exceeds 1000"
|
|
780
|
+
)
|
|
781
|
+
return page_size
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
def _require_safe_cursor(value: str) -> None:
|
|
785
|
+
validate_datagov_cursor_text(value, "harvest.response.after")
|
|
786
|
+
continuation = f"{DATAGOV_V4_ENDPOINT}?" + urlencode(
|
|
787
|
+
(
|
|
788
|
+
("sort", max(DATAGOV_V4_SORT_ORDERS, key=len)),
|
|
789
|
+
("per_page", "1000"),
|
|
790
|
+
("after", value),
|
|
791
|
+
)
|
|
792
|
+
)
|
|
793
|
+
if len(continuation) > MAX_URL_LENGTH:
|
|
794
|
+
raise CatalogHarvestError(
|
|
795
|
+
"HARVEST_CURSOR",
|
|
796
|
+
"harvest.response.after",
|
|
797
|
+
"Data.gov cursor exceeds the canonical continuation URL boundary",
|
|
798
|
+
)
|