mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,445 @@
|
|
|
1
|
+
"""The SDMX harvester: read one 2.1 dataflow structure message into public-source records.
|
|
2
|
+
|
|
3
|
+
XML is the one genuinely new attack surface in the harvest package, and it is hardened in three
|
|
4
|
+
ordered steps.
|
|
5
|
+
|
|
6
|
+
**First, a byte-level pre-check, before any parser object exists.** A payload declaring
|
|
7
|
+
``<!DOCTYPE`` or ``<!ENTITY`` is refused outright. That single check closes XXE, external-entity
|
|
8
|
+
SSRF and entity-expansion (the billion-laughs family) together, because all three require a
|
|
9
|
+
document type or an entity declaration. Refusing rather than sanitising is right here: a
|
|
10
|
+
legitimate SDMX structure message has no reason to declare either, so there is nothing to lose.
|
|
11
|
+
|
|
12
|
+
The check walks the whole prolog rather than scanning a fixed-width prefix of it, and it walks it
|
|
13
|
+
in the encoding the parser will read. Both halves are load-bearing, and each was found missing in
|
|
14
|
+
turn.
|
|
15
|
+
|
|
16
|
+
A prefix scan looks equivalent to a walk and is not: comments, processing instructions and
|
|
17
|
+
whitespace may all legally precede the root element at any length, so padding past the window
|
|
18
|
+
would carry a declaration through to the parser. The walk is bounded by ``MAX_PROLOG_BYTES`` and
|
|
19
|
+
fails closed, because walking an unbounded prolog is itself attacker-controlled work.
|
|
20
|
+
|
|
21
|
+
A UTF-8-only scan looks equivalent to an encoding-aware one and is not: expat auto-detects UTF-16
|
|
22
|
+
and UTF-32 from a byte-order mark or from the byte spelling of the mandatory leading ``<``, so a
|
|
23
|
+
declaration written in either was invisible to a search for the ASCII pattern and reached the
|
|
24
|
+
parser unseen. The payload is therefore normalised to UTF-8 before the walk, by the parser's own
|
|
25
|
+
detection rules. This was not theoretical: a UTF-16 payload well inside the byte cap defined an
|
|
26
|
+
entity that expanded to sixteen million characters, which the element budget cannot see because
|
|
27
|
+
one element with an enormous body is still one element.
|
|
28
|
+
|
|
29
|
+
Nothing behind the pre-check would catch either evasion -- ``ElementTree`` expands a defined
|
|
30
|
+
internal entity, and the bundled expat's amplification limit is a build-dependent accident rather
|
|
31
|
+
than a control this module may lean on.
|
|
32
|
+
|
|
33
|
+
**Second, an element budget enforced during a pull parse.** ``MAX_XML_ELEMENTS`` stops work in
|
|
34
|
+
progress. A one-shot ``fromstring`` would build the entire tree before anyone could count it,
|
|
35
|
+
which turns a size-capped response into an unbounded allocation.
|
|
36
|
+
|
|
37
|
+
**Third, typed failure.** A raw ``ParseError`` escaping a trust boundary is an untyped failure at
|
|
38
|
+
exactly the place a caller needs to distinguish "this endpoint is broken" from "this endpoint is
|
|
39
|
+
hostile", so parser errors are wrapped.
|
|
40
|
+
|
|
41
|
+
No XML dependency is added. The three checks above are what a hardening library would do for this
|
|
42
|
+
document shape, and pulling one in would change the governed dependency closure for no gain.
|
|
43
|
+
|
|
44
|
+
**Query shape.** ``dataflow/{agency}/{id}/latest`` is what this harvester expects. ``all/latest``
|
|
45
|
+
is expected to exceed the byte cap on a large provider -- the observed Eurostat response is
|
|
46
|
+
37,166,239 bytes against a 4 MiB cap. That is a usage constraint, not a defect: an agent asking a
|
|
47
|
+
statistical office for its entire structure catalogue in one request is asking the wrong question.
|
|
48
|
+
The endpoint path itself is per-provider configuration, so this module never constructs one.
|
|
49
|
+
|
|
50
|
+
**Payload formats.** A structure message describes dataflows, not payloads, so every record here
|
|
51
|
+
carries ``data_formats=()``. That is honest and it is also a scope limit worth stating plainly: an
|
|
52
|
+
SDMX record cannot be composed into a catalog entry on the record's own evidence, because an entry
|
|
53
|
+
must declare at least one format the harness can read. The author supplies them at the composition
|
|
54
|
+
seam -- ``compose_catalog_entry(..., data_formats=("csv",))`` -- and the harvested record remains
|
|
55
|
+
the cited evidence for everything else. Guessing a format here would be this module claiming a
|
|
56
|
+
capability the structure message never stated.
|
|
57
|
+
|
|
58
|
+
**Licences.** An SDMX structure message carries none, so every record has ``license_id=None``.
|
|
59
|
+
That is honest, and it maps to ``unclear`` rights, which the facts gate escalates to a human
|
|
60
|
+
rather than guessing at -- the correct outcome for a source whose terms are genuinely not
|
|
61
|
+
machine-readable.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
from __future__ import annotations
|
|
65
|
+
|
|
66
|
+
import re
|
|
67
|
+
from xml.etree.ElementTree import ParseError, XMLPullParser
|
|
68
|
+
|
|
69
|
+
from mostlyright.data_harness.sources.catalog.harvest.ckan import MAX_HARVEST_RECORDS
|
|
70
|
+
from mostlyright.data_harness.sources.catalog.harvest.protocol import (
|
|
71
|
+
MAX_RECORD_PUBLISHER,
|
|
72
|
+
MAX_RECORD_TEXT,
|
|
73
|
+
MAX_RECORD_TITLE,
|
|
74
|
+
CatalogHarvestError,
|
|
75
|
+
HarvestCursor,
|
|
76
|
+
HarvestedRecord,
|
|
77
|
+
HarvesterDescriptor,
|
|
78
|
+
HarvestPage,
|
|
79
|
+
HarvestResponseEvidence,
|
|
80
|
+
bounded_record_text,
|
|
81
|
+
harvest_limits,
|
|
82
|
+
usable_https_uri,
|
|
83
|
+
usable_record_id,
|
|
84
|
+
)
|
|
85
|
+
from mostlyright.data_harness.sources.contracts import EvidenceReference
|
|
86
|
+
|
|
87
|
+
SDMX_LIMITS = harvest_limits(
|
|
88
|
+
"application/vnd.sdmx.structure+xml",
|
|
89
|
+
max_response_bytes=4 * 1024 * 1024,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
MAX_XML_ELEMENTS = 200_000
|
|
93
|
+
|
|
94
|
+
# How much prolog the declaration pre-check will walk before refusing. A declaration must precede
|
|
95
|
+
# the root element, but the prolog itself is unbounded -- comments, processing instructions and
|
|
96
|
+
# whitespace may all be arbitrarily long -- so the walk is bounded and fails closed. A real SDMX
|
|
97
|
+
# structure message carries a prolog of a few dozen bytes.
|
|
98
|
+
MAX_PROLOG_BYTES = 64 * 1024
|
|
99
|
+
|
|
100
|
+
# Feed size for the bounded pull parse.
|
|
101
|
+
FEED_CHUNK_BYTES = 64 * 1024
|
|
102
|
+
|
|
103
|
+
SDMX_STRUCTURE_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/structure"
|
|
104
|
+
SDMX_MESSAGE_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/message"
|
|
105
|
+
SDMX_COMMON_NS = "http://www.sdmx.org/resources/sdmxml/schemas/v2_1/common"
|
|
106
|
+
XML_NS = "http://www.w3.org/XML/1998/namespace"
|
|
107
|
+
|
|
108
|
+
_DATAFLOW_TAG = f"{{{SDMX_STRUCTURE_NS}}}Dataflow"
|
|
109
|
+
_NAME_TAG = f"{{{SDMX_COMMON_NS}}}Name"
|
|
110
|
+
_DESCRIPTION_TAG = f"{{{SDMX_COMMON_NS}}}Description"
|
|
111
|
+
_LANG_ATTR = f"{{{XML_NS}}}lang"
|
|
112
|
+
|
|
113
|
+
_DECLARATION = re.compile(rb"(?i)<!\s*(doctype|entity)")
|
|
114
|
+
|
|
115
|
+
# How a conforming XML parser works out the byte order of a document before it has read the
|
|
116
|
+
# encoding declaration -- XML 1.0 appendix F, and what expat implements. Longest patterns first,
|
|
117
|
+
# because a UTF-32LE mark begins with a UTF-16LE one. The four unmarked patterns are the byte
|
|
118
|
+
# spellings of the mandatory leading `<`.
|
|
119
|
+
_AUTODETECTED_ENCODINGS = (
|
|
120
|
+
(b"\x00\x00\xfe\xff", "utf-32"),
|
|
121
|
+
(b"\xff\xfe\x00\x00", "utf-32"),
|
|
122
|
+
(b"\x00\x00\x00\x3c", "utf-32-be"),
|
|
123
|
+
(b"\x3c\x00\x00\x00", "utf-32-le"),
|
|
124
|
+
(b"\xfe\xff", "utf-16"),
|
|
125
|
+
(b"\xff\xfe", "utf-16"),
|
|
126
|
+
(b"\x00\x3c", "utf-16-be"),
|
|
127
|
+
(b"\x3c\x00", "utf-16-le"),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class SdmxHarvester:
|
|
132
|
+
"""Read an SDMX 2.1 dataflow structure message. Takes bytes; never fetches."""
|
|
133
|
+
|
|
134
|
+
_DESCRIPTOR = HarvesterDescriptor(
|
|
135
|
+
protocol="sdmx",
|
|
136
|
+
harvester_id="sdmx.dataflow",
|
|
137
|
+
harvester_version="1.0.0",
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
@property
|
|
141
|
+
def descriptor(self) -> HarvesterDescriptor:
|
|
142
|
+
return self._DESCRIPTOR
|
|
143
|
+
|
|
144
|
+
def parse(
|
|
145
|
+
self,
|
|
146
|
+
payload: bytes,
|
|
147
|
+
*,
|
|
148
|
+
uri: str,
|
|
149
|
+
observed_at: str,
|
|
150
|
+
evidence: EvidenceReference,
|
|
151
|
+
) -> tuple[HarvestedRecord, ...]:
|
|
152
|
+
page = self.parse_page(
|
|
153
|
+
payload,
|
|
154
|
+
uri=uri,
|
|
155
|
+
observed_at=observed_at,
|
|
156
|
+
evidence=evidence,
|
|
157
|
+
response_evidence=None,
|
|
158
|
+
cursor=None,
|
|
159
|
+
)
|
|
160
|
+
if not page.records:
|
|
161
|
+
raise CatalogHarvestError(
|
|
162
|
+
"HARVEST_SHAPE", "sdmx.Structures.Dataflows", "message carries no dataflow"
|
|
163
|
+
)
|
|
164
|
+
return page.records
|
|
165
|
+
|
|
166
|
+
def parse_page(
|
|
167
|
+
self,
|
|
168
|
+
payload: bytes,
|
|
169
|
+
*,
|
|
170
|
+
uri: str,
|
|
171
|
+
observed_at: str,
|
|
172
|
+
evidence: EvidenceReference,
|
|
173
|
+
response_evidence: HarvestResponseEvidence | None,
|
|
174
|
+
cursor: HarvestCursor | None,
|
|
175
|
+
) -> HarvestPage:
|
|
176
|
+
if cursor is not None:
|
|
177
|
+
raise CatalogHarvestError(
|
|
178
|
+
"HARVEST_CURSOR", "sdmx.cursor", "SDMX dataflow harvest is one-shot"
|
|
179
|
+
)
|
|
180
|
+
if not isinstance(payload, bytes):
|
|
181
|
+
raise CatalogHarvestError("TYPE", "sdmx.payload", "payload must be bytes")
|
|
182
|
+
require_no_xml_declarations(payload)
|
|
183
|
+
root = _parse_bounded(payload)
|
|
184
|
+
dataflows = [element for element in root.iter() if element.tag == _DATAFLOW_TAG]
|
|
185
|
+
if len(dataflows) > MAX_HARVEST_RECORDS:
|
|
186
|
+
raise CatalogHarvestError(
|
|
187
|
+
"HARVEST_RECORD_LIMIT",
|
|
188
|
+
"sdmx.Structures.Dataflows",
|
|
189
|
+
f"a single message may not carry more than {MAX_HARVEST_RECORDS} dataflows",
|
|
190
|
+
)
|
|
191
|
+
records: list[HarvestedRecord] = []
|
|
192
|
+
skipped: list[str] = []
|
|
193
|
+
for position, dataflow in enumerate(dataflows):
|
|
194
|
+
record = _record(
|
|
195
|
+
dataflow,
|
|
196
|
+
position=position,
|
|
197
|
+
uri=uri,
|
|
198
|
+
evidence=evidence,
|
|
199
|
+
skipped=skipped,
|
|
200
|
+
)
|
|
201
|
+
if record is not None:
|
|
202
|
+
records.append(record)
|
|
203
|
+
return HarvestPage(
|
|
204
|
+
protocol="sdmx",
|
|
205
|
+
records=tuple(records),
|
|
206
|
+
next_cursor=None,
|
|
207
|
+
provider_count=len(dataflows),
|
|
208
|
+
count_basis="one_shot_response",
|
|
209
|
+
skipped_record_ids=tuple(sorted(skipped)),
|
|
210
|
+
evidence=evidence,
|
|
211
|
+
response_evidence=response_evidence,
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def require_no_xml_declarations(payload: bytes) -> None:
|
|
216
|
+
"""Refuse a payload declaring a document type or an entity, before any parser exists.
|
|
217
|
+
|
|
218
|
+
The whole prolog is walked, not a fixed-width prefix of it. A prefix scan is an evadable
|
|
219
|
+
control: comments, processing instructions and whitespace are all legal before the root element
|
|
220
|
+
and all three can be padded to any length, so a declaration hidden behind enough padding would
|
|
221
|
+
reach the parser -- and ElementTree does expand a defined internal entity, so the parser is not
|
|
222
|
+
a fallback control. The walk consumes exactly the constructs that may legally precede the root
|
|
223
|
+
element and stops at the first thing that is not one of them.
|
|
224
|
+
|
|
225
|
+
The walk runs over UTF-8 bytes, whichever byte order the sender chose. expat auto-detects
|
|
226
|
+
UTF-16 and UTF-32 from a byte-order mark or from the first character's byte pattern, so a
|
|
227
|
+
declaration spelled in either was invisible to a scan for the ASCII pattern and reached the
|
|
228
|
+
parser unseen -- a walk that only reads one encoding is an evadable control for the same
|
|
229
|
+
reason a prefix scan is. Normalising rather than refusing keeps the refusal in one place and
|
|
230
|
+
keeps the reported code the same whatever the encoding.
|
|
231
|
+
"""
|
|
232
|
+
|
|
233
|
+
payload = _utf8_normalised(payload)
|
|
234
|
+
cursor = _skip_bom_and_space(payload, 0)
|
|
235
|
+
while cursor < len(payload):
|
|
236
|
+
if cursor > MAX_PROLOG_BYTES:
|
|
237
|
+
raise _declaration_refused(
|
|
238
|
+
f"the prolog exceeds {MAX_PROLOG_BYTES} bytes before any element begins, so it "
|
|
239
|
+
"cannot be cleared of declarations within its budget"
|
|
240
|
+
)
|
|
241
|
+
window = payload[cursor : cursor + 32]
|
|
242
|
+
match = _DECLARATION.match(window)
|
|
243
|
+
if match is not None:
|
|
244
|
+
raise _declaration_refused(
|
|
245
|
+
f"the document declares a {match.group(1).decode('ascii').lower()}; document type "
|
|
246
|
+
"and entity declarations are refused outright, because an SDMX structure message "
|
|
247
|
+
"has no reason to carry one"
|
|
248
|
+
)
|
|
249
|
+
if window.startswith(b"<!--"):
|
|
250
|
+
end = payload.find(b"-->", cursor + 4)
|
|
251
|
+
if end < 0:
|
|
252
|
+
# Unterminated: nothing can follow it, so no declaration can be hiding behind it.
|
|
253
|
+
return
|
|
254
|
+
cursor = _skip_bom_and_space(payload, end + 3)
|
|
255
|
+
continue
|
|
256
|
+
if window.startswith(b"<?"):
|
|
257
|
+
end = payload.find(b"?>", cursor + 2)
|
|
258
|
+
if end < 0:
|
|
259
|
+
return
|
|
260
|
+
cursor = _skip_bom_and_space(payload, end + 2)
|
|
261
|
+
continue
|
|
262
|
+
# Anything else is the root element or malformed input; either way the prolog is over and
|
|
263
|
+
# no declaration may legally appear past this point.
|
|
264
|
+
return
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _utf8_normalised(payload: bytes) -> bytes:
|
|
268
|
+
"""Return the payload as UTF-8 bytes, whatever byte order it arrived in.
|
|
269
|
+
|
|
270
|
+
The detection rules are the parser's own -- XML 1.0 appendix F, which is what expat implements:
|
|
271
|
+
a byte-order mark if there is one, otherwise the byte pattern of the mandatory leading ``<``.
|
|
272
|
+
Reading it any other way would leave a gap between what this pre-check inspects and what the
|
|
273
|
+
parser will act on, which is precisely the defect this closes.
|
|
274
|
+
|
|
275
|
+
A payload that announces one of these encodings and then is not valid in it is refused rather
|
|
276
|
+
than passed along, because at that point nothing can say what the parser would make of it.
|
|
277
|
+
"""
|
|
278
|
+
|
|
279
|
+
encoding = _autodetected_encoding(payload)
|
|
280
|
+
if encoding is None:
|
|
281
|
+
return payload
|
|
282
|
+
try:
|
|
283
|
+
text = payload.decode(encoding)
|
|
284
|
+
except UnicodeDecodeError:
|
|
285
|
+
raise CatalogHarvestError(
|
|
286
|
+
"HARVEST_XML_ENCODING",
|
|
287
|
+
"sdmx.payload",
|
|
288
|
+
f"the payload announces {encoding} and is not valid in it, so what the parser would "
|
|
289
|
+
"make of it cannot be established before parsing it",
|
|
290
|
+
) from None
|
|
291
|
+
return text.encode("utf-8")
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _autodetected_encoding(payload: bytes) -> str | None:
|
|
295
|
+
"""The encoding a conforming XML parser would detect from the first four bytes.
|
|
296
|
+
|
|
297
|
+
``None`` means UTF-8 or another ASCII-compatible single-byte encoding, for which the ASCII
|
|
298
|
+
declaration pattern is already directly readable in the bytes. The byte-order-mark codecs are
|
|
299
|
+
named without an endianness suffix on purpose: those forms consume the mark, so the decoded
|
|
300
|
+
text starts at the first real character.
|
|
301
|
+
"""
|
|
302
|
+
|
|
303
|
+
prefix = payload[:4]
|
|
304
|
+
for pattern, encoding in _AUTODETECTED_ENCODINGS:
|
|
305
|
+
if prefix.startswith(pattern):
|
|
306
|
+
return encoding
|
|
307
|
+
return None
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _skip_bom_and_space(payload: bytes, cursor: int) -> int:
|
|
311
|
+
if cursor == 0 and payload.startswith(b"\xef\xbb\xbf"):
|
|
312
|
+
cursor = 3
|
|
313
|
+
while cursor < len(payload) and payload[cursor : cursor + 1].isspace():
|
|
314
|
+
cursor += 1
|
|
315
|
+
return cursor
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _declaration_refused(detail: str) -> CatalogHarvestError:
|
|
319
|
+
return CatalogHarvestError("HARVEST_XML_DOCTYPE", "sdmx.payload", detail)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _parse_bounded(payload: bytes): # type: ignore[no-untyped-def]
|
|
323
|
+
# Fed in chunks and drained between them, so the budget stops work in progress rather than
|
|
324
|
+
# after the whole tree has already been built.
|
|
325
|
+
parser = XMLPullParser(events=("start",))
|
|
326
|
+
root = None
|
|
327
|
+
elements = 0
|
|
328
|
+
try:
|
|
329
|
+
for offset in range(0, len(payload), FEED_CHUNK_BYTES):
|
|
330
|
+
parser.feed(payload[offset : offset + FEED_CHUNK_BYTES])
|
|
331
|
+
for _event, element in parser.read_events():
|
|
332
|
+
elements += 1
|
|
333
|
+
if root is None:
|
|
334
|
+
root = element
|
|
335
|
+
if elements > MAX_XML_ELEMENTS:
|
|
336
|
+
raise CatalogHarvestError(
|
|
337
|
+
"HARVEST_XML_ELEMENTS",
|
|
338
|
+
"sdmx.payload",
|
|
339
|
+
f"the document exceeds {MAX_XML_ELEMENTS} elements",
|
|
340
|
+
)
|
|
341
|
+
parser.close()
|
|
342
|
+
for _event, element in parser.read_events():
|
|
343
|
+
elements += 1
|
|
344
|
+
if root is None:
|
|
345
|
+
root = element
|
|
346
|
+
except ParseError as error:
|
|
347
|
+
raise CatalogHarvestError(
|
|
348
|
+
"HARVEST_SHAPE",
|
|
349
|
+
"sdmx.payload",
|
|
350
|
+
f"the response is not well-formed XML: {error}",
|
|
351
|
+
) from None
|
|
352
|
+
if root is None:
|
|
353
|
+
raise CatalogHarvestError(
|
|
354
|
+
"HARVEST_SHAPE",
|
|
355
|
+
"sdmx.payload",
|
|
356
|
+
"the response carries no XML element",
|
|
357
|
+
)
|
|
358
|
+
return root
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _record(
|
|
362
|
+
dataflow, # type: ignore[no-untyped-def]
|
|
363
|
+
*,
|
|
364
|
+
position: int,
|
|
365
|
+
uri: str,
|
|
366
|
+
evidence: EvidenceReference,
|
|
367
|
+
skipped: list[str],
|
|
368
|
+
) -> HarvestedRecord | None:
|
|
369
|
+
flow_id = _attribute(dataflow, "id")
|
|
370
|
+
agency = _attribute(dataflow, "agencyID")
|
|
371
|
+
version = _attribute(dataflow, "version") or "1.0"
|
|
372
|
+
if flow_id is None or agency is None:
|
|
373
|
+
skipped.append(f"<dataflow {position} without id or agencyID>")
|
|
374
|
+
return None
|
|
375
|
+
# The dataflow's natural coordinate is an identifier, so it degrades this record rather than
|
|
376
|
+
# being cut: an agency, id or version outside the record grammar would raise out of `parse` and
|
|
377
|
+
# take every other dataflow in the message with it.
|
|
378
|
+
record_id = usable_record_id(f"{agency}:{flow_id}({version})")
|
|
379
|
+
if record_id is None:
|
|
380
|
+
skipped.append(f"<dataflow {position} whose coordinate is not a usable record id>")
|
|
381
|
+
return None
|
|
382
|
+
landing = usable_https_uri(_structure_uri(uri, agency, flow_id, version))
|
|
383
|
+
if landing is None:
|
|
384
|
+
skipped.append(record_id)
|
|
385
|
+
return None
|
|
386
|
+
name = _localised(dataflow, _NAME_TAG) or flow_id
|
|
387
|
+
description = _localised(dataflow, _DESCRIPTION_TAG) or (
|
|
388
|
+
f"SDMX dataflow {agency}:{flow_id}({version}) published as a 2.1 structure message."
|
|
389
|
+
)
|
|
390
|
+
return HarvestedRecord(
|
|
391
|
+
protocol="sdmx",
|
|
392
|
+
record_id=record_id,
|
|
393
|
+
title=bounded_record_text(name, maximum=MAX_RECORD_TITLE),
|
|
394
|
+
description=bounded_record_text(description, maximum=MAX_RECORD_TEXT),
|
|
395
|
+
publisher=bounded_record_text(agency, maximum=MAX_RECORD_PUBLISHER),
|
|
396
|
+
landing_uri=landing,
|
|
397
|
+
data_formats=(),
|
|
398
|
+
evidence=evidence,
|
|
399
|
+
# A structure message states no licence. Saying so honestly maps to unclear rights, which
|
|
400
|
+
# escalates rather than admits.
|
|
401
|
+
license_id=None,
|
|
402
|
+
license_uri=None,
|
|
403
|
+
declared_updated_at=None,
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _structure_uri(uri: str, agency: str, flow_id: str, version: str) -> str | None:
|
|
408
|
+
if not uri.startswith("https://"):
|
|
409
|
+
return None
|
|
410
|
+
base = uri.split("?", 1)[0].rstrip("/")
|
|
411
|
+
marker = "/dataflow"
|
|
412
|
+
if marker in base:
|
|
413
|
+
base = base[: base.index(marker) + len(marker)]
|
|
414
|
+
return f"{base}/{agency}/{flow_id}/{version}"
|
|
415
|
+
return base
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _attribute(element, name: str) -> str | None: # type: ignore[no-untyped-def]
|
|
419
|
+
value = element.get(name)
|
|
420
|
+
if not isinstance(value, str):
|
|
421
|
+
return None
|
|
422
|
+
text = value.strip()
|
|
423
|
+
return text or None
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _localised(element, tag: str) -> str | None: # type: ignore[no-untyped-def]
|
|
427
|
+
"""The English text of this child, or the first declared one, deterministically.
|
|
428
|
+
|
|
429
|
+
Falling back to "whichever came out of the iterator" would make the harvested title depend on
|
|
430
|
+
document order in a way nobody documented, so the fallback is explicit and ordered.
|
|
431
|
+
"""
|
|
432
|
+
|
|
433
|
+
candidates = [child for child in element if child.tag == tag]
|
|
434
|
+
if not candidates:
|
|
435
|
+
return None
|
|
436
|
+
for child in candidates:
|
|
437
|
+
if (child.get(_LANG_ATTR) or "").lower().startswith("en"):
|
|
438
|
+
text = (child.text or "").strip()
|
|
439
|
+
if text:
|
|
440
|
+
return text
|
|
441
|
+
for child in candidates:
|
|
442
|
+
text = (child.text or "").strip()
|
|
443
|
+
if text:
|
|
444
|
+
return text
|
|
445
|
+
return None
|