mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,3889 @@
|
|
|
1
|
+
"""Bounded, resumable local fills over the existing catalog harvesters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import sqlite3
|
|
7
|
+
import tempfile
|
|
8
|
+
import time
|
|
9
|
+
from collections.abc import Callable, Iterator
|
|
10
|
+
from dataclasses import dataclass, replace
|
|
11
|
+
from itertools import islice
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
|
15
|
+
|
|
16
|
+
from mostlyright.data_harness.acquisition.http import (
|
|
17
|
+
DATAGOV_V4_AUTHENTICATION_KIND,
|
|
18
|
+
DatagovRateObservation,
|
|
19
|
+
DatagovV4Authorization,
|
|
20
|
+
PeerAttemptError,
|
|
21
|
+
)
|
|
22
|
+
from mostlyright.data_harness.acquisition.url_policy import AcquisitionSecurityError
|
|
23
|
+
from mostlyright.data_harness.canonical import (
|
|
24
|
+
CanonicalJSONError,
|
|
25
|
+
canonical_json_bytes,
|
|
26
|
+
canonical_sha256,
|
|
27
|
+
sha256_bytes,
|
|
28
|
+
)
|
|
29
|
+
from mostlyright.data_harness.sources.catalog.fill_partitions import (
|
|
30
|
+
dry_run_external_merge_pass,
|
|
31
|
+
external_merge_pass,
|
|
32
|
+
plan_partition_runs,
|
|
33
|
+
read_partition_run,
|
|
34
|
+
validate_pass_map,
|
|
35
|
+
write_partition_runs,
|
|
36
|
+
)
|
|
37
|
+
from mostlyright.data_harness.sources.catalog.fill_staging import (
|
|
38
|
+
SHARD_SCHEMA,
|
|
39
|
+
CatalogFillRefused,
|
|
40
|
+
FillStaging,
|
|
41
|
+
shard_byte_limit,
|
|
42
|
+
)
|
|
43
|
+
from mostlyright.data_harness.sources.catalog.harvest.ckan import (
|
|
44
|
+
CKAN_LIMITS,
|
|
45
|
+
CkanHarvester,
|
|
46
|
+
ckan_licence_rights,
|
|
47
|
+
)
|
|
48
|
+
from mostlyright.data_harness.sources.catalog.harvest.datagov_v4 import (
|
|
49
|
+
DATAGOV_V4_ENDPOINT,
|
|
50
|
+
DATAGOV_V4_LIMITS,
|
|
51
|
+
DatagovProjectedField,
|
|
52
|
+
DatagovV4Harvester,
|
|
53
|
+
DatagovV4Record,
|
|
54
|
+
)
|
|
55
|
+
from mostlyright.data_harness.sources.catalog.harvest.protocol import (
|
|
56
|
+
CatalogHarvestError,
|
|
57
|
+
DatagovFetchRefused,
|
|
58
|
+
DatagovRefusedResponseEvidence,
|
|
59
|
+
HarvestCursor,
|
|
60
|
+
HarvestedRecord,
|
|
61
|
+
HarvestPage,
|
|
62
|
+
HarvestResponseEvidence,
|
|
63
|
+
SourceHarvester,
|
|
64
|
+
fetch_datagov_v4_response,
|
|
65
|
+
fetch_harvest_response,
|
|
66
|
+
)
|
|
67
|
+
from mostlyright.data_harness.sources.catalog.harvest.sdmx import SDMX_LIMITS, SdmxHarvester
|
|
68
|
+
from mostlyright.data_harness.sources.catalog.harvest.stac import (
|
|
69
|
+
STAC_LIMITS,
|
|
70
|
+
StacHarvester,
|
|
71
|
+
stac_licence_rights,
|
|
72
|
+
)
|
|
73
|
+
from mostlyright.data_harness.sources.contracts import EvidenceReference
|
|
74
|
+
|
|
75
|
+
FILL_CONFIG_SCHEMA = "harness-catalog-fill-config.v1"
|
|
76
|
+
FILL_RECEIPT_SCHEMA = "harness-catalog-fill-receipt.v1"
|
|
77
|
+
DATAGOV_V4_CHECKPOINT_SCHEMA = "harness-datagov-v4-fill-checkpoint.v8"
|
|
78
|
+
|
|
79
|
+
#: The provider sorts this cursor contract admits. ``last_harvested_date`` orders by the one
|
|
80
|
+
#: field data.gov rewrites every time it re-harvests a dataset, so a record can move ahead of a
|
|
81
|
+
#: cursor that already passed it and never be seen; ``relevance`` orders by the constant
|
|
82
|
+
#: match-all score, then popularity, then the record's stable UUID.
|
|
83
|
+
DATAGOV_V4_SORT_ORDERS = ("last_harvested_date", "relevance")
|
|
84
|
+
DATAGOV_V4_DEFAULT_SORT = "last_harvested_date"
|
|
85
|
+
DATAGOV_V4_MAX_NORMALIZED_PAGE_BYTES = 64 * 1024 * 1024
|
|
86
|
+
DATAGOV_V4_MAX_NORMALIZED_EXPANSION = 7
|
|
87
|
+
DATAGOV_V4_LOOP_SET_SCHEMA = "harness-datagov-v4-loop-set.v1"
|
|
88
|
+
DATAGOV_V4_LOOP_SET_LEAF_ITEMS = 512
|
|
89
|
+
_V4_VALIDATED_CACHE_LIMIT = 32
|
|
90
|
+
_MAX_SAFE_EPOCH_SECONDS = (1 << 53) - 1
|
|
91
|
+
_V4_POST_RESPONSE_OVERRUN_REASONS = frozenset(
|
|
92
|
+
{
|
|
93
|
+
"FILL_RESPONSE_BYTE_LIMIT",
|
|
94
|
+
"FILL_AGGREGATE_BYTE_LIMIT",
|
|
95
|
+
"FILL_WALL_TIME_LIMIT",
|
|
96
|
+
}
|
|
97
|
+
)
|
|
98
|
+
_V4_PREFLIGHT_TERMINAL_REASONS = frozenset(
|
|
99
|
+
{
|
|
100
|
+
"FILL_REQUEST_LIMIT",
|
|
101
|
+
"FILL_RESPONSE_LIMIT",
|
|
102
|
+
"FILL_PAGE_LIMIT",
|
|
103
|
+
"FILL_RECORD_LIMIT",
|
|
104
|
+
"FILL_AGGREGATE_BYTE_LIMIT",
|
|
105
|
+
"FILL_WALL_TIME_LIMIT",
|
|
106
|
+
}
|
|
107
|
+
)
|
|
108
|
+
#: Terminal reasons that still seal a final map, so the corpus they name is real.
|
|
109
|
+
_V4_MAP_SEALING_REASONS = frozenset({"FILL_NONCONVERGENT", "FILL_CONFLICTING_IDENTIFIERS"})
|
|
110
|
+
_V4_IDEMPOTENT_TERMINAL_REASONS = (
|
|
111
|
+
_V4_POST_RESPONSE_OVERRUN_REASONS
|
|
112
|
+
| _V4_PREFLIGHT_TERMINAL_REASONS
|
|
113
|
+
| {
|
|
114
|
+
"FILL_ENDPOINT_DRIFT",
|
|
115
|
+
"FILL_NONCONVERGENT",
|
|
116
|
+
"FILL_CONFLICTING_IDENTIFIERS",
|
|
117
|
+
}
|
|
118
|
+
)
|
|
119
|
+
_V4_EVIDENCE_STREAMS = (
|
|
120
|
+
"request",
|
|
121
|
+
"response",
|
|
122
|
+
"page",
|
|
123
|
+
"record",
|
|
124
|
+
"cursor_transition",
|
|
125
|
+
)
|
|
126
|
+
_SHA256 = re.compile(r"^[0-9a-f]{64}$")
|
|
127
|
+
_TIMESTAMP = re.compile(
|
|
128
|
+
r"^[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]{1,9})?Z$"
|
|
129
|
+
)
|
|
130
|
+
_V4_VALIDATED_CHECKPOINTS: dict[str, tuple[str, tuple[tuple[str, int, int, int, int], ...]]] = {}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@dataclass(frozen=True)
|
|
134
|
+
class CatalogFillLimits:
|
|
135
|
+
max_requests: int
|
|
136
|
+
max_pages: int
|
|
137
|
+
max_records: int
|
|
138
|
+
max_response_bytes: int
|
|
139
|
+
max_aggregate_bytes: int
|
|
140
|
+
max_responses: int
|
|
141
|
+
max_wall_time_ns: int
|
|
142
|
+
|
|
143
|
+
def __post_init__(self) -> None:
|
|
144
|
+
maxima = {
|
|
145
|
+
"max_requests": 10_000_000,
|
|
146
|
+
"max_pages": 1_000_000,
|
|
147
|
+
"max_records": 10_000_000,
|
|
148
|
+
"max_response_bytes": 64 * 1024 * 1024,
|
|
149
|
+
"max_aggregate_bytes": 1 << 50,
|
|
150
|
+
"max_responses": 10_000_000,
|
|
151
|
+
"max_wall_time_ns": 30 * 24 * 60 * 60 * 1_000_000_000,
|
|
152
|
+
}
|
|
153
|
+
for name, maximum in maxima.items():
|
|
154
|
+
value = getattr(self, name)
|
|
155
|
+
if type(value) is not int or not 1 <= value <= maximum:
|
|
156
|
+
raise CatalogFillRefused(
|
|
157
|
+
"FILL_LIMIT", f"fill.limits.{name}", f"must be in [1, {maximum}]"
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def to_dict(self) -> dict[str, int]:
|
|
161
|
+
return {
|
|
162
|
+
"max_requests": self.max_requests,
|
|
163
|
+
"max_pages": self.max_pages,
|
|
164
|
+
"max_records": self.max_records,
|
|
165
|
+
"max_response_bytes": self.max_response_bytes,
|
|
166
|
+
"max_aggregate_bytes": self.max_aggregate_bytes,
|
|
167
|
+
"max_responses": self.max_responses,
|
|
168
|
+
"max_wall_time_ns": self.max_wall_time_ns,
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@dataclass(frozen=True)
|
|
173
|
+
class CatalogFillConfig:
|
|
174
|
+
protocol: str
|
|
175
|
+
endpoint: str
|
|
176
|
+
observed_at: str
|
|
177
|
+
page_size: int
|
|
178
|
+
max_convergence_passes: int
|
|
179
|
+
limits: CatalogFillLimits
|
|
180
|
+
authentication_kind: str | None = None
|
|
181
|
+
minimum_request_interval_ns: int = 0
|
|
182
|
+
sort: str = DATAGOV_V4_DEFAULT_SORT
|
|
183
|
+
max_conflicting_identifiers: int = 0
|
|
184
|
+
harvester_id: str | None = None
|
|
185
|
+
harvester_version: str = "1.0.0"
|
|
186
|
+
predecessor_sha256: str | None = None
|
|
187
|
+
schema_version: str = FILL_CONFIG_SCHEMA
|
|
188
|
+
|
|
189
|
+
def __post_init__(self) -> None:
|
|
190
|
+
if self.schema_version != FILL_CONFIG_SCHEMA:
|
|
191
|
+
raise CatalogFillRefused("FILL_CONFIG_VERSION", "fill.config", "unsupported version")
|
|
192
|
+
if self.protocol not in {"ckan", "datagov_v4", "stac", "sdmx"}:
|
|
193
|
+
raise CatalogFillRefused("FILL_PROTOCOL", "fill.protocol", "unsupported protocol")
|
|
194
|
+
parsed = urlsplit(self.endpoint)
|
|
195
|
+
if (
|
|
196
|
+
parsed.scheme != "https"
|
|
197
|
+
or not parsed.hostname
|
|
198
|
+
or parsed.username is not None
|
|
199
|
+
or parsed.password is not None
|
|
200
|
+
or parsed.query
|
|
201
|
+
or parsed.fragment
|
|
202
|
+
):
|
|
203
|
+
raise CatalogFillRefused(
|
|
204
|
+
"FILL_ENDPOINT",
|
|
205
|
+
"fill.endpoint",
|
|
206
|
+
"endpoint must be a credential-free public HTTPS base URL without query or "
|
|
207
|
+
"fragment",
|
|
208
|
+
)
|
|
209
|
+
if not isinstance(self.limits, CatalogFillLimits):
|
|
210
|
+
raise CatalogFillRefused("FILL_LIMIT", "fill.limits", "strict limits are required")
|
|
211
|
+
if self.protocol == "datagov_v4":
|
|
212
|
+
if self.endpoint != DATAGOV_V4_ENDPOINT:
|
|
213
|
+
raise CatalogFillRefused(
|
|
214
|
+
"DATAGOV_ENDPOINT",
|
|
215
|
+
"fill.endpoint",
|
|
216
|
+
"Data.gov v4 uses the exact official search endpoint",
|
|
217
|
+
)
|
|
218
|
+
if self.authentication_kind != DATAGOV_V4_AUTHENTICATION_KIND:
|
|
219
|
+
raise CatalogFillRefused(
|
|
220
|
+
"DATAGOV_AUTHENTICATION",
|
|
221
|
+
"fill.authentication_kind",
|
|
222
|
+
"Data.gov v4 requires the closed key-file authentication kind",
|
|
223
|
+
)
|
|
224
|
+
if self.limits.max_response_bytes != 16 * 1024 * 1024:
|
|
225
|
+
raise CatalogFillRefused(
|
|
226
|
+
"DATAGOV_RESPONSE_LIMIT",
|
|
227
|
+
"fill.limits.max_response_bytes",
|
|
228
|
+
"Data.gov v4 uses the exact 16 MiB response cap",
|
|
229
|
+
)
|
|
230
|
+
if self.sort not in DATAGOV_V4_SORT_ORDERS:
|
|
231
|
+
raise CatalogFillRefused(
|
|
232
|
+
"DATAGOV_SORT",
|
|
233
|
+
"fill.sort",
|
|
234
|
+
"Data.gov v4 admits exactly the sorts its cursor contract can page",
|
|
235
|
+
)
|
|
236
|
+
if (
|
|
237
|
+
type(self.max_conflicting_identifiers) is not int
|
|
238
|
+
or self.max_conflicting_identifiers < 0
|
|
239
|
+
):
|
|
240
|
+
raise CatalogFillRefused(
|
|
241
|
+
"DATAGOV_CONFLICT_BOUND",
|
|
242
|
+
"fill.max_conflicting_identifiers",
|
|
243
|
+
"the admitted conflicting-identifier count must be a non-negative integer",
|
|
244
|
+
)
|
|
245
|
+
elif (
|
|
246
|
+
self.authentication_kind is not None
|
|
247
|
+
or self.minimum_request_interval_ns != 0
|
|
248
|
+
or self.sort != DATAGOV_V4_DEFAULT_SORT
|
|
249
|
+
or self.max_conflicting_identifiers != 0
|
|
250
|
+
):
|
|
251
|
+
raise CatalogFillRefused(
|
|
252
|
+
"FILL_AUTHENTICATION",
|
|
253
|
+
"fill.authentication_kind",
|
|
254
|
+
"legacy public protocols remain credential-free",
|
|
255
|
+
)
|
|
256
|
+
if (
|
|
257
|
+
type(self.minimum_request_interval_ns) is not int
|
|
258
|
+
or not 0 <= self.minimum_request_interval_ns <= 3_600_000_000_000
|
|
259
|
+
):
|
|
260
|
+
raise CatalogFillRefused(
|
|
261
|
+
"DATAGOV_RATE",
|
|
262
|
+
"fill.minimum_request_interval_ns",
|
|
263
|
+
"minimum interval must be in [0, 3600 seconds]",
|
|
264
|
+
)
|
|
265
|
+
if type(self.page_size) is not int or not 1 <= self.page_size <= 1_000:
|
|
266
|
+
raise CatalogFillRefused("FILL_PAGE_SIZE", "fill.page_size", "must be in [1, 1000]")
|
|
267
|
+
if not isinstance(self.observed_at, str) or not _TIMESTAMP.fullmatch(self.observed_at):
|
|
268
|
+
raise CatalogFillRefused(
|
|
269
|
+
"FILL_OBSERVED_AT", "fill.observed_at", "must be a canonical UTC timestamp"
|
|
270
|
+
)
|
|
271
|
+
if (
|
|
272
|
+
type(self.max_convergence_passes) is not int
|
|
273
|
+
or not 1 <= self.max_convergence_passes <= 100
|
|
274
|
+
):
|
|
275
|
+
raise CatalogFillRefused(
|
|
276
|
+
"FILL_CONVERGENCE_LIMIT",
|
|
277
|
+
"fill.max_convergence_passes",
|
|
278
|
+
"must be in [1, 100]",
|
|
279
|
+
)
|
|
280
|
+
expected = _protocol_parts(self.protocol)[1].descriptor.harvester_id
|
|
281
|
+
if self.harvester_id is not None and self.harvester_id != expected:
|
|
282
|
+
raise CatalogFillRefused(
|
|
283
|
+
"FILL_HARVESTER", "fill.harvester_id", "harvester does not match protocol"
|
|
284
|
+
)
|
|
285
|
+
if self.harvester_version != "1.0.0":
|
|
286
|
+
raise CatalogFillRefused(
|
|
287
|
+
"FILL_HARVESTER", "fill.harvester_version", "harvester version is unavailable"
|
|
288
|
+
)
|
|
289
|
+
if self.predecessor_sha256 is not None and not _SHA256.fullmatch(self.predecessor_sha256):
|
|
290
|
+
raise CatalogFillRefused(
|
|
291
|
+
"FILL_PREDECESSOR", "fill.predecessor_sha256", "must be SHA-256 or null"
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
@property
|
|
295
|
+
def coordinate(self) -> str:
|
|
296
|
+
identifier = self.harvester_id or _protocol_parts(self.protocol)[1].descriptor.harvester_id
|
|
297
|
+
return f"{identifier}@{self.harvester_version}"
|
|
298
|
+
|
|
299
|
+
@property
|
|
300
|
+
def digest(self) -> str:
|
|
301
|
+
return canonical_sha256(self.to_dict())
|
|
302
|
+
|
|
303
|
+
def to_dict(self) -> dict[str, Any]:
|
|
304
|
+
result = {
|
|
305
|
+
"schema_version": self.schema_version,
|
|
306
|
+
"protocol": self.protocol,
|
|
307
|
+
"endpoint": self.endpoint,
|
|
308
|
+
"observed_at": self.observed_at,
|
|
309
|
+
"page_size": self.page_size,
|
|
310
|
+
"max_convergence_passes": self.max_convergence_passes,
|
|
311
|
+
"limits": self.limits.to_dict(),
|
|
312
|
+
"harvester_id": self.harvester_id
|
|
313
|
+
or _protocol_parts(self.protocol)[1].descriptor.harvester_id,
|
|
314
|
+
"harvester_version": self.harvester_version,
|
|
315
|
+
"predecessor_sha256": self.predecessor_sha256,
|
|
316
|
+
}
|
|
317
|
+
if self.protocol == "datagov_v4":
|
|
318
|
+
result["authentication_kind"] = self.authentication_kind
|
|
319
|
+
result["minimum_request_interval_ns"] = self.minimum_request_interval_ns
|
|
320
|
+
result["sort"] = self.sort
|
|
321
|
+
result["max_conflicting_identifiers"] = self.max_conflicting_identifiers
|
|
322
|
+
return result
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
@dataclass(frozen=True)
|
|
326
|
+
class CatalogFillReceipt:
|
|
327
|
+
config_sha256: str
|
|
328
|
+
endpoint: str
|
|
329
|
+
harvester_coordinate: str
|
|
330
|
+
predecessor_sha256: str | None
|
|
331
|
+
status: str
|
|
332
|
+
reason_code: str | None
|
|
333
|
+
requests: int
|
|
334
|
+
responses: int
|
|
335
|
+
pages: int
|
|
336
|
+
records: int
|
|
337
|
+
unique_records: int
|
|
338
|
+
network_bytes: int
|
|
339
|
+
elapsed_ns: int
|
|
340
|
+
provider_count: int | None
|
|
341
|
+
count_basis: str | None
|
|
342
|
+
skipped_records: int
|
|
343
|
+
flagged_rights: int
|
|
344
|
+
convergence_passes: int
|
|
345
|
+
response_evidence_digests: tuple[str, ...]
|
|
346
|
+
page_shard_digests: tuple[str, ...]
|
|
347
|
+
flagged_queue_sha256: str
|
|
348
|
+
rate_observation: dict[str, int | None] | None = None
|
|
349
|
+
resume_not_before_epoch_seconds: int | None = None
|
|
350
|
+
evidence_indexes: dict[str, dict[str, Any]] | None = None
|
|
351
|
+
pass_map_sha256: str | None = None
|
|
352
|
+
normalized_bytes: int | None = None
|
|
353
|
+
conflicting_identifiers: int | None = None
|
|
354
|
+
schema_version: str = FILL_RECEIPT_SCHEMA
|
|
355
|
+
|
|
356
|
+
@property
|
|
357
|
+
def digest(self) -> str:
|
|
358
|
+
return canonical_sha256(self.to_dict())
|
|
359
|
+
|
|
360
|
+
def to_dict(self) -> dict[str, Any]:
|
|
361
|
+
result = {
|
|
362
|
+
"schema_version": self.schema_version,
|
|
363
|
+
"config_sha256": self.config_sha256,
|
|
364
|
+
"endpoint": self.endpoint,
|
|
365
|
+
"harvester_coordinate": self.harvester_coordinate,
|
|
366
|
+
"predecessor_sha256": self.predecessor_sha256,
|
|
367
|
+
"status": self.status,
|
|
368
|
+
"reason_code": self.reason_code,
|
|
369
|
+
"requests": self.requests,
|
|
370
|
+
"responses": self.responses,
|
|
371
|
+
"pages": self.pages,
|
|
372
|
+
"records": self.records,
|
|
373
|
+
"unique_records": self.unique_records,
|
|
374
|
+
"network_bytes": self.network_bytes,
|
|
375
|
+
"elapsed_ns": self.elapsed_ns,
|
|
376
|
+
"provider_count": self.provider_count,
|
|
377
|
+
"count_basis": self.count_basis,
|
|
378
|
+
"skipped_records": self.skipped_records,
|
|
379
|
+
"flagged_rights": self.flagged_rights,
|
|
380
|
+
"convergence_passes": self.convergence_passes,
|
|
381
|
+
"response_evidence_digests": list(self.response_evidence_digests),
|
|
382
|
+
"page_shard_digests": list(self.page_shard_digests),
|
|
383
|
+
"flagged_queue_sha256": self.flagged_queue_sha256,
|
|
384
|
+
}
|
|
385
|
+
if self.rate_observation is not None or self.resume_not_before_epoch_seconds is not None:
|
|
386
|
+
result["rate_observation"] = self.rate_observation
|
|
387
|
+
result["resume_not_before_epoch_seconds"] = self.resume_not_before_epoch_seconds
|
|
388
|
+
if self.evidence_indexes is not None:
|
|
389
|
+
result["evidence_indexes"] = self.evidence_indexes
|
|
390
|
+
result["pass_map_sha256"] = self.pass_map_sha256
|
|
391
|
+
if self.normalized_bytes is not None:
|
|
392
|
+
result["normalized_bytes"] = self.normalized_bytes
|
|
393
|
+
if self.conflicting_identifiers is not None:
|
|
394
|
+
result["conflicting_identifiers"] = self.conflicting_identifiers
|
|
395
|
+
return result
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
@dataclass(frozen=True)
|
|
399
|
+
class CatalogFillResult:
|
|
400
|
+
status: str
|
|
401
|
+
reason_code: str | None
|
|
402
|
+
receipt: CatalogFillReceipt
|
|
403
|
+
checkpoint_sha256: str
|
|
404
|
+
eligible_staging_sha256: str | None
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _run_datagov_v4_fill(
|
|
408
|
+
*,
|
|
409
|
+
config: CatalogFillConfig,
|
|
410
|
+
staging_root: Path,
|
|
411
|
+
retriever: Any,
|
|
412
|
+
authorization: DatagovV4Authorization,
|
|
413
|
+
expected_checkpoint_digest: str | None,
|
|
414
|
+
clock_ns: Callable[[], int],
|
|
415
|
+
wall_clock_seconds: Callable[[], int],
|
|
416
|
+
max_pages_this_invocation: int | None,
|
|
417
|
+
staging_descriptor: int | None,
|
|
418
|
+
pacing_sleep: Callable[[float], None] | None,
|
|
419
|
+
) -> CatalogFillResult:
|
|
420
|
+
"""Run v4 through compact index heads and partitioned pass maps only."""
|
|
421
|
+
|
|
422
|
+
staging = FillStaging(staging_root, root_descriptor=staging_descriptor)
|
|
423
|
+
with staging.locked():
|
|
424
|
+
state = staging.load()
|
|
425
|
+
if state is None:
|
|
426
|
+
if expected_checkpoint_digest is not None:
|
|
427
|
+
raise CatalogFillRefused(
|
|
428
|
+
"FILL_CHECKPOINT_MISMATCH",
|
|
429
|
+
"fill.expected_checkpoint_digest",
|
|
430
|
+
"no live checkpoint matches the caller's expected digest",
|
|
431
|
+
)
|
|
432
|
+
state = staging.commit(
|
|
433
|
+
_initial_v4_state(config, staging),
|
|
434
|
+
expected_generation=-1,
|
|
435
|
+
expected_checkpoint_digest=None,
|
|
436
|
+
)
|
|
437
|
+
else:
|
|
438
|
+
live_digest = staging.state_digest()
|
|
439
|
+
journal_predecessor = _v4_journal_predecessor(state)
|
|
440
|
+
handoff_predecessor = state.get("handoff_predecessor_sha256")
|
|
441
|
+
if (
|
|
442
|
+
not isinstance(expected_checkpoint_digest, str)
|
|
443
|
+
or not _SHA256.fullmatch(expected_checkpoint_digest)
|
|
444
|
+
or expected_checkpoint_digest
|
|
445
|
+
not in {live_digest, journal_predecessor, handoff_predecessor}
|
|
446
|
+
):
|
|
447
|
+
raise CatalogFillRefused(
|
|
448
|
+
"FILL_CHECKPOINT_MISMATCH",
|
|
449
|
+
"fill.expected_checkpoint_digest",
|
|
450
|
+
"resume requires the live or journal-authorized predecessor digest",
|
|
451
|
+
)
|
|
452
|
+
state = dict(state)
|
|
453
|
+
if expected_checkpoint_digest == live_digest:
|
|
454
|
+
state["handoff_predecessor_sha256"] = expected_checkpoint_digest
|
|
455
|
+
elif state.get("handoff_predecessor_sha256") is None:
|
|
456
|
+
state["handoff_predecessor_sha256"] = expected_checkpoint_digest
|
|
457
|
+
_validate_v4_state(state, config)
|
|
458
|
+
live_digest = staging.state_digest()
|
|
459
|
+
if _v4_cached_checkpoint_valid(staging, live_digest):
|
|
460
|
+
_validate_v4_tail_evidence(staging, state, config=config)
|
|
461
|
+
else:
|
|
462
|
+
_validate_v4_evidence(staging, state, config=config)
|
|
463
|
+
_remember_v4_checkpoint(staging, live_digest)
|
|
464
|
+
if state["reason_code"] in _V4_IDEMPOTENT_TERMINAL_REASONS:
|
|
465
|
+
return _v4_result(staging, state, config=config, reason=state["reason_code"])
|
|
466
|
+
if state["inflight_request"] is not None:
|
|
467
|
+
state, reconciled_reason = _reconcile_v4_inflight(
|
|
468
|
+
staging, state, config, clock_ns, wall_clock_seconds
|
|
469
|
+
)
|
|
470
|
+
if reconciled_reason is not None:
|
|
471
|
+
return _v4_result(staging, state, config=config, reason=reconciled_reason)
|
|
472
|
+
if state["completed"]:
|
|
473
|
+
return _v4_result(staging, state, config=config, reason=None)
|
|
474
|
+
paced_response_ns: int | None = None
|
|
475
|
+
|
|
476
|
+
def pace_in_process(deadline_epoch_seconds: int) -> None:
|
|
477
|
+
"""Wait the politeness interval out here instead of handing it back to the caller.
|
|
478
|
+
|
|
479
|
+
The checkpoint has already been committed carrying this deadline, so a process that
|
|
480
|
+
dies inside this wait resumes exactly where a returned ``DATAGOV_PACING_REQUIRED``
|
|
481
|
+
would have: the durable journal does not know the difference. What changes is who
|
|
482
|
+
waits, and therefore how many times the accumulated evidence tree is re-audited --
|
|
483
|
+
once per invocation rather than once per page.
|
|
484
|
+
"""
|
|
485
|
+
|
|
486
|
+
assert pacing_sleep is not None
|
|
487
|
+
seconds = _pacing_sleep_seconds(
|
|
488
|
+
config,
|
|
489
|
+
deadline_epoch_seconds=deadline_epoch_seconds,
|
|
490
|
+
now_epoch_seconds=_wall_clock(wall_clock_seconds),
|
|
491
|
+
anchor_ns=paced_response_ns,
|
|
492
|
+
now_ns=_clock(clock_ns),
|
|
493
|
+
)
|
|
494
|
+
if seconds > 0:
|
|
495
|
+
pacing_sleep(seconds)
|
|
496
|
+
|
|
497
|
+
now = _wall_clock(wall_clock_seconds)
|
|
498
|
+
wait_until = state["resume_not_before_epoch_seconds"]
|
|
499
|
+
if state["pending_response"] is None and wait_until is not None and now < wait_until:
|
|
500
|
+
if pacing_sleep is None or state["wait_reason"] != "DATAGOV_PACING_REQUIRED":
|
|
501
|
+
return _v4_result(staging, state, config=config, reason=state["wait_reason"])
|
|
502
|
+
pace_in_process(wait_until)
|
|
503
|
+
invocation_pages = 0
|
|
504
|
+
seen_cursors, seen_responses = _v4_pass_loop_coordinates(staging, state)
|
|
505
|
+
while True:
|
|
506
|
+
if state["pending_response"] is None:
|
|
507
|
+
preflight = _preflight_limit(state, config.limits)
|
|
508
|
+
if preflight is not None:
|
|
509
|
+
state = _v4_stop(staging, state, preflight)
|
|
510
|
+
return _v4_result(staging, state, config=config, reason=preflight)
|
|
511
|
+
cursor = _cursor_from_dict(state["next_cursor"])
|
|
512
|
+
request_url = _request_url(config, cursor)
|
|
513
|
+
transport_limits = _remaining_transport_limits(
|
|
514
|
+
state, config.limits, DATAGOV_V4_LIMITS
|
|
515
|
+
)
|
|
516
|
+
authorized_predecessor = staging.state_digest()
|
|
517
|
+
state = dict(state)
|
|
518
|
+
state["inflight_request"] = {
|
|
519
|
+
"authorized_predecessor_sha256": authorized_predecessor,
|
|
520
|
+
"attempt": state["budget"]["requests"] + 1,
|
|
521
|
+
"request_url": request_url,
|
|
522
|
+
"cursor": cursor.to_dict() if cursor is not None else None,
|
|
523
|
+
"started": False,
|
|
524
|
+
"started_ns": None,
|
|
525
|
+
"response_observed_ns": None,
|
|
526
|
+
"response_observed_epoch_seconds": None,
|
|
527
|
+
"observed_response": None,
|
|
528
|
+
}
|
|
529
|
+
state["reason_code"] = None
|
|
530
|
+
state = _v4_commit(staging, state)
|
|
531
|
+
|
|
532
|
+
def mark_started() -> None:
|
|
533
|
+
nonlocal state
|
|
534
|
+
inflight = dict(state["inflight_request"])
|
|
535
|
+
if not inflight["started"]:
|
|
536
|
+
started_ns = _clock(clock_ns)
|
|
537
|
+
inflight["started"] = True
|
|
538
|
+
inflight["started_ns"] = started_ns
|
|
539
|
+
state = dict(state)
|
|
540
|
+
state["inflight_request"] = inflight
|
|
541
|
+
state = _v4_commit(staging, state)
|
|
542
|
+
|
|
543
|
+
def retain_response(
|
|
544
|
+
payload: bytes,
|
|
545
|
+
evidence: EvidenceReference,
|
|
546
|
+
response: HarvestResponseEvidence,
|
|
547
|
+
_authorized_predecessor: str = authorized_predecessor,
|
|
548
|
+
) -> None:
|
|
549
|
+
nonlocal state
|
|
550
|
+
if state["pending_response"] is not None:
|
|
551
|
+
return
|
|
552
|
+
inflight = state["inflight_request"]
|
|
553
|
+
if not inflight["started"]:
|
|
554
|
+
raise _v4_evidence_corrupt("response preceded request start")
|
|
555
|
+
raw_digest = staging.write_blob("raw_response", payload)
|
|
556
|
+
elapsed = _v4_elapsed_ns(
|
|
557
|
+
inflight["started_ns"], inflight["response_observed_ns"]
|
|
558
|
+
)
|
|
559
|
+
state = _promote_v4_inflight(
|
|
560
|
+
staging,
|
|
561
|
+
state,
|
|
562
|
+
elapsed_ns=elapsed,
|
|
563
|
+
observed_epoch_seconds=inflight["response_observed_epoch_seconds"],
|
|
564
|
+
)
|
|
565
|
+
state = _append_v4_index(
|
|
566
|
+
staging,
|
|
567
|
+
state,
|
|
568
|
+
"response",
|
|
569
|
+
[
|
|
570
|
+
{
|
|
571
|
+
"request_attempt": inflight["attempt"],
|
|
572
|
+
"request_url": inflight["request_url"],
|
|
573
|
+
"observed_epoch_seconds": inflight[
|
|
574
|
+
"response_observed_epoch_seconds"
|
|
575
|
+
],
|
|
576
|
+
"raw_response_sha256": raw_digest,
|
|
577
|
+
"evidence": response.to_dict(),
|
|
578
|
+
}
|
|
579
|
+
],
|
|
580
|
+
)
|
|
581
|
+
budget = dict(state["budget"])
|
|
582
|
+
budget["responses"] += response.responses
|
|
583
|
+
budget["network_bytes"] += response.network_bytes
|
|
584
|
+
budget["elapsed_ns"] += elapsed
|
|
585
|
+
state["budget"] = budget
|
|
586
|
+
rate = response.rate_observation
|
|
587
|
+
state["last_rate_observation"] = rate.to_dict() if rate else None
|
|
588
|
+
wait_until, wait_reason = _v4_response_wait(
|
|
589
|
+
config,
|
|
590
|
+
response.status,
|
|
591
|
+
rate,
|
|
592
|
+
inflight["response_observed_epoch_seconds"],
|
|
593
|
+
)
|
|
594
|
+
state["resume_not_before_epoch_seconds"] = wait_until
|
|
595
|
+
state["wait_reason"] = wait_reason
|
|
596
|
+
state["pending_response"] = {
|
|
597
|
+
"authorized_predecessor_sha256": _authorized_predecessor,
|
|
598
|
+
"request_url": inflight["request_url"],
|
|
599
|
+
"cursor": inflight["cursor"],
|
|
600
|
+
"raw_response_sha256": raw_digest,
|
|
601
|
+
"evidence": evidence.to_dict(),
|
|
602
|
+
"response_evidence": response.to_dict(),
|
|
603
|
+
}
|
|
604
|
+
state = _v4_commit(staging, state)
|
|
605
|
+
|
|
606
|
+
def observe_response(
|
|
607
|
+
response: HarvestResponseEvidence | DatagovRefusedResponseEvidence,
|
|
608
|
+
) -> None:
|
|
609
|
+
nonlocal state, paced_response_ns
|
|
610
|
+
inflight = dict(state["inflight_request"])
|
|
611
|
+
if not inflight["started"]:
|
|
612
|
+
raise _v4_evidence_corrupt("response preceded request start")
|
|
613
|
+
if response.requests != 1 or response.responses != 1:
|
|
614
|
+
raise CatalogHarvestError(
|
|
615
|
+
"DATAGOV_TRANSPORT_SHAPE",
|
|
616
|
+
"harvest.response",
|
|
617
|
+
"v4 transport evidence exceeded its one-attempt closure",
|
|
618
|
+
)
|
|
619
|
+
observed_ns = _clock(clock_ns)
|
|
620
|
+
observed_epoch_seconds = _wall_clock(wall_clock_seconds)
|
|
621
|
+
_v4_elapsed_ns(inflight["started_ns"], observed_ns)
|
|
622
|
+
inflight["response_observed_ns"] = observed_ns
|
|
623
|
+
inflight["response_observed_epoch_seconds"] = observed_epoch_seconds
|
|
624
|
+
# The monotonic instant an in-process pace measures its interval from. The
|
|
625
|
+
# durable checkpoint keeps only whole seconds; this keeps the fraction.
|
|
626
|
+
paced_response_ns = observed_ns
|
|
627
|
+
rate = response.rate_observation
|
|
628
|
+
inflight["observed_response"] = {
|
|
629
|
+
"status": response.status,
|
|
630
|
+
"responses": response.responses,
|
|
631
|
+
"content_bytes": response.content_bytes,
|
|
632
|
+
"network_bytes": response.network_bytes,
|
|
633
|
+
"transport_evidence_digest": getattr(
|
|
634
|
+
response, "transport_evidence_digest", None
|
|
635
|
+
),
|
|
636
|
+
"rate_observation": rate.to_dict() if rate else None,
|
|
637
|
+
}
|
|
638
|
+
state = dict(state)
|
|
639
|
+
state["inflight_request"] = inflight
|
|
640
|
+
state["last_rate_observation"] = rate.to_dict() if rate else None
|
|
641
|
+
wait_until, wait_reason = _v4_response_wait(
|
|
642
|
+
config, response.status, rate, observed_epoch_seconds
|
|
643
|
+
)
|
|
644
|
+
state["resume_not_before_epoch_seconds"] = wait_until
|
|
645
|
+
state["wait_reason"] = wait_reason
|
|
646
|
+
state = _v4_commit(staging, state)
|
|
647
|
+
|
|
648
|
+
try:
|
|
649
|
+
fetched = fetch_datagov_v4_response(
|
|
650
|
+
retriever=retriever,
|
|
651
|
+
url=request_url,
|
|
652
|
+
observed_at=config.observed_at,
|
|
653
|
+
limits=transport_limits,
|
|
654
|
+
authorization=authorization,
|
|
655
|
+
on_request_started=mark_started,
|
|
656
|
+
on_response_observed=observe_response,
|
|
657
|
+
on_response_received=retain_response,
|
|
658
|
+
)
|
|
659
|
+
except DatagovFetchRefused as error:
|
|
660
|
+
response = error.response_evidence
|
|
661
|
+
if not state["inflight_request"]["started"]:
|
|
662
|
+
mark_started()
|
|
663
|
+
inflight = state["inflight_request"]
|
|
664
|
+
elapsed = _v4_elapsed_ns(
|
|
665
|
+
inflight["started_ns"], inflight["response_observed_ns"]
|
|
666
|
+
)
|
|
667
|
+
state = _promote_v4_inflight(
|
|
668
|
+
staging,
|
|
669
|
+
state,
|
|
670
|
+
elapsed_ns=elapsed,
|
|
671
|
+
observed_epoch_seconds=inflight["response_observed_epoch_seconds"],
|
|
672
|
+
)
|
|
673
|
+
state = _append_v4_index(
|
|
674
|
+
staging,
|
|
675
|
+
state,
|
|
676
|
+
"response",
|
|
677
|
+
[
|
|
678
|
+
{
|
|
679
|
+
"request_attempt": inflight["attempt"],
|
|
680
|
+
"request_url": inflight["request_url"],
|
|
681
|
+
"observed_epoch_seconds": inflight[
|
|
682
|
+
"response_observed_epoch_seconds"
|
|
683
|
+
],
|
|
684
|
+
"refusal": {
|
|
685
|
+
"code": error.code,
|
|
686
|
+
"status": response.status,
|
|
687
|
+
"content_bytes": response.content_bytes,
|
|
688
|
+
"network_bytes": response.network_bytes,
|
|
689
|
+
"rate_observation": (
|
|
690
|
+
response.rate_observation.to_dict()
|
|
691
|
+
if response.rate_observation
|
|
692
|
+
else None
|
|
693
|
+
),
|
|
694
|
+
},
|
|
695
|
+
}
|
|
696
|
+
],
|
|
697
|
+
)
|
|
698
|
+
budget = dict(state["budget"])
|
|
699
|
+
budget["responses"] += response.responses
|
|
700
|
+
budget["network_bytes"] += response.network_bytes
|
|
701
|
+
budget["elapsed_ns"] += elapsed
|
|
702
|
+
state["budget"] = budget
|
|
703
|
+
rate = response.rate_observation
|
|
704
|
+
state["last_rate_observation"] = rate.to_dict() if rate else None
|
|
705
|
+
wait_until, wait_reason = _v4_response_wait(
|
|
706
|
+
config,
|
|
707
|
+
response.status,
|
|
708
|
+
rate,
|
|
709
|
+
inflight["response_observed_epoch_seconds"],
|
|
710
|
+
)
|
|
711
|
+
state["resume_not_before_epoch_seconds"] = wait_until
|
|
712
|
+
state["wait_reason"] = wait_reason
|
|
713
|
+
reason = _v4_transport_terminal_reason(error.code, state, config.limits)
|
|
714
|
+
state = _v4_stop(staging, state, reason)
|
|
715
|
+
return _v4_result(staging, state, config=config, reason=reason)
|
|
716
|
+
except (CatalogHarvestError, AcquisitionSecurityError) as error:
|
|
717
|
+
inflight = state["inflight_request"]
|
|
718
|
+
if state["inflight_request"]["started"]:
|
|
719
|
+
elapsed = _v4_elapsed_ns(inflight["started_ns"], _clock(clock_ns))
|
|
720
|
+
observed_epoch_seconds = _wall_clock(wall_clock_seconds)
|
|
721
|
+
state = _promote_v4_inflight(
|
|
722
|
+
staging,
|
|
723
|
+
state,
|
|
724
|
+
elapsed_ns=elapsed,
|
|
725
|
+
observed_epoch_seconds=observed_epoch_seconds,
|
|
726
|
+
)
|
|
727
|
+
else:
|
|
728
|
+
state = dict(state)
|
|
729
|
+
state["inflight_request"] = None
|
|
730
|
+
budget = dict(state["budget"])
|
|
731
|
+
if inflight["started"]:
|
|
732
|
+
if isinstance(error, PeerAttemptError):
|
|
733
|
+
budget["network_bytes"] += error.total_response_body_size_bytes
|
|
734
|
+
budget["elapsed_ns"] += elapsed
|
|
735
|
+
wait_until = _pacing_wait(config, observed_epoch_seconds)
|
|
736
|
+
state["resume_not_before_epoch_seconds"] = wait_until
|
|
737
|
+
state["wait_reason"] = (
|
|
738
|
+
"DATAGOV_PACING_REQUIRED" if wait_until is not None else None
|
|
739
|
+
)
|
|
740
|
+
state["budget"] = budget
|
|
741
|
+
reason = _v4_transport_terminal_reason(error.code, state, config.limits)
|
|
742
|
+
state = _v4_stop(staging, state, reason)
|
|
743
|
+
return _v4_result(staging, state, config=config, reason=reason)
|
|
744
|
+
if state["pending_response"] is None:
|
|
745
|
+
retain_response(*fetched)
|
|
746
|
+
continue
|
|
747
|
+
|
|
748
|
+
pending = state["pending_response"]
|
|
749
|
+
response_wait_until = state["resume_not_before_epoch_seconds"]
|
|
750
|
+
request_url = pending["request_url"]
|
|
751
|
+
cursor = _cursor_from_dict(pending["cursor"])
|
|
752
|
+
raw_digest = pending["raw_response_sha256"]
|
|
753
|
+
payload = staging.read_blob("raw_response", raw_digest)
|
|
754
|
+
evidence = _evidence_from_dict(pending["evidence"])
|
|
755
|
+
response_evidence = _response_evidence_from_dict(pending["response_evidence"])
|
|
756
|
+
rate = response_evidence.rate_observation
|
|
757
|
+
late = _late_limit(state, config.limits, response_evidence)
|
|
758
|
+
if late is not None:
|
|
759
|
+
state = _clear_v4_pending_and_stop(staging, state, late)
|
|
760
|
+
return _v4_result(staging, state, config=config, reason=late)
|
|
761
|
+
try:
|
|
762
|
+
_validate_response_endpoint(config.endpoint, request_url, response_evidence)
|
|
763
|
+
except CatalogFillRefused as error:
|
|
764
|
+
state = _clear_v4_pending_and_stop(staging, state, error.code)
|
|
765
|
+
return _v4_result(staging, state, config=config, reason=error.code)
|
|
766
|
+
if response_evidence.status == 429:
|
|
767
|
+
state = _clear_v4_pending_and_stop(staging, state, "DATAGOV_RATE_LIMITED")
|
|
768
|
+
return _v4_result(staging, state, config=config, reason="DATAGOV_RATE_LIMITED")
|
|
769
|
+
if response_evidence.status == 403:
|
|
770
|
+
state = _clear_v4_pending_and_stop(staging, state, "DATAGOV_AUTHENTICATION_FAILED")
|
|
771
|
+
return _v4_result(
|
|
772
|
+
staging, state, config=config, reason="DATAGOV_AUTHENTICATION_FAILED"
|
|
773
|
+
)
|
|
774
|
+
if response_evidence.status != 200:
|
|
775
|
+
state = _clear_v4_pending_and_stop(staging, state, "DATAGOV_HTTP_STATUS")
|
|
776
|
+
return _v4_result(staging, state, config=config, reason="DATAGOV_HTTP_STATUS")
|
|
777
|
+
if rate is None:
|
|
778
|
+
state = _clear_v4_pending_and_stop(staging, state, "DATAGOV_RATE_METADATA")
|
|
779
|
+
return _v4_result(staging, state, config=config, reason="DATAGOV_RATE_METADATA")
|
|
780
|
+
|
|
781
|
+
try:
|
|
782
|
+
page = DatagovV4Harvester().parse_page(
|
|
783
|
+
payload,
|
|
784
|
+
uri=request_url,
|
|
785
|
+
observed_at=config.observed_at,
|
|
786
|
+
evidence=evidence,
|
|
787
|
+
response_evidence=response_evidence,
|
|
788
|
+
cursor=cursor,
|
|
789
|
+
)
|
|
790
|
+
except CatalogHarvestError as error:
|
|
791
|
+
state = _clear_v4_pending_and_stop(staging, state, error.code)
|
|
792
|
+
return _v4_result(staging, state, config=config, reason=error.code)
|
|
793
|
+
requested_cursor = cursor.to_dict() if cursor is not None else None
|
|
794
|
+
requested_digest = canonical_sha256(requested_cursor)
|
|
795
|
+
next_cursor = page.next_cursor.to_dict() if page.next_cursor is not None else None
|
|
796
|
+
next_digest = canonical_sha256(next_cursor) if next_cursor is not None else None
|
|
797
|
+
if raw_digest in seen_responses:
|
|
798
|
+
state = _clear_v4_pending_and_stop(staging, state, "FILL_RESPONSE_LOOP")
|
|
799
|
+
return _v4_result(staging, state, config=config, reason="FILL_RESPONSE_LOOP")
|
|
800
|
+
if requested_digest in seen_cursors or (
|
|
801
|
+
next_cursor is not None
|
|
802
|
+
and (next_cursor == requested_cursor or next_digest in seen_cursors)
|
|
803
|
+
):
|
|
804
|
+
state = _clear_v4_pending_and_stop(staging, state, "FILL_CURSOR_LOOP")
|
|
805
|
+
return _v4_result(staging, state, config=config, reason="FILL_CURSOR_LOOP")
|
|
806
|
+
projected = _projected_page_limit(state, config.limits, page)
|
|
807
|
+
if projected is not None:
|
|
808
|
+
state = _clear_v4_pending_and_stop(staging, state, projected)
|
|
809
|
+
return _v4_result(staging, state, config=config, reason=projected)
|
|
810
|
+
try:
|
|
811
|
+
normalized_digest, normalized_bytes = _write_v4_normalized_page(
|
|
812
|
+
staging,
|
|
813
|
+
raw_response_sha256=raw_digest,
|
|
814
|
+
records=[record.to_dict() for record in page.records],
|
|
815
|
+
skipped_record_ids=list(page.skipped_record_ids),
|
|
816
|
+
maximum_bytes=min(
|
|
817
|
+
DATAGOV_V4_MAX_NORMALIZED_PAGE_BYTES,
|
|
818
|
+
config.limits.max_aggregate_bytes * DATAGOV_V4_MAX_NORMALIZED_EXPANSION
|
|
819
|
+
- state["normalized_bytes"],
|
|
820
|
+
),
|
|
821
|
+
)
|
|
822
|
+
except CatalogFillRefused as error:
|
|
823
|
+
if error.code != "DATAGOV_NORMALIZED_LIMIT":
|
|
824
|
+
raise
|
|
825
|
+
state = _clear_v4_pending_and_stop(staging, state, error.code)
|
|
826
|
+
return _v4_result(staging, state, config=config, reason=error.code)
|
|
827
|
+
if (
|
|
828
|
+
state["normalized_bytes"] + normalized_bytes
|
|
829
|
+
> config.limits.max_aggregate_bytes * DATAGOV_V4_MAX_NORMALIZED_EXPANSION
|
|
830
|
+
):
|
|
831
|
+
state = _clear_v4_pending_and_stop(staging, state, "DATAGOV_NORMALIZED_LIMIT")
|
|
832
|
+
return _v4_result(staging, state, config=config, reason="DATAGOV_NORMALIZED_LIMIT")
|
|
833
|
+
run_descriptors = write_partition_runs(staging, page.records)
|
|
834
|
+
if run_descriptors:
|
|
835
|
+
partition_head = staging.append_index(
|
|
836
|
+
state["partition_run_index"],
|
|
837
|
+
stream="partition_run",
|
|
838
|
+
items=run_descriptors,
|
|
839
|
+
)
|
|
840
|
+
state["partition_run_index"] = partition_head
|
|
841
|
+
state = _append_v4_index(
|
|
842
|
+
staging,
|
|
843
|
+
state,
|
|
844
|
+
"page",
|
|
845
|
+
[
|
|
846
|
+
{
|
|
847
|
+
"pass": state["pass_number"],
|
|
848
|
+
"raw_response_sha256": raw_digest,
|
|
849
|
+
"normalized_sha256": normalized_digest,
|
|
850
|
+
"record_count": len(page.records),
|
|
851
|
+
}
|
|
852
|
+
],
|
|
853
|
+
)
|
|
854
|
+
if page.records:
|
|
855
|
+
state = _append_v4_index(
|
|
856
|
+
staging,
|
|
857
|
+
state,
|
|
858
|
+
"record",
|
|
859
|
+
[
|
|
860
|
+
{
|
|
861
|
+
"identifier_sha256": canonical_sha256(record.record_id),
|
|
862
|
+
"semantic_sha256": record.digest,
|
|
863
|
+
"normalized_page_sha256": normalized_digest,
|
|
864
|
+
}
|
|
865
|
+
for record in page.records
|
|
866
|
+
],
|
|
867
|
+
)
|
|
868
|
+
state = _append_v4_index(
|
|
869
|
+
staging,
|
|
870
|
+
state,
|
|
871
|
+
"cursor_transition",
|
|
872
|
+
[
|
|
873
|
+
{
|
|
874
|
+
"request_cursor": cursor.to_dict() if cursor is not None else None,
|
|
875
|
+
"next_cursor": (
|
|
876
|
+
page.next_cursor.to_dict() if page.next_cursor is not None else None
|
|
877
|
+
),
|
|
878
|
+
"normalized_page_sha256": normalized_digest,
|
|
879
|
+
}
|
|
880
|
+
],
|
|
881
|
+
)
|
|
882
|
+
budget = dict(state["budget"])
|
|
883
|
+
budget["pages"] += 1
|
|
884
|
+
budget["records"] += len(page.records)
|
|
885
|
+
state["normalized_bytes"] += normalized_bytes
|
|
886
|
+
budget["skipped_records"] += len(page.skipped_record_ids)
|
|
887
|
+
state["budget"] = budget
|
|
888
|
+
state["next_cursor"] = page.next_cursor.to_dict() if page.next_cursor else None
|
|
889
|
+
state["reason_code"] = None
|
|
890
|
+
state["pending_response"] = None
|
|
891
|
+
seen_cursors.add(requested_digest)
|
|
892
|
+
seen_responses.add(raw_digest)
|
|
893
|
+
state["pass_cursor_set_sha256"] = getattr(
|
|
894
|
+
seen_cursors, "root_sha256", state["pass_cursor_set_sha256"]
|
|
895
|
+
)
|
|
896
|
+
state["pass_response_set_sha256"] = getattr(
|
|
897
|
+
seen_responses, "root_sha256", state["pass_response_set_sha256"]
|
|
898
|
+
)
|
|
899
|
+
invocation_pages += 1
|
|
900
|
+
pacing_wait = response_wait_until
|
|
901
|
+
|
|
902
|
+
if page.next_cursor is not None:
|
|
903
|
+
if pacing_wait is not None:
|
|
904
|
+
state["resume_not_before_epoch_seconds"] = pacing_wait
|
|
905
|
+
state["wait_reason"] = "DATAGOV_PACING_REQUIRED"
|
|
906
|
+
state = _v4_commit(staging, state)
|
|
907
|
+
if max_pages_this_invocation == invocation_pages:
|
|
908
|
+
state = _v4_stop(staging, state, "INVOCATION_PAGE_LIMIT")
|
|
909
|
+
return _v4_result(staging, state, config=config, reason="INVOCATION_PAGE_LIMIT")
|
|
910
|
+
if pacing_wait is not None:
|
|
911
|
+
if pacing_sleep is None:
|
|
912
|
+
state = _v4_stop(staging, state, "DATAGOV_PACING_REQUIRED")
|
|
913
|
+
return _v4_result(
|
|
914
|
+
staging, state, config=config, reason="DATAGOV_PACING_REQUIRED"
|
|
915
|
+
)
|
|
916
|
+
pace_in_process(pacing_wait)
|
|
917
|
+
continue
|
|
918
|
+
|
|
919
|
+
run_items = staging.iter_index(
|
|
920
|
+
state["partition_run_index"],
|
|
921
|
+
stream="partition_run",
|
|
922
|
+
start=state["pass_start_run_count"],
|
|
923
|
+
)
|
|
924
|
+
pass_map = external_merge_pass(staging, run_items)
|
|
925
|
+
state["last_pass_manifest_sha256"] = pass_map.manifest_sha256
|
|
926
|
+
stable = state["previous_pass_map_sha256"] == pass_map.root_sha256
|
|
927
|
+
state["conflicting_identifiers"] = pass_map.conflicting_identifiers
|
|
928
|
+
arbitrated = pass_map.conflicting_identifiers <= config.max_conflicting_identifiers
|
|
929
|
+
if stable and arbitrated:
|
|
930
|
+
state["completed"] = True
|
|
931
|
+
state["final_map_sha256"] = pass_map.root_sha256
|
|
932
|
+
state["unique_records"] = pass_map.unique_records
|
|
933
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
934
|
+
state["wait_reason"] = None
|
|
935
|
+
state = _v4_commit(staging, state)
|
|
936
|
+
return _v4_result(staging, state, config=config, reason=None)
|
|
937
|
+
if state["pass_number"] >= config.max_convergence_passes:
|
|
938
|
+
state["final_map_sha256"] = pass_map.root_sha256
|
|
939
|
+
state["unique_records"] = pass_map.unique_records
|
|
940
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
941
|
+
state["wait_reason"] = None
|
|
942
|
+
reason = "FILL_CONFLICTING_IDENTIFIERS" if stable else "FILL_NONCONVERGENT"
|
|
943
|
+
state = _v4_stop(staging, state, reason)
|
|
944
|
+
return _v4_result(staging, state, config=config, reason=reason)
|
|
945
|
+
state["previous_pass_map_sha256"] = pass_map.root_sha256
|
|
946
|
+
state["pass_number"] += 1
|
|
947
|
+
state["pass_start_run_count"] = state["partition_run_index"]["count"]
|
|
948
|
+
state["next_cursor"] = None
|
|
949
|
+
state["unique_records"] = pass_map.unique_records
|
|
950
|
+
seen_cursors.clear()
|
|
951
|
+
seen_responses.clear()
|
|
952
|
+
state["pass_cursor_set_sha256"] = None
|
|
953
|
+
state["pass_response_set_sha256"] = None
|
|
954
|
+
if pacing_wait is not None:
|
|
955
|
+
state["resume_not_before_epoch_seconds"] = pacing_wait
|
|
956
|
+
state["wait_reason"] = "DATAGOV_PACING_REQUIRED"
|
|
957
|
+
state = _v4_commit(staging, state)
|
|
958
|
+
if max_pages_this_invocation == invocation_pages:
|
|
959
|
+
state = _v4_stop(staging, state, "INVOCATION_PAGE_LIMIT")
|
|
960
|
+
return _v4_result(staging, state, config=config, reason="INVOCATION_PAGE_LIMIT")
|
|
961
|
+
if pacing_wait is not None:
|
|
962
|
+
if pacing_sleep is None:
|
|
963
|
+
state = _v4_stop(staging, state, "DATAGOV_PACING_REQUIRED")
|
|
964
|
+
return _v4_result(
|
|
965
|
+
staging, state, config=config, reason="DATAGOV_PACING_REQUIRED"
|
|
966
|
+
)
|
|
967
|
+
pace_in_process(pacing_wait)
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
def _initial_v4_state(config: CatalogFillConfig, staging: FillStaging) -> dict[str, Any]:
|
|
971
|
+
state = {
|
|
972
|
+
"v4_checkpoint_schema_version": DATAGOV_V4_CHECKPOINT_SCHEMA,
|
|
973
|
+
"config_sha256": config.digest,
|
|
974
|
+
"endpoint": config.endpoint,
|
|
975
|
+
"harvester_coordinate": config.coordinate,
|
|
976
|
+
"predecessor_sha256": config.predecessor_sha256,
|
|
977
|
+
"next_cursor": None,
|
|
978
|
+
"budget": {
|
|
979
|
+
"requests": 0,
|
|
980
|
+
"responses": 0,
|
|
981
|
+
"pages": 0,
|
|
982
|
+
"records": 0,
|
|
983
|
+
"network_bytes": 0,
|
|
984
|
+
"elapsed_ns": 0,
|
|
985
|
+
"skipped_records": 0,
|
|
986
|
+
},
|
|
987
|
+
"provider_count": None,
|
|
988
|
+
"count_basis": "not_reported",
|
|
989
|
+
"pass_number": 1,
|
|
990
|
+
"previous_pass_map_sha256": None,
|
|
991
|
+
"pass_start_run_count": 0,
|
|
992
|
+
"partition_run_index": staging.empty_index_head(),
|
|
993
|
+
"evidence_indexes": {stream: staging.empty_index_head() for stream in _V4_EVIDENCE_STREAMS},
|
|
994
|
+
"last_rate_observation": None,
|
|
995
|
+
"resume_not_before_epoch_seconds": None,
|
|
996
|
+
"wait_reason": None,
|
|
997
|
+
"completed": False,
|
|
998
|
+
"reason_code": None,
|
|
999
|
+
"final_map_sha256": None,
|
|
1000
|
+
"last_pass_manifest_sha256": None,
|
|
1001
|
+
"unique_records": 0,
|
|
1002
|
+
"conflicting_identifiers": 0,
|
|
1003
|
+
"normalized_bytes": 0,
|
|
1004
|
+
"inflight_request": None,
|
|
1005
|
+
"pending_response": None,
|
|
1006
|
+
"handoff_predecessor_sha256": None,
|
|
1007
|
+
"validated_tail_sha256": None,
|
|
1008
|
+
"pass_cursor_set_sha256": None,
|
|
1009
|
+
"pass_response_set_sha256": None,
|
|
1010
|
+
}
|
|
1011
|
+
state["validated_tail_sha256"] = _v4_tail_seal(state)
|
|
1012
|
+
return state
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def _validate_v4_state(state: dict[str, Any], config: CatalogFillConfig) -> None:
|
|
1016
|
+
required = {
|
|
1017
|
+
"v4_checkpoint_schema_version",
|
|
1018
|
+
"config_sha256",
|
|
1019
|
+
"endpoint",
|
|
1020
|
+
"harvester_coordinate",
|
|
1021
|
+
"predecessor_sha256",
|
|
1022
|
+
"next_cursor",
|
|
1023
|
+
"budget",
|
|
1024
|
+
"provider_count",
|
|
1025
|
+
"count_basis",
|
|
1026
|
+
"pass_number",
|
|
1027
|
+
"previous_pass_map_sha256",
|
|
1028
|
+
"pass_start_run_count",
|
|
1029
|
+
"partition_run_index",
|
|
1030
|
+
"evidence_indexes",
|
|
1031
|
+
"last_rate_observation",
|
|
1032
|
+
"resume_not_before_epoch_seconds",
|
|
1033
|
+
"wait_reason",
|
|
1034
|
+
"completed",
|
|
1035
|
+
"reason_code",
|
|
1036
|
+
"final_map_sha256",
|
|
1037
|
+
"last_pass_manifest_sha256",
|
|
1038
|
+
"unique_records",
|
|
1039
|
+
"conflicting_identifiers",
|
|
1040
|
+
"normalized_bytes",
|
|
1041
|
+
"inflight_request",
|
|
1042
|
+
"pending_response",
|
|
1043
|
+
"handoff_predecessor_sha256",
|
|
1044
|
+
"validated_tail_sha256",
|
|
1045
|
+
"pass_cursor_set_sha256",
|
|
1046
|
+
"pass_response_set_sha256",
|
|
1047
|
+
"schema_version",
|
|
1048
|
+
"generation",
|
|
1049
|
+
}
|
|
1050
|
+
if set(state) != required:
|
|
1051
|
+
if state.get("v4_checkpoint_schema_version") != DATAGOV_V4_CHECKPOINT_SCHEMA:
|
|
1052
|
+
raise CatalogFillRefused(
|
|
1053
|
+
"DATAGOV_CHECKPOINT_VERSION",
|
|
1054
|
+
"fill.staging.state",
|
|
1055
|
+
"in-progress v4 checkpoint predates the crash-exact schema",
|
|
1056
|
+
)
|
|
1057
|
+
raise CatalogFillRefused(
|
|
1058
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.state", "v4 state keys differ"
|
|
1059
|
+
)
|
|
1060
|
+
if state["v4_checkpoint_schema_version"] != DATAGOV_V4_CHECKPOINT_SCHEMA:
|
|
1061
|
+
raise CatalogFillRefused(
|
|
1062
|
+
"DATAGOV_CHECKPOINT_VERSION",
|
|
1063
|
+
"fill.staging.state",
|
|
1064
|
+
"v4 checkpoint schema is unsupported",
|
|
1065
|
+
)
|
|
1066
|
+
if not _SHA256.fullmatch(state["validated_tail_sha256"] or "") or state[
|
|
1067
|
+
"validated_tail_sha256"
|
|
1068
|
+
] != _v4_tail_seal(state):
|
|
1069
|
+
raise _v4_evidence_corrupt("validated tail seal differs from checkpoint coordinates")
|
|
1070
|
+
budget = state["budget"]
|
|
1071
|
+
budget_keys = {
|
|
1072
|
+
"requests",
|
|
1073
|
+
"responses",
|
|
1074
|
+
"pages",
|
|
1075
|
+
"records",
|
|
1076
|
+
"network_bytes",
|
|
1077
|
+
"elapsed_ns",
|
|
1078
|
+
"skipped_records",
|
|
1079
|
+
}
|
|
1080
|
+
if (
|
|
1081
|
+
type(state["completed"]) is not bool
|
|
1082
|
+
or type(state["pass_number"]) is not int
|
|
1083
|
+
or not 1 <= state["pass_number"] <= config.max_convergence_passes
|
|
1084
|
+
or type(state["unique_records"]) is not int
|
|
1085
|
+
or state["unique_records"] < 0
|
|
1086
|
+
or type(state["normalized_bytes"]) is not int
|
|
1087
|
+
or state["normalized_bytes"] < 0
|
|
1088
|
+
or state["provider_count"] is not None
|
|
1089
|
+
or state["count_basis"] != "not_reported"
|
|
1090
|
+
or not isinstance(budget, dict)
|
|
1091
|
+
or set(budget) != budget_keys
|
|
1092
|
+
or any(type(budget[name]) is not int or budget[name] < 0 for name in budget_keys)
|
|
1093
|
+
or (state["reason_code"] is not None and not isinstance(state["reason_code"], str))
|
|
1094
|
+
or (state["wait_reason"] is not None and not isinstance(state["wait_reason"], str))
|
|
1095
|
+
or (
|
|
1096
|
+
state["resume_not_before_epoch_seconds"] is not None
|
|
1097
|
+
and (
|
|
1098
|
+
type(state["resume_not_before_epoch_seconds"]) is not int
|
|
1099
|
+
or state["resume_not_before_epoch_seconds"] < 0
|
|
1100
|
+
or state["resume_not_before_epoch_seconds"] > _MAX_SAFE_EPOCH_SECONDS
|
|
1101
|
+
or state["wait_reason"] is None
|
|
1102
|
+
)
|
|
1103
|
+
)
|
|
1104
|
+
or (state["wait_reason"] is not None and state["resume_not_before_epoch_seconds"] is None)
|
|
1105
|
+
):
|
|
1106
|
+
raise _v4_evidence_corrupt("v4 scalar state is invalid")
|
|
1107
|
+
transport_overrun = (
|
|
1108
|
+
budget["requests"] > config.limits.max_requests
|
|
1109
|
+
or budget["responses"] > config.limits.max_responses
|
|
1110
|
+
or budget["network_bytes"] > config.limits.max_aggregate_bytes
|
|
1111
|
+
or budget["elapsed_ns"] > config.limits.max_wall_time_ns
|
|
1112
|
+
)
|
|
1113
|
+
if (
|
|
1114
|
+
budget["pages"] > config.limits.max_pages
|
|
1115
|
+
or budget["records"] > config.limits.max_records
|
|
1116
|
+
or state["normalized_bytes"]
|
|
1117
|
+
> config.limits.max_aggregate_bytes * DATAGOV_V4_MAX_NORMALIZED_EXPANSION
|
|
1118
|
+
or state["unique_records"] > budget["records"]
|
|
1119
|
+
or (
|
|
1120
|
+
transport_overrun
|
|
1121
|
+
and not (
|
|
1122
|
+
(
|
|
1123
|
+
state["reason_code"] is None
|
|
1124
|
+
and not state["completed"]
|
|
1125
|
+
and state["inflight_request"] is None
|
|
1126
|
+
and state["pending_response"] is not None
|
|
1127
|
+
)
|
|
1128
|
+
or (
|
|
1129
|
+
state["reason_code"] in _V4_POST_RESPONSE_OVERRUN_REASONS
|
|
1130
|
+
and not state["completed"]
|
|
1131
|
+
and state["inflight_request"] is None
|
|
1132
|
+
and state["pending_response"] is None
|
|
1133
|
+
)
|
|
1134
|
+
)
|
|
1135
|
+
)
|
|
1136
|
+
):
|
|
1137
|
+
raise _v4_evidence_corrupt("v4 budget exceeds configuration")
|
|
1138
|
+
if state["reason_code"] in _V4_IDEMPOTENT_TERMINAL_REASONS and (
|
|
1139
|
+
state["inflight_request"] is not None or state["pending_response"] is not None
|
|
1140
|
+
):
|
|
1141
|
+
raise _v4_evidence_corrupt("terminal stop retains an unresolved journal")
|
|
1142
|
+
rate_value = state["last_rate_observation"]
|
|
1143
|
+
try:
|
|
1144
|
+
if rate_value is not None:
|
|
1145
|
+
if not isinstance(rate_value, dict):
|
|
1146
|
+
raise TypeError
|
|
1147
|
+
DatagovRateObservation(**rate_value)
|
|
1148
|
+
except (AcquisitionSecurityError, TypeError):
|
|
1149
|
+
raise _v4_evidence_corrupt("last rate observation is invalid") from None
|
|
1150
|
+
if state["handoff_predecessor_sha256"] is not None and not _SHA256.fullmatch(
|
|
1151
|
+
state["handoff_predecessor_sha256"]
|
|
1152
|
+
):
|
|
1153
|
+
raise _v4_evidence_corrupt("handoff predecessor is invalid")
|
|
1154
|
+
for coordinate in (
|
|
1155
|
+
"previous_pass_map_sha256",
|
|
1156
|
+
"final_map_sha256",
|
|
1157
|
+
"last_pass_manifest_sha256",
|
|
1158
|
+
"pass_cursor_set_sha256",
|
|
1159
|
+
"pass_response_set_sha256",
|
|
1160
|
+
):
|
|
1161
|
+
if state[coordinate] is not None and not _SHA256.fullmatch(state[coordinate]):
|
|
1162
|
+
raise _v4_evidence_corrupt(f"{coordinate} is invalid")
|
|
1163
|
+
if (
|
|
1164
|
+
state["config_sha256"] != config.digest
|
|
1165
|
+
or state["endpoint"] != config.endpoint
|
|
1166
|
+
or state["harvester_coordinate"] != config.coordinate
|
|
1167
|
+
or state["predecessor_sha256"] != config.predecessor_sha256
|
|
1168
|
+
):
|
|
1169
|
+
raise CatalogFillRefused(
|
|
1170
|
+
"FILL_CHECKPOINT_CONFIG", "fill.staging.state", "checkpoint coordinate differs"
|
|
1171
|
+
)
|
|
1172
|
+
next_cursor = _cursor_from_dict(state["next_cursor"])
|
|
1173
|
+
indexes = state["evidence_indexes"]
|
|
1174
|
+
if not isinstance(indexes, dict) or set(indexes) != set(_V4_EVIDENCE_STREAMS):
|
|
1175
|
+
raise CatalogFillRefused(
|
|
1176
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.indexes", "evidence indexes differ"
|
|
1177
|
+
)
|
|
1178
|
+
if any(
|
|
1179
|
+
not isinstance(indexes[name], dict) or type(indexes[name].get("count")) is not int
|
|
1180
|
+
for name in _V4_EVIDENCE_STREAMS
|
|
1181
|
+
):
|
|
1182
|
+
raise _v4_evidence_corrupt("evidence index head is invalid")
|
|
1183
|
+
if not isinstance(state["partition_run_index"], dict):
|
|
1184
|
+
raise _v4_evidence_corrupt("partition index head is invalid")
|
|
1185
|
+
expected_counts = {
|
|
1186
|
+
"request": budget["requests"],
|
|
1187
|
+
"response": budget["responses"],
|
|
1188
|
+
"page": budget["pages"],
|
|
1189
|
+
"record": budget["records"],
|
|
1190
|
+
"cursor_transition": budget["pages"],
|
|
1191
|
+
}
|
|
1192
|
+
if any(indexes[name]["count"] != count for name, count in expected_counts.items()):
|
|
1193
|
+
raise CatalogFillRefused(
|
|
1194
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.indexes", "evidence counts differ"
|
|
1195
|
+
)
|
|
1196
|
+
if state["completed"] and (
|
|
1197
|
+
state["pass_number"] < 2
|
|
1198
|
+
or state["final_map_sha256"] is None
|
|
1199
|
+
or state["final_map_sha256"] != state["previous_pass_map_sha256"]
|
|
1200
|
+
or state["last_pass_manifest_sha256"] is None
|
|
1201
|
+
or state["next_cursor"] is not None
|
|
1202
|
+
or state["inflight_request"] is not None
|
|
1203
|
+
or state["pending_response"] is not None
|
|
1204
|
+
or state["resume_not_before_epoch_seconds"] is not None
|
|
1205
|
+
or state["wait_reason"] is not None
|
|
1206
|
+
or state["reason_code"] is not None
|
|
1207
|
+
):
|
|
1208
|
+
raise _v4_evidence_corrupt("completed state lacks converged terminal coordinates")
|
|
1209
|
+
if (
|
|
1210
|
+
not state["completed"]
|
|
1211
|
+
and state["final_map_sha256"] is not None
|
|
1212
|
+
and (
|
|
1213
|
+
state["reason_code"] not in _V4_MAP_SEALING_REASONS
|
|
1214
|
+
or state["pass_number"] != config.max_convergence_passes
|
|
1215
|
+
or state["last_pass_manifest_sha256"] is None
|
|
1216
|
+
or state["next_cursor"] is not None
|
|
1217
|
+
or state["inflight_request"] is not None
|
|
1218
|
+
or state["pending_response"] is not None
|
|
1219
|
+
or state["resume_not_before_epoch_seconds"] is not None
|
|
1220
|
+
or state["wait_reason"] is not None
|
|
1221
|
+
)
|
|
1222
|
+
):
|
|
1223
|
+
raise _v4_evidence_corrupt("map-sealing state lacks terminal max-pass coordinates")
|
|
1224
|
+
if state["reason_code"] in _V4_MAP_SEALING_REASONS and state["final_map_sha256"] is None:
|
|
1225
|
+
raise _v4_evidence_corrupt("map-sealing state lacks its final pass map")
|
|
1226
|
+
if (
|
|
1227
|
+
type(state["pass_start_run_count"]) is not int
|
|
1228
|
+
or not 0 <= state["pass_start_run_count"] <= state["partition_run_index"]["count"]
|
|
1229
|
+
):
|
|
1230
|
+
raise CatalogFillRefused(
|
|
1231
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.partition", "pass run offset is invalid"
|
|
1232
|
+
)
|
|
1233
|
+
inflight = state["inflight_request"]
|
|
1234
|
+
pending = state["pending_response"]
|
|
1235
|
+
if inflight is not None and pending is not None:
|
|
1236
|
+
raise _v4_evidence_corrupt("request and response journals overlap")
|
|
1237
|
+
if inflight is not None:
|
|
1238
|
+
if (
|
|
1239
|
+
not isinstance(inflight, dict)
|
|
1240
|
+
or set(inflight)
|
|
1241
|
+
!= {
|
|
1242
|
+
"authorized_predecessor_sha256",
|
|
1243
|
+
"attempt",
|
|
1244
|
+
"request_url",
|
|
1245
|
+
"cursor",
|
|
1246
|
+
"started",
|
|
1247
|
+
"started_ns",
|
|
1248
|
+
"response_observed_ns",
|
|
1249
|
+
"response_observed_epoch_seconds",
|
|
1250
|
+
"observed_response",
|
|
1251
|
+
}
|
|
1252
|
+
or not _SHA256.fullmatch(inflight["authorized_predecessor_sha256"] or "")
|
|
1253
|
+
or inflight["attempt"] != budget["requests"] + 1
|
|
1254
|
+
or not _valid_v4_request_url(
|
|
1255
|
+
inflight["request_url"], page_size=config.page_size, sort=config.sort
|
|
1256
|
+
)
|
|
1257
|
+
or type(inflight["started"]) is not bool
|
|
1258
|
+
or (
|
|
1259
|
+
inflight["started"]
|
|
1260
|
+
and (type(inflight["started_ns"]) is not int or inflight["started_ns"] < 0)
|
|
1261
|
+
)
|
|
1262
|
+
or (not inflight["started"] and inflight["started_ns"] is not None)
|
|
1263
|
+
or (
|
|
1264
|
+
inflight["response_observed_ns"] is not None
|
|
1265
|
+
and (
|
|
1266
|
+
type(inflight["response_observed_ns"]) is not int
|
|
1267
|
+
or not inflight["started"]
|
|
1268
|
+
or inflight["response_observed_ns"] < inflight["started_ns"]
|
|
1269
|
+
)
|
|
1270
|
+
)
|
|
1271
|
+
or (
|
|
1272
|
+
inflight["response_observed_epoch_seconds"] is not None
|
|
1273
|
+
and (
|
|
1274
|
+
type(inflight["response_observed_epoch_seconds"]) is not int
|
|
1275
|
+
or inflight["response_observed_epoch_seconds"] < 0
|
|
1276
|
+
or inflight["response_observed_epoch_seconds"] > _MAX_SAFE_EPOCH_SECONDS
|
|
1277
|
+
or inflight["response_observed_ns"] is None
|
|
1278
|
+
)
|
|
1279
|
+
)
|
|
1280
|
+
):
|
|
1281
|
+
raise _v4_evidence_corrupt("inflight request journal is invalid")
|
|
1282
|
+
try:
|
|
1283
|
+
inflight_cursor = _cursor_from_dict(inflight["cursor"])
|
|
1284
|
+
_sort, _page_size, request_cursor = _parse_v4_request_url(inflight["request_url"])
|
|
1285
|
+
except (CatalogHarvestError, TypeError, ValueError):
|
|
1286
|
+
raise _v4_evidence_corrupt("inflight cursor is invalid") from None
|
|
1287
|
+
if inflight_cursor != request_cursor or inflight_cursor != next_cursor:
|
|
1288
|
+
raise _v4_evidence_corrupt("inflight cursor differs from request/state")
|
|
1289
|
+
observed = inflight["observed_response"]
|
|
1290
|
+
if (observed is None) != (inflight["response_observed_ns"] is None) or (
|
|
1291
|
+
observed is None
|
|
1292
|
+
) != (inflight["response_observed_epoch_seconds"] is None):
|
|
1293
|
+
raise _v4_evidence_corrupt("response observation timing differs")
|
|
1294
|
+
if observed is not None and (
|
|
1295
|
+
not isinstance(observed, dict)
|
|
1296
|
+
or set(observed)
|
|
1297
|
+
!= {
|
|
1298
|
+
"status",
|
|
1299
|
+
"responses",
|
|
1300
|
+
"content_bytes",
|
|
1301
|
+
"network_bytes",
|
|
1302
|
+
"transport_evidence_digest",
|
|
1303
|
+
"rate_observation",
|
|
1304
|
+
}
|
|
1305
|
+
or type(observed["status"]) is not int
|
|
1306
|
+
or not 100 <= observed["status"] <= 599
|
|
1307
|
+
or observed["responses"] != 1
|
|
1308
|
+
or type(observed["content_bytes"]) is not int
|
|
1309
|
+
or type(observed["network_bytes"]) is not int
|
|
1310
|
+
or observed["content_bytes"] < 0
|
|
1311
|
+
or observed["network_bytes"] < observed["content_bytes"]
|
|
1312
|
+
or (
|
|
1313
|
+
observed["transport_evidence_digest"] is not None
|
|
1314
|
+
and not _SHA256.fullmatch(observed["transport_evidence_digest"])
|
|
1315
|
+
)
|
|
1316
|
+
):
|
|
1317
|
+
raise _v4_evidence_corrupt("observed response journal is invalid")
|
|
1318
|
+
if observed is not None:
|
|
1319
|
+
try:
|
|
1320
|
+
rate_value = observed["rate_observation"]
|
|
1321
|
+
if rate_value is not None:
|
|
1322
|
+
if not isinstance(rate_value, dict):
|
|
1323
|
+
raise TypeError
|
|
1324
|
+
DatagovRateObservation(**rate_value)
|
|
1325
|
+
except (AcquisitionSecurityError, TypeError):
|
|
1326
|
+
raise _v4_evidence_corrupt("observed response rate is invalid") from None
|
|
1327
|
+
if pending is not None:
|
|
1328
|
+
if (
|
|
1329
|
+
not isinstance(pending, dict)
|
|
1330
|
+
or set(pending)
|
|
1331
|
+
!= {
|
|
1332
|
+
"authorized_predecessor_sha256",
|
|
1333
|
+
"request_url",
|
|
1334
|
+
"cursor",
|
|
1335
|
+
"raw_response_sha256",
|
|
1336
|
+
"evidence",
|
|
1337
|
+
"response_evidence",
|
|
1338
|
+
}
|
|
1339
|
+
or not _SHA256.fullmatch(pending["authorized_predecessor_sha256"] or "")
|
|
1340
|
+
or not _SHA256.fullmatch(pending["raw_response_sha256"] or "")
|
|
1341
|
+
or not _valid_v4_request_url(
|
|
1342
|
+
pending["request_url"], page_size=config.page_size, sort=config.sort
|
|
1343
|
+
)
|
|
1344
|
+
):
|
|
1345
|
+
raise _v4_evidence_corrupt("pending response journal is invalid")
|
|
1346
|
+
try:
|
|
1347
|
+
pending_cursor = _cursor_from_dict(pending["cursor"])
|
|
1348
|
+
_sort, _page_size, request_cursor = _parse_v4_request_url(pending["request_url"])
|
|
1349
|
+
evidence = _evidence_from_dict(pending["evidence"])
|
|
1350
|
+
parsed = _response_evidence_from_dict(pending["response_evidence"])
|
|
1351
|
+
except (CatalogHarvestError, AcquisitionSecurityError, TypeError, ValueError):
|
|
1352
|
+
raise _v4_evidence_corrupt("pending response contract is invalid") from None
|
|
1353
|
+
if (
|
|
1354
|
+
pending_cursor != request_cursor
|
|
1355
|
+
or pending_cursor != next_cursor
|
|
1356
|
+
or evidence.uri != parsed.final_url
|
|
1357
|
+
or evidence.content_sha256 != pending["raw_response_sha256"]
|
|
1358
|
+
or evidence.media_type != "application/json"
|
|
1359
|
+
or parsed.content_sha256 != pending["raw_response_sha256"]
|
|
1360
|
+
or parsed.media_type != "application/json"
|
|
1361
|
+
or parsed.requests != 1
|
|
1362
|
+
or parsed.responses != 1
|
|
1363
|
+
or parsed.network_bytes != parsed.content_bytes
|
|
1364
|
+
):
|
|
1365
|
+
raise _v4_evidence_corrupt("pending response transport closure differs")
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
def _append_v4_index(
|
|
1369
|
+
staging: FillStaging,
|
|
1370
|
+
state: dict[str, Any],
|
|
1371
|
+
stream: str,
|
|
1372
|
+
items: list[dict[str, Any]],
|
|
1373
|
+
) -> dict[str, Any]:
|
|
1374
|
+
if not items:
|
|
1375
|
+
return state
|
|
1376
|
+
next_state = dict(state)
|
|
1377
|
+
indexes = dict(state["evidence_indexes"])
|
|
1378
|
+
indexes[stream] = staging.append_index(indexes[stream], stream=stream, items=items)
|
|
1379
|
+
next_state["evidence_indexes"] = indexes
|
|
1380
|
+
return next_state
|
|
1381
|
+
|
|
1382
|
+
|
|
1383
|
+
def _write_v4_normalized_page(
|
|
1384
|
+
staging: FillStaging,
|
|
1385
|
+
*,
|
|
1386
|
+
raw_response_sha256: str,
|
|
1387
|
+
records: list[dict[str, Any]],
|
|
1388
|
+
skipped_record_ids: list[str],
|
|
1389
|
+
maximum_bytes: int = DATAGOV_V4_MAX_NORMALIZED_PAGE_BYTES,
|
|
1390
|
+
) -> tuple[str, int]:
|
|
1391
|
+
"""Install one bounded page manifest over adaptively split normalized record shards."""
|
|
1392
|
+
|
|
1393
|
+
normalized_bytes = 0
|
|
1394
|
+
planned_segments: list[tuple[dict[str, Any], str]] = []
|
|
1395
|
+
|
|
1396
|
+
def plan_segment(first_record: int, values: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1397
|
+
nonlocal normalized_bytes
|
|
1398
|
+
payload = {
|
|
1399
|
+
"schema_version": "harness-datagov-v4-normalized-record-segment.v1",
|
|
1400
|
+
"raw_response_sha256": raw_response_sha256,
|
|
1401
|
+
"first_record": first_record,
|
|
1402
|
+
"records": values,
|
|
1403
|
+
}
|
|
1404
|
+
try:
|
|
1405
|
+
raw = _v4_normalized_shard_bytes(payload)
|
|
1406
|
+
except CanonicalJSONError:
|
|
1407
|
+
if len(values) <= 1:
|
|
1408
|
+
raise CatalogFillRefused(
|
|
1409
|
+
"DATAGOV_NORMALIZED_LIMIT",
|
|
1410
|
+
"fill.normalized",
|
|
1411
|
+
"one normalized record exceeds the canonical shard contract",
|
|
1412
|
+
) from None
|
|
1413
|
+
midpoint = len(values) // 2
|
|
1414
|
+
return plan_segment(first_record, values[:midpoint]) + plan_segment(
|
|
1415
|
+
first_record + midpoint, values[midpoint:]
|
|
1416
|
+
)
|
|
1417
|
+
if len(raw) > shard_byte_limit("normalized"):
|
|
1418
|
+
if len(values) <= 1:
|
|
1419
|
+
raise CatalogFillRefused(
|
|
1420
|
+
"DATAGOV_NORMALIZED_LIMIT",
|
|
1421
|
+
"fill.normalized",
|
|
1422
|
+
"one normalized record exceeds the shard byte contract",
|
|
1423
|
+
)
|
|
1424
|
+
midpoint = len(values) // 2
|
|
1425
|
+
return plan_segment(first_record, values[:midpoint]) + plan_segment(
|
|
1426
|
+
first_record + midpoint, values[midpoint:]
|
|
1427
|
+
)
|
|
1428
|
+
if normalized_bytes + len(raw) > maximum_bytes:
|
|
1429
|
+
raise CatalogFillRefused(
|
|
1430
|
+
"DATAGOV_NORMALIZED_LIMIT",
|
|
1431
|
+
"fill.normalized",
|
|
1432
|
+
"normalized page exceeds its aggregate staging byte budget",
|
|
1433
|
+
)
|
|
1434
|
+
digest = sha256_bytes(raw)
|
|
1435
|
+
normalized_bytes += len(raw)
|
|
1436
|
+
planned_segments.append((payload, digest))
|
|
1437
|
+
return [{"sha256": digest, "first_record": first_record, "record_count": len(values)}]
|
|
1438
|
+
|
|
1439
|
+
segments: list[dict[str, Any]] = []
|
|
1440
|
+
for first_record in range(0, len(records), 64):
|
|
1441
|
+
segments.extend(plan_segment(first_record, records[first_record : first_record + 64]))
|
|
1442
|
+
manifest = {
|
|
1443
|
+
"schema_version": "harness-datagov-v4-normalized-page.v2",
|
|
1444
|
+
"raw_response_sha256": raw_response_sha256,
|
|
1445
|
+
"record_count": len(records),
|
|
1446
|
+
"record_segments": segments,
|
|
1447
|
+
"skipped_record_ids": skipped_record_ids,
|
|
1448
|
+
}
|
|
1449
|
+
manifest_raw = _v4_normalized_shard_bytes(manifest)
|
|
1450
|
+
if (
|
|
1451
|
+
len(manifest_raw) > shard_byte_limit("normalized")
|
|
1452
|
+
or normalized_bytes + len(manifest_raw) > maximum_bytes
|
|
1453
|
+
):
|
|
1454
|
+
raise CatalogFillRefused(
|
|
1455
|
+
"DATAGOV_NORMALIZED_LIMIT",
|
|
1456
|
+
"fill.normalized",
|
|
1457
|
+
"normalized page exceeds its aggregate staging byte budget",
|
|
1458
|
+
)
|
|
1459
|
+
for payload, expected_digest in planned_segments:
|
|
1460
|
+
if staging.write_shard("normalized", payload) != expected_digest:
|
|
1461
|
+
raise _v4_evidence_corrupt("normalized segment write digest differs")
|
|
1462
|
+
digest = staging.write_shard("normalized", manifest)
|
|
1463
|
+
return digest, normalized_bytes + len(manifest_raw)
|
|
1464
|
+
|
|
1465
|
+
|
|
1466
|
+
def _v4_normalized_shard_bytes(payload: dict[str, Any]) -> bytes:
|
|
1467
|
+
return canonical_json_bytes(
|
|
1468
|
+
{"schema_version": SHARD_SCHEMA, "kind": "normalized", "payload": payload}
|
|
1469
|
+
)
|
|
1470
|
+
|
|
1471
|
+
|
|
1472
|
+
def _v4_normalized_shard_size(payload: dict[str, Any]) -> int:
|
|
1473
|
+
return len(_v4_normalized_shard_bytes(payload))
|
|
1474
|
+
|
|
1475
|
+
|
|
1476
|
+
def _v4_tail_seal(state: dict[str, Any]) -> str:
|
|
1477
|
+
"""Bind the producer-validated immutable tails without self-reference."""
|
|
1478
|
+
|
|
1479
|
+
excluded = {
|
|
1480
|
+
"generation",
|
|
1481
|
+
"schema_version",
|
|
1482
|
+
"handoff_predecessor_sha256",
|
|
1483
|
+
"validated_tail_sha256",
|
|
1484
|
+
}
|
|
1485
|
+
return canonical_sha256({key: value for key, value in state.items() if key not in excluded})
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
def _v4_cache_key(staging: FillStaging) -> str:
|
|
1489
|
+
return str(staging.root.resolve(strict=False))
|
|
1490
|
+
|
|
1491
|
+
|
|
1492
|
+
def _v4_cached_checkpoint_valid(staging: FillStaging, digest: str) -> bool:
|
|
1493
|
+
"""Trust an ordinary resume only while the retained object tree is unchanged."""
|
|
1494
|
+
|
|
1495
|
+
key = _v4_cache_key(staging)
|
|
1496
|
+
cached = _V4_VALIDATED_CHECKPOINTS.get(key)
|
|
1497
|
+
if cached is None or cached[0] != digest or cached[1] != staging.integrity_epoch():
|
|
1498
|
+
_V4_VALIDATED_CHECKPOINTS.pop(key, None)
|
|
1499
|
+
return False
|
|
1500
|
+
return True
|
|
1501
|
+
|
|
1502
|
+
|
|
1503
|
+
def _remember_v4_checkpoint(staging: FillStaging, digest: str) -> None:
|
|
1504
|
+
"""Remember one fully or producer-validated checkpoint for this process only."""
|
|
1505
|
+
|
|
1506
|
+
key = _v4_cache_key(staging)
|
|
1507
|
+
_V4_VALIDATED_CHECKPOINTS.pop(key, None)
|
|
1508
|
+
_V4_VALIDATED_CHECKPOINTS[key] = (digest, staging.integrity_epoch())
|
|
1509
|
+
while len(_V4_VALIDATED_CHECKPOINTS) > _V4_VALIDATED_CACHE_LIMIT:
|
|
1510
|
+
del _V4_VALIDATED_CHECKPOINTS[next(iter(_V4_VALIDATED_CHECKPOINTS))]
|
|
1511
|
+
|
|
1512
|
+
|
|
1513
|
+
def _v4_commit(staging: FillStaging, state: dict[str, Any]) -> dict[str, Any]:
|
|
1514
|
+
next_state = dict(state)
|
|
1515
|
+
next_state["validated_tail_sha256"] = _v4_tail_seal(next_state)
|
|
1516
|
+
return staging.commit(
|
|
1517
|
+
next_state,
|
|
1518
|
+
expected_generation=state["generation"],
|
|
1519
|
+
expected_checkpoint_digest=staging.state_digest(),
|
|
1520
|
+
)
|
|
1521
|
+
|
|
1522
|
+
|
|
1523
|
+
def _v4_journal_predecessor(state: dict[str, Any]) -> str | None:
|
|
1524
|
+
for name in ("pending_response", "inflight_request"):
|
|
1525
|
+
journal = state.get(name)
|
|
1526
|
+
if isinstance(journal, dict):
|
|
1527
|
+
value = journal.get("authorized_predecessor_sha256")
|
|
1528
|
+
if isinstance(value, str):
|
|
1529
|
+
return value
|
|
1530
|
+
return None
|
|
1531
|
+
|
|
1532
|
+
|
|
1533
|
+
def _v4_elapsed_ns(started_ns: Any, ended_ns: Any) -> int:
|
|
1534
|
+
if (
|
|
1535
|
+
type(started_ns) is not int
|
|
1536
|
+
or type(ended_ns) is not int
|
|
1537
|
+
or started_ns < 0
|
|
1538
|
+
or ended_ns < started_ns
|
|
1539
|
+
):
|
|
1540
|
+
raise CatalogFillRefused(
|
|
1541
|
+
"DATAGOV_CLOCK_DISCONTINUITY",
|
|
1542
|
+
"fill.clock_ns",
|
|
1543
|
+
"monotonic request timing moved backward or changed origin",
|
|
1544
|
+
)
|
|
1545
|
+
return ended_ns - started_ns
|
|
1546
|
+
|
|
1547
|
+
|
|
1548
|
+
def _promote_v4_inflight(
|
|
1549
|
+
staging: FillStaging,
|
|
1550
|
+
state: dict[str, Any],
|
|
1551
|
+
*,
|
|
1552
|
+
elapsed_ns: int,
|
|
1553
|
+
observed_epoch_seconds: int,
|
|
1554
|
+
) -> dict[str, Any]:
|
|
1555
|
+
inflight = state.get("inflight_request")
|
|
1556
|
+
if not isinstance(inflight, dict) or not inflight.get("started"):
|
|
1557
|
+
raise _v4_evidence_corrupt("started request journal is absent")
|
|
1558
|
+
next_state = _append_v4_index(
|
|
1559
|
+
staging,
|
|
1560
|
+
dict(state),
|
|
1561
|
+
"request",
|
|
1562
|
+
[
|
|
1563
|
+
{
|
|
1564
|
+
"attempt": inflight["attempt"],
|
|
1565
|
+
"request_url": inflight["request_url"],
|
|
1566
|
+
"cursor": inflight["cursor"],
|
|
1567
|
+
"elapsed_ns": elapsed_ns,
|
|
1568
|
+
"observed_epoch_seconds": observed_epoch_seconds,
|
|
1569
|
+
}
|
|
1570
|
+
],
|
|
1571
|
+
)
|
|
1572
|
+
budget = dict(next_state["budget"])
|
|
1573
|
+
budget["requests"] += 1
|
|
1574
|
+
next_state["budget"] = budget
|
|
1575
|
+
next_state["inflight_request"] = None
|
|
1576
|
+
return next_state
|
|
1577
|
+
|
|
1578
|
+
|
|
1579
|
+
def _reconcile_v4_inflight(
|
|
1580
|
+
staging: FillStaging,
|
|
1581
|
+
state: dict[str, Any],
|
|
1582
|
+
config: CatalogFillConfig,
|
|
1583
|
+
clock_ns: Callable[[], int],
|
|
1584
|
+
wall_clock_seconds: Callable[[], int],
|
|
1585
|
+
) -> tuple[dict[str, Any], str | None]:
|
|
1586
|
+
inflight = state["inflight_request"]
|
|
1587
|
+
if not inflight["started"]:
|
|
1588
|
+
state = dict(state)
|
|
1589
|
+
state["inflight_request"] = None
|
|
1590
|
+
return _v4_commit(staging, state), None
|
|
1591
|
+
observed = inflight["observed_response"]
|
|
1592
|
+
ended_ns = _clock(clock_ns) if observed is None else inflight["response_observed_ns"]
|
|
1593
|
+
observed_epoch_seconds = (
|
|
1594
|
+
_wall_clock(wall_clock_seconds)
|
|
1595
|
+
if observed is None
|
|
1596
|
+
else inflight["response_observed_epoch_seconds"]
|
|
1597
|
+
)
|
|
1598
|
+
elapsed_ns = _v4_elapsed_ns(inflight["started_ns"], ended_ns)
|
|
1599
|
+
state = _promote_v4_inflight(
|
|
1600
|
+
staging,
|
|
1601
|
+
state,
|
|
1602
|
+
elapsed_ns=elapsed_ns,
|
|
1603
|
+
observed_epoch_seconds=observed_epoch_seconds,
|
|
1604
|
+
)
|
|
1605
|
+
if observed is None:
|
|
1606
|
+
budget = dict(state["budget"])
|
|
1607
|
+
budget["elapsed_ns"] += elapsed_ns
|
|
1608
|
+
state["budget"] = budget
|
|
1609
|
+
state["resume_not_before_epoch_seconds"] = _pacing_wait(config, observed_epoch_seconds)
|
|
1610
|
+
state["wait_reason"] = (
|
|
1611
|
+
"DATAGOV_PACING_REQUIRED"
|
|
1612
|
+
if state["resume_not_before_epoch_seconds"] is not None
|
|
1613
|
+
else None
|
|
1614
|
+
)
|
|
1615
|
+
reason = "DATAGOV_REQUEST_OUTCOME_UNKNOWN"
|
|
1616
|
+
return _v4_stop(staging, state, reason), reason
|
|
1617
|
+
reason = "DATAGOV_RESPONSE_SCAN_INTERRUPTED"
|
|
1618
|
+
redacted = {
|
|
1619
|
+
key: value
|
|
1620
|
+
for key, value in observed.items()
|
|
1621
|
+
if key not in {"responses", "transport_evidence_digest"}
|
|
1622
|
+
}
|
|
1623
|
+
redacted["code"] = reason
|
|
1624
|
+
redacted["content_bytes"] = 0
|
|
1625
|
+
state = _append_v4_index(
|
|
1626
|
+
staging,
|
|
1627
|
+
state,
|
|
1628
|
+
"response",
|
|
1629
|
+
[
|
|
1630
|
+
{
|
|
1631
|
+
"request_attempt": inflight["attempt"],
|
|
1632
|
+
"request_url": inflight["request_url"],
|
|
1633
|
+
"observed_epoch_seconds": observed_epoch_seconds,
|
|
1634
|
+
"refusal": redacted,
|
|
1635
|
+
}
|
|
1636
|
+
],
|
|
1637
|
+
)
|
|
1638
|
+
budget = dict(state["budget"])
|
|
1639
|
+
budget["responses"] += observed["responses"]
|
|
1640
|
+
budget["network_bytes"] += observed["network_bytes"]
|
|
1641
|
+
budget["elapsed_ns"] += elapsed_ns
|
|
1642
|
+
state["budget"] = budget
|
|
1643
|
+
rate_value = observed["rate_observation"]
|
|
1644
|
+
rate = DatagovRateObservation(**rate_value) if isinstance(rate_value, dict) else None
|
|
1645
|
+
state["last_rate_observation"] = rate_value
|
|
1646
|
+
wait_until, wait_reason = _v4_response_wait(
|
|
1647
|
+
config, observed["status"], rate, observed_epoch_seconds
|
|
1648
|
+
)
|
|
1649
|
+
state["resume_not_before_epoch_seconds"] = wait_until
|
|
1650
|
+
state["wait_reason"] = wait_reason
|
|
1651
|
+
return _v4_stop(staging, state, reason), reason
|
|
1652
|
+
|
|
1653
|
+
|
|
1654
|
+
def _clear_v4_pending_and_stop(
|
|
1655
|
+
staging: FillStaging, state: dict[str, Any], reason: str
|
|
1656
|
+
) -> dict[str, Any]:
|
|
1657
|
+
state = dict(state)
|
|
1658
|
+
state["pending_response"] = None
|
|
1659
|
+
return _v4_stop(staging, state, reason)
|
|
1660
|
+
|
|
1661
|
+
|
|
1662
|
+
def _evidence_from_dict(value: Any) -> EvidenceReference:
|
|
1663
|
+
if not isinstance(value, dict) or set(value) != {
|
|
1664
|
+
"uri",
|
|
1665
|
+
"observed_at",
|
|
1666
|
+
"content_sha256",
|
|
1667
|
+
"media_type",
|
|
1668
|
+
}:
|
|
1669
|
+
raise _v4_evidence_corrupt("pending evidence is invalid")
|
|
1670
|
+
return EvidenceReference(**value)
|
|
1671
|
+
|
|
1672
|
+
|
|
1673
|
+
def _response_evidence_from_dict(value: Any) -> HarvestResponseEvidence:
|
|
1674
|
+
if not isinstance(value, dict):
|
|
1675
|
+
raise _v4_evidence_corrupt("pending response evidence is invalid")
|
|
1676
|
+
payload = dict(value)
|
|
1677
|
+
rate_value = payload.pop("rate_observation", None)
|
|
1678
|
+
status = payload.pop("status", 200)
|
|
1679
|
+
if rate_value is not None and not isinstance(rate_value, dict):
|
|
1680
|
+
raise _v4_evidence_corrupt("pending response rate observation is invalid")
|
|
1681
|
+
rate = DatagovRateObservation(**rate_value) if rate_value is not None else None
|
|
1682
|
+
return HarvestResponseEvidence(**payload, status=status, rate_observation=rate)
|
|
1683
|
+
|
|
1684
|
+
|
|
1685
|
+
def _v4_stop(staging: FillStaging, state: dict[str, Any], reason: str) -> dict[str, Any]:
|
|
1686
|
+
if state.get("reason_code") == reason:
|
|
1687
|
+
return state
|
|
1688
|
+
next_state = dict(state)
|
|
1689
|
+
next_state["reason_code"] = reason
|
|
1690
|
+
return _v4_commit(staging, next_state)
|
|
1691
|
+
|
|
1692
|
+
|
|
1693
|
+
def _v4_result(
|
|
1694
|
+
staging: FillStaging,
|
|
1695
|
+
state: dict[str, Any],
|
|
1696
|
+
*,
|
|
1697
|
+
config: CatalogFillConfig,
|
|
1698
|
+
reason: str | None,
|
|
1699
|
+
) -> CatalogFillResult:
|
|
1700
|
+
if state["completed"] or reason in _V4_MAP_SEALING_REASONS:
|
|
1701
|
+
_validate_v4_evidence(staging, state, config=config)
|
|
1702
|
+
receipt = CatalogFillReceipt(
|
|
1703
|
+
config_sha256=state["config_sha256"],
|
|
1704
|
+
endpoint=state["endpoint"],
|
|
1705
|
+
harvester_coordinate=state["harvester_coordinate"],
|
|
1706
|
+
predecessor_sha256=state["predecessor_sha256"],
|
|
1707
|
+
status="complete" if state["completed"] else "incomplete",
|
|
1708
|
+
reason_code=reason,
|
|
1709
|
+
requests=state["budget"]["requests"],
|
|
1710
|
+
responses=state["budget"]["responses"],
|
|
1711
|
+
pages=state["budget"]["pages"],
|
|
1712
|
+
records=state["budget"]["records"],
|
|
1713
|
+
unique_records=state["unique_records"],
|
|
1714
|
+
network_bytes=state["budget"]["network_bytes"],
|
|
1715
|
+
elapsed_ns=state["budget"]["elapsed_ns"],
|
|
1716
|
+
provider_count=None,
|
|
1717
|
+
count_basis="not_reported",
|
|
1718
|
+
skipped_records=state["budget"]["skipped_records"],
|
|
1719
|
+
flagged_rights=0,
|
|
1720
|
+
convergence_passes=state["pass_number"],
|
|
1721
|
+
response_evidence_digests=(),
|
|
1722
|
+
page_shard_digests=(),
|
|
1723
|
+
flagged_queue_sha256=canonical_sha256([]),
|
|
1724
|
+
rate_observation=state["last_rate_observation"],
|
|
1725
|
+
resume_not_before_epoch_seconds=state["resume_not_before_epoch_seconds"],
|
|
1726
|
+
evidence_indexes=state["evidence_indexes"],
|
|
1727
|
+
pass_map_sha256=state["final_map_sha256"] or state["previous_pass_map_sha256"],
|
|
1728
|
+
normalized_bytes=state["normalized_bytes"],
|
|
1729
|
+
conflicting_identifiers=state["conflicting_identifiers"],
|
|
1730
|
+
)
|
|
1731
|
+
digest = staging.state_digest()
|
|
1732
|
+
_remember_v4_checkpoint(staging, digest)
|
|
1733
|
+
return CatalogFillResult(
|
|
1734
|
+
status=receipt.status,
|
|
1735
|
+
reason_code=reason,
|
|
1736
|
+
receipt=receipt,
|
|
1737
|
+
checkpoint_sha256=digest,
|
|
1738
|
+
eligible_staging_sha256=digest if state["completed"] else None,
|
|
1739
|
+
)
|
|
1740
|
+
|
|
1741
|
+
|
|
1742
|
+
def _validate_v4_tail_evidence(
|
|
1743
|
+
staging: FillStaging,
|
|
1744
|
+
state: dict[str, Any],
|
|
1745
|
+
*,
|
|
1746
|
+
config: CatalogFillConfig,
|
|
1747
|
+
) -> None:
|
|
1748
|
+
"""Authenticate only current immutable tails on ordinary resumptions."""
|
|
1749
|
+
|
|
1750
|
+
indexes = state["evidence_indexes"]
|
|
1751
|
+
tails = {
|
|
1752
|
+
stream: staging.index_tail_items(indexes[stream], stream=stream)
|
|
1753
|
+
for stream in _V4_EVIDENCE_STREAMS
|
|
1754
|
+
}
|
|
1755
|
+
staging.index_tail_items(state["partition_run_index"], stream="partition_run")
|
|
1756
|
+
request_tail = tails["request"][-1] if tails["request"] else None
|
|
1757
|
+
if state["budget"]["requests"]:
|
|
1758
|
+
if (
|
|
1759
|
+
not isinstance(request_tail, dict)
|
|
1760
|
+
or request_tail.get("attempt") != state["budget"]["requests"]
|
|
1761
|
+
or not _valid_v4_request_url(
|
|
1762
|
+
request_tail.get("request_url"), page_size=config.page_size, sort=config.sort
|
|
1763
|
+
)
|
|
1764
|
+
or type(request_tail.get("observed_epoch_seconds")) is not int
|
|
1765
|
+
or not 0 <= request_tail["observed_epoch_seconds"] <= _MAX_SAFE_EPOCH_SECONDS
|
|
1766
|
+
):
|
|
1767
|
+
raise _v4_evidence_corrupt("request tail differs from checkpoint budget")
|
|
1768
|
+
response_tail = tails["response"][-1] if tails["response"] else None
|
|
1769
|
+
if response_tail is not None and (
|
|
1770
|
+
type(response_tail.get("request_attempt")) is not int
|
|
1771
|
+
or not 1 <= response_tail["request_attempt"] <= state["budget"]["requests"]
|
|
1772
|
+
or type(response_tail.get("observed_epoch_seconds")) is not int
|
|
1773
|
+
or not 0 <= response_tail["observed_epoch_seconds"] <= _MAX_SAFE_EPOCH_SECONDS
|
|
1774
|
+
):
|
|
1775
|
+
raise _v4_evidence_corrupt("response tail is invalid")
|
|
1776
|
+
page_tail = tails["page"][-1] if tails["page"] else None
|
|
1777
|
+
cursor_tail = tails["cursor_transition"][-1] if tails["cursor_transition"] else None
|
|
1778
|
+
if state["budget"]["pages"] and (
|
|
1779
|
+
not isinstance(page_tail, dict)
|
|
1780
|
+
or not isinstance(cursor_tail, dict)
|
|
1781
|
+
or page_tail.get("normalized_sha256") != cursor_tail.get("normalized_page_sha256")
|
|
1782
|
+
or cursor_tail.get("next_cursor") != state["next_cursor"]
|
|
1783
|
+
):
|
|
1784
|
+
raise _v4_evidence_corrupt("page and cursor tails differ")
|
|
1785
|
+
if page_tail is not None:
|
|
1786
|
+
observed_records = 0
|
|
1787
|
+
try:
|
|
1788
|
+
for _record in iter_v4_normalized_records(
|
|
1789
|
+
staging,
|
|
1790
|
+
page_tail["normalized_sha256"],
|
|
1791
|
+
raw_response_sha256=page_tail["raw_response_sha256"],
|
|
1792
|
+
record_count=page_tail["record_count"],
|
|
1793
|
+
):
|
|
1794
|
+
observed_records += 1
|
|
1795
|
+
except (CatalogFillRefused, CanonicalJSONError, TypeError, ValueError):
|
|
1796
|
+
raise _v4_evidence_corrupt("normalized page tail is invalid") from None
|
|
1797
|
+
if observed_records != page_tail["record_count"]:
|
|
1798
|
+
raise _v4_evidence_corrupt("normalized page tail count differs")
|
|
1799
|
+
if page_tail.get("pass") == state["pass_number"]:
|
|
1800
|
+
requested_digest = canonical_sha256(cursor_tail["request_cursor"])
|
|
1801
|
+
if not _v4_loop_contains(
|
|
1802
|
+
staging,
|
|
1803
|
+
state["pass_cursor_set_sha256"],
|
|
1804
|
+
"cursor",
|
|
1805
|
+
requested_digest,
|
|
1806
|
+
) or not _v4_loop_contains(
|
|
1807
|
+
staging,
|
|
1808
|
+
state["pass_response_set_sha256"],
|
|
1809
|
+
"response",
|
|
1810
|
+
page_tail["raw_response_sha256"],
|
|
1811
|
+
):
|
|
1812
|
+
raise _v4_evidence_corrupt("current pass loop-set omits its page tail")
|
|
1813
|
+
for set_kind, coordinate in (
|
|
1814
|
+
("cursor", state["pass_cursor_set_sha256"]),
|
|
1815
|
+
("response", state["pass_response_set_sha256"]),
|
|
1816
|
+
):
|
|
1817
|
+
if coordinate is not None:
|
|
1818
|
+
_v4_loop_node(staging, coordinate, set_kind=set_kind, depth=0, prefix="")
|
|
1819
|
+
pending = state["pending_response"]
|
|
1820
|
+
if pending is not None:
|
|
1821
|
+
try:
|
|
1822
|
+
response = _response_evidence_from_dict(pending["response_evidence"])
|
|
1823
|
+
except (CatalogHarvestError, AcquisitionSecurityError, TypeError, ValueError):
|
|
1824
|
+
raise _v4_evidence_corrupt("pending response tail is invalid") from None
|
|
1825
|
+
if (
|
|
1826
|
+
not isinstance(response_tail, dict)
|
|
1827
|
+
or response_tail.get("request_attempt") != state["budget"]["requests"]
|
|
1828
|
+
or response_tail.get("request_url") != pending["request_url"]
|
|
1829
|
+
or response_tail.get("raw_response_sha256") != pending["raw_response_sha256"]
|
|
1830
|
+
or response_tail.get("evidence") != response.to_dict()
|
|
1831
|
+
):
|
|
1832
|
+
raise _v4_evidence_corrupt("pending response differs from response tail")
|
|
1833
|
+
latest_status: int | None = None
|
|
1834
|
+
latest_rate: DatagovRateObservation | None = None
|
|
1835
|
+
latest_epoch: int | None = None
|
|
1836
|
+
latest_attempt = 0
|
|
1837
|
+
latest_full: HarvestResponseEvidence | None = None
|
|
1838
|
+
latest_refusal_code: str | None = None
|
|
1839
|
+
if response_tail is not None:
|
|
1840
|
+
latest_attempt = response_tail["request_attempt"]
|
|
1841
|
+
latest_epoch = response_tail["observed_epoch_seconds"]
|
|
1842
|
+
if "evidence" in response_tail:
|
|
1843
|
+
try:
|
|
1844
|
+
latest_full = _response_evidence_from_dict(response_tail["evidence"])
|
|
1845
|
+
except (CatalogHarvestError, AcquisitionSecurityError, TypeError, ValueError):
|
|
1846
|
+
raise _v4_evidence_corrupt("full response tail is invalid") from None
|
|
1847
|
+
latest_status = latest_full.status
|
|
1848
|
+
latest_rate = latest_full.rate_observation
|
|
1849
|
+
elif "refusal" in response_tail and isinstance(response_tail["refusal"], dict):
|
|
1850
|
+
refusal = response_tail["refusal"]
|
|
1851
|
+
latest_refusal_code = refusal.get("code")
|
|
1852
|
+
latest_status = refusal.get("status")
|
|
1853
|
+
rate_value = refusal.get("rate_observation")
|
|
1854
|
+
try:
|
|
1855
|
+
latest_rate = (
|
|
1856
|
+
DatagovRateObservation(**rate_value) if rate_value is not None else None
|
|
1857
|
+
)
|
|
1858
|
+
except (AcquisitionSecurityError, TypeError):
|
|
1859
|
+
raise _v4_evidence_corrupt("refusal response tail is invalid") from None
|
|
1860
|
+
else:
|
|
1861
|
+
raise _v4_evidence_corrupt("response tail shape is invalid")
|
|
1862
|
+
inflight = state.get("inflight_request")
|
|
1863
|
+
observed = inflight.get("observed_response") if isinstance(inflight, dict) else None
|
|
1864
|
+
if isinstance(observed, dict):
|
|
1865
|
+
latest_attempt = inflight["attempt"]
|
|
1866
|
+
latest_status = observed["status"]
|
|
1867
|
+
rate_value = observed["rate_observation"]
|
|
1868
|
+
try:
|
|
1869
|
+
latest_rate = DatagovRateObservation(**rate_value) if rate_value is not None else None
|
|
1870
|
+
except (AcquisitionSecurityError, TypeError):
|
|
1871
|
+
raise _v4_evidence_corrupt("inflight response rate is invalid") from None
|
|
1872
|
+
latest_epoch = inflight["response_observed_epoch_seconds"]
|
|
1873
|
+
expected_wait: tuple[int | None, str | None] = (None, None)
|
|
1874
|
+
try:
|
|
1875
|
+
if latest_epoch is not None and latest_attempt == (
|
|
1876
|
+
inflight["attempt"]
|
|
1877
|
+
if isinstance(inflight, dict) and observed is not None
|
|
1878
|
+
else state["budget"]["requests"]
|
|
1879
|
+
):
|
|
1880
|
+
expected_wait = _v4_response_wait(config, latest_status, latest_rate, latest_epoch)
|
|
1881
|
+
elif request_tail is not None:
|
|
1882
|
+
wait_until = _pacing_wait(config, request_tail["observed_epoch_seconds"])
|
|
1883
|
+
expected_wait = (
|
|
1884
|
+
wait_until,
|
|
1885
|
+
"DATAGOV_PACING_REQUIRED" if wait_until is not None else None,
|
|
1886
|
+
)
|
|
1887
|
+
except CatalogFillRefused:
|
|
1888
|
+
raise _v4_evidence_corrupt("tail observation time cannot produce a safe wait") from None
|
|
1889
|
+
if state["completed"] or state["reason_code"] in _V4_MAP_SEALING_REASONS:
|
|
1890
|
+
expected_wait = (None, None)
|
|
1891
|
+
if (
|
|
1892
|
+
state["resume_not_before_epoch_seconds"],
|
|
1893
|
+
state["wait_reason"],
|
|
1894
|
+
) != expected_wait:
|
|
1895
|
+
raise _v4_evidence_corrupt("durable wait differs from authenticated tail")
|
|
1896
|
+
reason = state["reason_code"]
|
|
1897
|
+
if reason in (_V4_POST_RESPONSE_OVERRUN_REASONS | _V4_PREFLIGHT_TERMINAL_REASONS):
|
|
1898
|
+
late_reason = (
|
|
1899
|
+
_late_limit(state, config.limits, latest_full) if latest_full is not None else None
|
|
1900
|
+
)
|
|
1901
|
+
prospective_reason = _v4_prospective_terminal_reason(
|
|
1902
|
+
staging,
|
|
1903
|
+
state,
|
|
1904
|
+
config=config,
|
|
1905
|
+
request_item=request_tail,
|
|
1906
|
+
response_item=response_tail,
|
|
1907
|
+
response=latest_full,
|
|
1908
|
+
)
|
|
1909
|
+
refusal_reason = (
|
|
1910
|
+
_v4_transport_terminal_reason(latest_refusal_code, state, config.limits)
|
|
1911
|
+
if isinstance(latest_refusal_code, str)
|
|
1912
|
+
else None
|
|
1913
|
+
)
|
|
1914
|
+
if reason not in {
|
|
1915
|
+
late_reason,
|
|
1916
|
+
_preflight_limit(state, config.limits),
|
|
1917
|
+
prospective_reason,
|
|
1918
|
+
refusal_reason,
|
|
1919
|
+
}:
|
|
1920
|
+
raise _v4_evidence_corrupt("terminal limit reason differs from authenticated tail")
|
|
1921
|
+
if reason == "FILL_ENDPOINT_DRIFT" and (
|
|
1922
|
+
latest_full is None
|
|
1923
|
+
or latest_full.final_url == response_tail.get("request_url")
|
|
1924
|
+
or response_tail.get("request_attempt") != state["budget"]["requests"]
|
|
1925
|
+
):
|
|
1926
|
+
raise _v4_evidence_corrupt("endpoint drift reason differs from authenticated tail")
|
|
1927
|
+
|
|
1928
|
+
|
|
1929
|
+
@dataclass(frozen=True)
|
|
1930
|
+
class _V4ValidatedResponseItem:
|
|
1931
|
+
item: dict[str, Any]
|
|
1932
|
+
full: HarvestResponseEvidence | None
|
|
1933
|
+
refusal: dict[str, Any] | None
|
|
1934
|
+
network_bytes: int
|
|
1935
|
+
endpoint_drift: bool
|
|
1936
|
+
|
|
1937
|
+
|
|
1938
|
+
def _validate_v4_request_member(
|
|
1939
|
+
item: Any, *, position: int, config: CatalogFillConfig
|
|
1940
|
+
) -> dict[str, Any]:
|
|
1941
|
+
if (
|
|
1942
|
+
not isinstance(item, dict)
|
|
1943
|
+
or set(item) != {"attempt", "request_url", "cursor", "elapsed_ns", "observed_epoch_seconds"}
|
|
1944
|
+
or item["attempt"] != position
|
|
1945
|
+
or not _valid_v4_request_url(
|
|
1946
|
+
item["request_url"], page_size=config.page_size, sort=config.sort
|
|
1947
|
+
)
|
|
1948
|
+
or type(item["elapsed_ns"]) is not int
|
|
1949
|
+
or item["elapsed_ns"] < 0
|
|
1950
|
+
or type(item["observed_epoch_seconds"]) is not int
|
|
1951
|
+
or not 0 <= item["observed_epoch_seconds"] <= _MAX_SAFE_EPOCH_SECONDS
|
|
1952
|
+
):
|
|
1953
|
+
raise _v4_evidence_corrupt("request member is invalid")
|
|
1954
|
+
request_cursor = _cursor_from_dict(item["cursor"])
|
|
1955
|
+
try:
|
|
1956
|
+
_sort, _page_size, url_cursor = _parse_v4_request_url(item["request_url"])
|
|
1957
|
+
except (CatalogHarvestError, TypeError, ValueError):
|
|
1958
|
+
raise _v4_evidence_corrupt("request URL is invalid") from None
|
|
1959
|
+
if request_cursor != url_cursor:
|
|
1960
|
+
raise _v4_evidence_corrupt("request cursor differs from URL")
|
|
1961
|
+
return item
|
|
1962
|
+
|
|
1963
|
+
|
|
1964
|
+
def _validate_v4_response_member(
|
|
1965
|
+
staging: FillStaging,
|
|
1966
|
+
item: Any,
|
|
1967
|
+
*,
|
|
1968
|
+
request: dict[str, Any],
|
|
1969
|
+
authenticate_raw: bool,
|
|
1970
|
+
) -> _V4ValidatedResponseItem:
|
|
1971
|
+
refusal_keys = {
|
|
1972
|
+
"request_attempt",
|
|
1973
|
+
"request_url",
|
|
1974
|
+
"observed_epoch_seconds",
|
|
1975
|
+
"refusal",
|
|
1976
|
+
}
|
|
1977
|
+
if isinstance(item, dict) and set(item) == refusal_keys:
|
|
1978
|
+
refusal = item["refusal"]
|
|
1979
|
+
if (
|
|
1980
|
+
request["request_url"] != item["request_url"]
|
|
1981
|
+
or item["observed_epoch_seconds"] != request["observed_epoch_seconds"]
|
|
1982
|
+
or not isinstance(refusal, dict)
|
|
1983
|
+
or set(refusal)
|
|
1984
|
+
!= {"code", "status", "content_bytes", "network_bytes", "rate_observation"}
|
|
1985
|
+
or not isinstance(refusal["code"], str)
|
|
1986
|
+
or type(refusal["status"]) is not int
|
|
1987
|
+
or not 100 <= refusal["status"] <= 599
|
|
1988
|
+
or refusal["content_bytes"] != 0
|
|
1989
|
+
or type(refusal["network_bytes"]) is not int
|
|
1990
|
+
or refusal["network_bytes"] < 0
|
|
1991
|
+
):
|
|
1992
|
+
raise _v4_evidence_corrupt("redacted response refusal is invalid")
|
|
1993
|
+
rate_value = refusal["rate_observation"]
|
|
1994
|
+
try:
|
|
1995
|
+
if rate_value is not None:
|
|
1996
|
+
if not isinstance(rate_value, dict):
|
|
1997
|
+
raise TypeError
|
|
1998
|
+
DatagovRateObservation(**rate_value)
|
|
1999
|
+
except (AcquisitionSecurityError, TypeError):
|
|
2000
|
+
raise _v4_evidence_corrupt("redacted rate observation is invalid") from None
|
|
2001
|
+
return _V4ValidatedResponseItem(
|
|
2002
|
+
item=item,
|
|
2003
|
+
full=None,
|
|
2004
|
+
refusal=refusal,
|
|
2005
|
+
network_bytes=refusal["network_bytes"],
|
|
2006
|
+
endpoint_drift=False,
|
|
2007
|
+
)
|
|
2008
|
+
full_keys = {
|
|
2009
|
+
"request_attempt",
|
|
2010
|
+
"request_url",
|
|
2011
|
+
"observed_epoch_seconds",
|
|
2012
|
+
"raw_response_sha256",
|
|
2013
|
+
"evidence",
|
|
2014
|
+
}
|
|
2015
|
+
if not isinstance(item, dict) or set(item) != full_keys:
|
|
2016
|
+
raise _v4_evidence_corrupt("response member is invalid")
|
|
2017
|
+
raw_digest = item["raw_response_sha256"]
|
|
2018
|
+
evidence = item["evidence"]
|
|
2019
|
+
if (
|
|
2020
|
+
request["request_url"] != item["request_url"]
|
|
2021
|
+
or item["observed_epoch_seconds"] != request["observed_epoch_seconds"]
|
|
2022
|
+
or not _valid_v4_request_url(item["request_url"])
|
|
2023
|
+
or not isinstance(evidence, dict)
|
|
2024
|
+
or not _SHA256.fullmatch(raw_digest or "")
|
|
2025
|
+
or evidence.get("content_sha256") != raw_digest
|
|
2026
|
+
):
|
|
2027
|
+
raise _v4_evidence_corrupt("response evidence is not bound to exact bytes")
|
|
2028
|
+
try:
|
|
2029
|
+
parsed = _response_evidence_from_dict(evidence)
|
|
2030
|
+
except (CatalogHarvestError, AcquisitionSecurityError, TypeError, ValueError):
|
|
2031
|
+
raise _v4_evidence_corrupt("response evidence is invalid") from None
|
|
2032
|
+
raw_size = len(staging.read_blob("raw_response", raw_digest)) if authenticate_raw else None
|
|
2033
|
+
if (
|
|
2034
|
+
parsed.requests != 1
|
|
2035
|
+
or parsed.responses != 1
|
|
2036
|
+
or parsed.media_type != "application/json"
|
|
2037
|
+
or (
|
|
2038
|
+
raw_size is not None
|
|
2039
|
+
and (parsed.content_bytes != raw_size or parsed.network_bytes != raw_size)
|
|
2040
|
+
)
|
|
2041
|
+
):
|
|
2042
|
+
raise _v4_evidence_corrupt("response byte/media accounting differs")
|
|
2043
|
+
return _V4ValidatedResponseItem(
|
|
2044
|
+
item=item,
|
|
2045
|
+
full=parsed,
|
|
2046
|
+
refusal=None,
|
|
2047
|
+
network_bytes=parsed.network_bytes,
|
|
2048
|
+
endpoint_drift=parsed.final_url != item["request_url"],
|
|
2049
|
+
)
|
|
2050
|
+
|
|
2051
|
+
|
|
2052
|
+
def _iter_v4_request_response_members(
|
|
2053
|
+
staging: FillStaging,
|
|
2054
|
+
indexes: dict[str, Any],
|
|
2055
|
+
*,
|
|
2056
|
+
config: CatalogFillConfig,
|
|
2057
|
+
authenticate_raw: bool = True,
|
|
2058
|
+
) -> Iterator[tuple[int, dict[str, Any], _V4ValidatedResponseItem | None]]:
|
|
2059
|
+
responses = iter(staging.iter_index(indexes["response"], stream="response"))
|
|
2060
|
+
response_item = next(responses, None)
|
|
2061
|
+
for position, request_value in enumerate(
|
|
2062
|
+
staging.iter_index(indexes["request"], stream="request"), start=1
|
|
2063
|
+
):
|
|
2064
|
+
request = _validate_v4_request_member(request_value, position=position, config=config)
|
|
2065
|
+
response: _V4ValidatedResponseItem | None = None
|
|
2066
|
+
if response_item is not None:
|
|
2067
|
+
attempt = (
|
|
2068
|
+
response_item.get("request_attempt") if isinstance(response_item, dict) else None
|
|
2069
|
+
)
|
|
2070
|
+
if type(attempt) is not int or attempt < position:
|
|
2071
|
+
raise _v4_evidence_corrupt("response attempt order differs from requests")
|
|
2072
|
+
if attempt == position:
|
|
2073
|
+
response = _validate_v4_response_member(
|
|
2074
|
+
staging,
|
|
2075
|
+
response_item,
|
|
2076
|
+
request=request,
|
|
2077
|
+
authenticate_raw=authenticate_raw,
|
|
2078
|
+
)
|
|
2079
|
+
response_item = next(responses, None)
|
|
2080
|
+
yield position, request, response
|
|
2081
|
+
if response_item is not None:
|
|
2082
|
+
raise _v4_evidence_corrupt("response index extends beyond request evidence")
|
|
2083
|
+
|
|
2084
|
+
|
|
2085
|
+
def _validate_v4_evidence(
|
|
2086
|
+
staging: FillStaging,
|
|
2087
|
+
state: dict[str, Any],
|
|
2088
|
+
*,
|
|
2089
|
+
config: CatalogFillConfig,
|
|
2090
|
+
) -> None:
|
|
2091
|
+
"""Authenticate compact chains and bind their ordered members to the fill budget."""
|
|
2092
|
+
|
|
2093
|
+
indexes = state["evidence_indexes"]
|
|
2094
|
+
staging.validate_index(state["partition_run_index"], stream="partition_run")
|
|
2095
|
+
request_count = 0
|
|
2096
|
+
request_elapsed_ns = 0
|
|
2097
|
+
latest_request_epoch_seconds: int | None = None
|
|
2098
|
+
response_count = 0
|
|
2099
|
+
response_network_bytes = 0
|
|
2100
|
+
latest_response_rate: dict[str, Any] | None = None
|
|
2101
|
+
latest_response_status: int | None = None
|
|
2102
|
+
latest_response_epoch_seconds: int | None = None
|
|
2103
|
+
latest_response_attempt = 0
|
|
2104
|
+
latest_full_response: HarvestResponseEvidence | None = None
|
|
2105
|
+
latest_refusal_code: str | None = None
|
|
2106
|
+
latest_response_item: dict[str, Any] | None = None
|
|
2107
|
+
latest_response_request: dict[str, Any] | None = None
|
|
2108
|
+
endpoint_drift_attempt = 0
|
|
2109
|
+
for position, request, response_item in _iter_v4_request_response_members(
|
|
2110
|
+
staging, indexes, config=config
|
|
2111
|
+
):
|
|
2112
|
+
request_count = position
|
|
2113
|
+
request_elapsed_ns += request["elapsed_ns"]
|
|
2114
|
+
latest_request_epoch_seconds = request["observed_epoch_seconds"]
|
|
2115
|
+
if response_item is None:
|
|
2116
|
+
continue
|
|
2117
|
+
response_count += 1
|
|
2118
|
+
response_network_bytes += response_item.network_bytes
|
|
2119
|
+
latest_response_item = response_item.item
|
|
2120
|
+
latest_response_request = request
|
|
2121
|
+
latest_response_epoch_seconds = response_item.item["observed_epoch_seconds"]
|
|
2122
|
+
latest_response_attempt = position
|
|
2123
|
+
if response_item.refusal is not None:
|
|
2124
|
+
refusal = response_item.refusal
|
|
2125
|
+
latest_response_rate = refusal["rate_observation"]
|
|
2126
|
+
latest_response_status = refusal["status"]
|
|
2127
|
+
latest_refusal_code = refusal["code"]
|
|
2128
|
+
latest_full_response = None
|
|
2129
|
+
else:
|
|
2130
|
+
parsed = response_item.full
|
|
2131
|
+
assert parsed is not None
|
|
2132
|
+
latest_response_rate = (
|
|
2133
|
+
parsed.rate_observation.to_dict() if parsed.rate_observation is not None else None
|
|
2134
|
+
)
|
|
2135
|
+
latest_response_status = parsed.status
|
|
2136
|
+
latest_refusal_code = None
|
|
2137
|
+
latest_full_response = parsed
|
|
2138
|
+
if response_item.endpoint_drift:
|
|
2139
|
+
endpoint_drift_attempt = position
|
|
2140
|
+
if (
|
|
2141
|
+
request_count != state["budget"]["requests"]
|
|
2142
|
+
or request_elapsed_ns != state["budget"]["elapsed_ns"]
|
|
2143
|
+
or response_count != state["budget"]["responses"]
|
|
2144
|
+
or response_network_bytes != state["budget"]["network_bytes"]
|
|
2145
|
+
):
|
|
2146
|
+
raise _v4_evidence_corrupt("request or response budget differs from evidence")
|
|
2147
|
+
inflight_observed = (
|
|
2148
|
+
state["inflight_request"].get("observed_response")
|
|
2149
|
+
if isinstance(state["inflight_request"], dict)
|
|
2150
|
+
else None
|
|
2151
|
+
)
|
|
2152
|
+
if isinstance(inflight_observed, dict):
|
|
2153
|
+
latest_response_rate = inflight_observed["rate_observation"]
|
|
2154
|
+
latest_response_status = inflight_observed["status"]
|
|
2155
|
+
latest_response_epoch_seconds = state["inflight_request"]["response_observed_epoch_seconds"]
|
|
2156
|
+
latest_response_attempt = state["inflight_request"]["attempt"]
|
|
2157
|
+
if state["last_rate_observation"] != latest_response_rate:
|
|
2158
|
+
raise _v4_evidence_corrupt("last rate observation differs from response evidence")
|
|
2159
|
+
expected_wait: tuple[int | None, str | None] = (None, None)
|
|
2160
|
+
try:
|
|
2161
|
+
if (
|
|
2162
|
+
latest_response_attempt
|
|
2163
|
+
== (
|
|
2164
|
+
state["inflight_request"]["attempt"]
|
|
2165
|
+
if isinstance(state["inflight_request"], dict) and inflight_observed is not None
|
|
2166
|
+
else state["budget"]["requests"]
|
|
2167
|
+
)
|
|
2168
|
+
and latest_response_epoch_seconds is not None
|
|
2169
|
+
):
|
|
2170
|
+
rate = (
|
|
2171
|
+
DatagovRateObservation(**latest_response_rate)
|
|
2172
|
+
if latest_response_rate is not None
|
|
2173
|
+
else None
|
|
2174
|
+
)
|
|
2175
|
+
expected_wait = _v4_response_wait(
|
|
2176
|
+
config, latest_response_status, rate, latest_response_epoch_seconds
|
|
2177
|
+
)
|
|
2178
|
+
elif latest_request_epoch_seconds is not None:
|
|
2179
|
+
wait_until = _pacing_wait(config, latest_request_epoch_seconds)
|
|
2180
|
+
expected_wait = (
|
|
2181
|
+
wait_until,
|
|
2182
|
+
"DATAGOV_PACING_REQUIRED" if wait_until is not None else None,
|
|
2183
|
+
)
|
|
2184
|
+
except CatalogFillRefused:
|
|
2185
|
+
raise _v4_evidence_corrupt("authenticated observation cannot produce a safe wait") from None
|
|
2186
|
+
if state["completed"] or state["reason_code"] in _V4_MAP_SEALING_REASONS:
|
|
2187
|
+
expected_wait = (None, None)
|
|
2188
|
+
if (
|
|
2189
|
+
state["resume_not_before_epoch_seconds"],
|
|
2190
|
+
state["wait_reason"],
|
|
2191
|
+
) != expected_wait:
|
|
2192
|
+
raise _v4_evidence_corrupt("durable wait differs from authenticated observation time")
|
|
2193
|
+
if state["reason_code"] in (_V4_POST_RESPONSE_OVERRUN_REASONS | _V4_PREFLIGHT_TERMINAL_REASONS):
|
|
2194
|
+
late_reason = (
|
|
2195
|
+
_late_limit(state, config.limits, latest_full_response)
|
|
2196
|
+
if latest_full_response is not None
|
|
2197
|
+
else None
|
|
2198
|
+
)
|
|
2199
|
+
preflight_reason = _preflight_limit(state, config.limits)
|
|
2200
|
+
prospective_reason = _v4_prospective_terminal_reason(
|
|
2201
|
+
staging,
|
|
2202
|
+
state,
|
|
2203
|
+
config=config,
|
|
2204
|
+
request_item=latest_response_request,
|
|
2205
|
+
response_item=latest_response_item,
|
|
2206
|
+
response=latest_full_response,
|
|
2207
|
+
)
|
|
2208
|
+
refusal_reason = (
|
|
2209
|
+
_v4_transport_terminal_reason(latest_refusal_code, state, config.limits)
|
|
2210
|
+
if latest_refusal_code is not None
|
|
2211
|
+
else None
|
|
2212
|
+
)
|
|
2213
|
+
if state["reason_code"] not in {
|
|
2214
|
+
late_reason,
|
|
2215
|
+
preflight_reason,
|
|
2216
|
+
prospective_reason,
|
|
2217
|
+
refusal_reason,
|
|
2218
|
+
}:
|
|
2219
|
+
raise _v4_evidence_corrupt("terminal limit reason differs from exact evidence")
|
|
2220
|
+
drift_is_latest = (
|
|
2221
|
+
endpoint_drift_attempt == state["budget"]["requests"] and endpoint_drift_attempt > 0
|
|
2222
|
+
)
|
|
2223
|
+
if endpoint_drift_attempt and (
|
|
2224
|
+
not drift_is_latest
|
|
2225
|
+
or (
|
|
2226
|
+
state["reason_code"]
|
|
2227
|
+
not in ({"FILL_ENDPOINT_DRIFT"} | _V4_POST_RESPONSE_OVERRUN_REASONS)
|
|
2228
|
+
and state["pending_response"] is None
|
|
2229
|
+
)
|
|
2230
|
+
):
|
|
2231
|
+
raise _v4_evidence_corrupt("endpoint drift reason differs from exact response evidence")
|
|
2232
|
+
if not endpoint_drift_attempt and state["reason_code"] == "FILL_ENDPOINT_DRIFT":
|
|
2233
|
+
raise _v4_evidence_corrupt("endpoint drift reason differs from exact response evidence")
|
|
2234
|
+
pending = state["pending_response"]
|
|
2235
|
+
if pending is not None:
|
|
2236
|
+
try:
|
|
2237
|
+
pending_evidence = _evidence_from_dict(pending["evidence"])
|
|
2238
|
+
pending_response = _response_evidence_from_dict(pending["response_evidence"])
|
|
2239
|
+
except (CatalogHarvestError, AcquisitionSecurityError, TypeError, ValueError):
|
|
2240
|
+
raise _v4_evidence_corrupt("pending response contract is invalid") from None
|
|
2241
|
+
if (
|
|
2242
|
+
latest_response_item is None
|
|
2243
|
+
or latest_response_attempt != state["budget"]["requests"]
|
|
2244
|
+
or latest_response_item["request_url"] != pending["request_url"]
|
|
2245
|
+
or latest_response_item["raw_response_sha256"] != pending["raw_response_sha256"]
|
|
2246
|
+
or latest_response_item["evidence"] != pending_response.to_dict()
|
|
2247
|
+
or pending_evidence.uri != pending_response.final_url
|
|
2248
|
+
or pending_evidence.content_sha256 != pending["raw_response_sha256"]
|
|
2249
|
+
or pending_evidence.media_type != "application/json"
|
|
2250
|
+
or pending_response.content_sha256 != pending["raw_response_sha256"]
|
|
2251
|
+
):
|
|
2252
|
+
raise _v4_evidence_corrupt("pending response differs from authenticated response tail")
|
|
2253
|
+
|
|
2254
|
+
record_items = iter(staging.iter_index(indexes["record"], stream="record"))
|
|
2255
|
+
partition_items = iter(staging.iter_index(state["partition_run_index"], stream="partition_run"))
|
|
2256
|
+
transitions = iter(staging.iter_index(indexes["cursor_transition"], stream="cursor_transition"))
|
|
2257
|
+
successful_responses = (
|
|
2258
|
+
(request, response_item)
|
|
2259
|
+
for _position, request, response_item in _iter_v4_request_response_members(
|
|
2260
|
+
staging, indexes, config=config, authenticate_raw=False
|
|
2261
|
+
)
|
|
2262
|
+
if response_item is not None
|
|
2263
|
+
and response_item.full is not None
|
|
2264
|
+
and response_item.full.status == 200
|
|
2265
|
+
and not response_item.endpoint_drift
|
|
2266
|
+
)
|
|
2267
|
+
observed_pages = 0
|
|
2268
|
+
observed_records = 0
|
|
2269
|
+
observed_skipped = 0
|
|
2270
|
+
observed_normalized_bytes = 0
|
|
2271
|
+
observed_partition_runs = 0
|
|
2272
|
+
expected_pass = 1
|
|
2273
|
+
pass_start_counts = {1: 0}
|
|
2274
|
+
prior_page_terminal = False
|
|
2275
|
+
expected_cursor: dict[str, Any] | None = None
|
|
2276
|
+
current_pass_members = 0
|
|
2277
|
+
try:
|
|
2278
|
+
with tempfile.TemporaryDirectory(prefix="mr-datagov-v4-audit-") as scratch:
|
|
2279
|
+
connection = sqlite3.connect(str(Path(scratch) / "paging.sqlite3"))
|
|
2280
|
+
try:
|
|
2281
|
+
connection.execute("PRAGMA journal_mode=OFF")
|
|
2282
|
+
connection.execute("PRAGMA synchronous=OFF")
|
|
2283
|
+
connection.execute("PRAGMA temp_store=FILE")
|
|
2284
|
+
connection.execute(
|
|
2285
|
+
"CREATE TABLE seen (kind INTEGER NOT NULL, digest TEXT NOT NULL, "
|
|
2286
|
+
"PRIMARY KEY (kind, digest)) WITHOUT ROWID"
|
|
2287
|
+
)
|
|
2288
|
+
for item in staging.iter_index(indexes["page"], stream="page"):
|
|
2289
|
+
if observed_pages and prior_page_terminal:
|
|
2290
|
+
expected_pass += 1
|
|
2291
|
+
pass_start_counts[expected_pass] = observed_partition_runs
|
|
2292
|
+
if (
|
|
2293
|
+
set(item)
|
|
2294
|
+
!= {"pass", "raw_response_sha256", "normalized_sha256", "record_count"}
|
|
2295
|
+
or type(item["pass"]) is not int
|
|
2296
|
+
or item["pass"] != expected_pass
|
|
2297
|
+
or item["pass"] > state["pass_number"]
|
|
2298
|
+
or not _SHA256.fullmatch(item["raw_response_sha256"] or "")
|
|
2299
|
+
or not _SHA256.fullmatch(item["normalized_sha256"] or "")
|
|
2300
|
+
or type(item["record_count"]) is not int
|
|
2301
|
+
or item["record_count"] < 0
|
|
2302
|
+
):
|
|
2303
|
+
raise _v4_evidence_corrupt("page member is invalid")
|
|
2304
|
+
while True:
|
|
2305
|
+
try:
|
|
2306
|
+
request, response_item = next(successful_responses)
|
|
2307
|
+
except StopIteration:
|
|
2308
|
+
raise _v4_evidence_corrupt(
|
|
2309
|
+
"page is not an ordered successful response member"
|
|
2310
|
+
) from None
|
|
2311
|
+
if response_item.item["raw_response_sha256"] == item["raw_response_sha256"]:
|
|
2312
|
+
break
|
|
2313
|
+
response = response_item.full
|
|
2314
|
+
if response is None:
|
|
2315
|
+
raise _v4_evidence_corrupt("successful response evidence is absent")
|
|
2316
|
+
raw_payload = staging.read_blob("raw_response", item["raw_response_sha256"])
|
|
2317
|
+
try:
|
|
2318
|
+
reparsed = DatagovV4Harvester().parse_page(
|
|
2319
|
+
raw_payload,
|
|
2320
|
+
uri=request["request_url"],
|
|
2321
|
+
observed_at=config.observed_at,
|
|
2322
|
+
evidence=EvidenceReference(
|
|
2323
|
+
uri=request["request_url"],
|
|
2324
|
+
observed_at=config.observed_at,
|
|
2325
|
+
content_sha256=item["raw_response_sha256"],
|
|
2326
|
+
media_type="application/json",
|
|
2327
|
+
),
|
|
2328
|
+
response_evidence=response,
|
|
2329
|
+
cursor=_cursor_from_dict(request["cursor"]),
|
|
2330
|
+
)
|
|
2331
|
+
except (
|
|
2332
|
+
CatalogHarvestError,
|
|
2333
|
+
CatalogFillRefused,
|
|
2334
|
+
CanonicalJSONError,
|
|
2335
|
+
TypeError,
|
|
2336
|
+
ValueError,
|
|
2337
|
+
):
|
|
2338
|
+
raise _v4_evidence_corrupt(
|
|
2339
|
+
"raw response cannot reproduce normalized page"
|
|
2340
|
+
) from None
|
|
2341
|
+
parsed_next_cursor = (
|
|
2342
|
+
reparsed.next_cursor.to_dict() if reparsed.next_cursor is not None else None
|
|
2343
|
+
)
|
|
2344
|
+
prior_page_terminal = parsed_next_cursor is None
|
|
2345
|
+
try:
|
|
2346
|
+
transition = next(transitions)
|
|
2347
|
+
except StopIteration:
|
|
2348
|
+
raise _v4_evidence_corrupt("cursor transition membership differs") from None
|
|
2349
|
+
if (
|
|
2350
|
+
set(transition)
|
|
2351
|
+
!= {"request_cursor", "next_cursor", "normalized_page_sha256"}
|
|
2352
|
+
or transition["normalized_page_sha256"] != item["normalized_sha256"]
|
|
2353
|
+
or transition["request_cursor"] != request["cursor"]
|
|
2354
|
+
or transition["request_cursor"] != expected_cursor
|
|
2355
|
+
or transition["next_cursor"] != parsed_next_cursor
|
|
2356
|
+
):
|
|
2357
|
+
raise _v4_evidence_corrupt("cursor transition order differs")
|
|
2358
|
+
_cursor_from_dict(transition["request_cursor"])
|
|
2359
|
+
_cursor_from_dict(transition["next_cursor"])
|
|
2360
|
+
requested_digest = canonical_sha256(transition["request_cursor"])
|
|
2361
|
+
next_digest = (
|
|
2362
|
+
canonical_sha256(transition["next_cursor"])
|
|
2363
|
+
if transition["next_cursor"] is not None
|
|
2364
|
+
else None
|
|
2365
|
+
)
|
|
2366
|
+
cursor_seen = connection.execute(
|
|
2367
|
+
"SELECT 1 FROM seen WHERE kind=0 AND digest=?", (requested_digest,)
|
|
2368
|
+
).fetchone()
|
|
2369
|
+
response_seen = connection.execute(
|
|
2370
|
+
"SELECT 1 FROM seen WHERE kind=1 AND digest=?",
|
|
2371
|
+
(item["raw_response_sha256"],),
|
|
2372
|
+
).fetchone()
|
|
2373
|
+
next_seen = (
|
|
2374
|
+
connection.execute(
|
|
2375
|
+
"SELECT 1 FROM seen WHERE kind=0 AND digest=?", (next_digest,)
|
|
2376
|
+
).fetchone()
|
|
2377
|
+
if next_digest is not None
|
|
2378
|
+
else None
|
|
2379
|
+
)
|
|
2380
|
+
if (
|
|
2381
|
+
cursor_seen is not None
|
|
2382
|
+
or response_seen is not None
|
|
2383
|
+
or (
|
|
2384
|
+
transition["next_cursor"] is not None
|
|
2385
|
+
and transition["next_cursor"] == transition["request_cursor"]
|
|
2386
|
+
)
|
|
2387
|
+
or next_seen is not None
|
|
2388
|
+
):
|
|
2389
|
+
raise _v4_evidence_corrupt("committed page contains a paging loop")
|
|
2390
|
+
connection.execute(
|
|
2391
|
+
"INSERT INTO seen(kind,digest) VALUES(0,?)", (requested_digest,)
|
|
2392
|
+
)
|
|
2393
|
+
connection.execute(
|
|
2394
|
+
"INSERT INTO seen(kind,digest) VALUES(1,?)",
|
|
2395
|
+
(item["raw_response_sha256"],),
|
|
2396
|
+
)
|
|
2397
|
+
if item["pass"] == state["pass_number"]:
|
|
2398
|
+
current_pass_members += 1
|
|
2399
|
+
if not _v4_loop_contains(
|
|
2400
|
+
staging,
|
|
2401
|
+
state["pass_cursor_set_sha256"],
|
|
2402
|
+
"cursor",
|
|
2403
|
+
requested_digest,
|
|
2404
|
+
) or not _v4_loop_contains(
|
|
2405
|
+
staging,
|
|
2406
|
+
state["pass_response_set_sha256"],
|
|
2407
|
+
"response",
|
|
2408
|
+
item["raw_response_sha256"],
|
|
2409
|
+
):
|
|
2410
|
+
raise _v4_evidence_corrupt("paging loop-set differs from current pass")
|
|
2411
|
+
expected_cursor = transition["next_cursor"]
|
|
2412
|
+
if expected_cursor is None:
|
|
2413
|
+
connection.execute("DELETE FROM seen")
|
|
2414
|
+
|
|
2415
|
+
manifest = staging.read_shard("normalized", item["normalized_sha256"])
|
|
2416
|
+
skipped_ids = manifest.get("skipped_record_ids")
|
|
2417
|
+
if (
|
|
2418
|
+
not isinstance(skipped_ids, list)
|
|
2419
|
+
or any(not isinstance(value, str) for value in skipped_ids)
|
|
2420
|
+
or skipped_ids != sorted(skipped_ids)
|
|
2421
|
+
):
|
|
2422
|
+
raise _v4_evidence_corrupt("normalized skipped identifiers differ")
|
|
2423
|
+
observed_skipped += len(skipped_ids)
|
|
2424
|
+
observed_normalized_bytes += _v4_normalized_shard_size(manifest)
|
|
2425
|
+
for descriptor in manifest.get("record_segments", []):
|
|
2426
|
+
if not isinstance(descriptor, dict) or not _SHA256.fullmatch(
|
|
2427
|
+
descriptor.get("sha256", "")
|
|
2428
|
+
):
|
|
2429
|
+
raise _v4_evidence_corrupt("normalized segment coordinate differs")
|
|
2430
|
+
observed_normalized_bytes += _v4_normalized_shard_size(
|
|
2431
|
+
staging.read_shard("normalized", descriptor["sha256"])
|
|
2432
|
+
)
|
|
2433
|
+
page_records: list[DatagovV4Record] = []
|
|
2434
|
+
for record_value in iter_v4_normalized_records(
|
|
2435
|
+
staging,
|
|
2436
|
+
item["normalized_sha256"],
|
|
2437
|
+
raw_response_sha256=item["raw_response_sha256"],
|
|
2438
|
+
record_count=item["record_count"],
|
|
2439
|
+
):
|
|
2440
|
+
record = _v4_record_from_dict(record_value)
|
|
2441
|
+
if record.raw_response_sha256 != item["raw_response_sha256"]:
|
|
2442
|
+
raise _v4_evidence_corrupt("normalized record raw provenance differs")
|
|
2443
|
+
page_records.append(record)
|
|
2444
|
+
try:
|
|
2445
|
+
record_item = next(record_items)
|
|
2446
|
+
except StopIteration:
|
|
2447
|
+
raise _v4_evidence_corrupt(
|
|
2448
|
+
"record index ended before normalized records"
|
|
2449
|
+
) from None
|
|
2450
|
+
if record_item != {
|
|
2451
|
+
"identifier_sha256": canonical_sha256(record.record_id),
|
|
2452
|
+
"semantic_sha256": record.digest,
|
|
2453
|
+
"normalized_page_sha256": item["normalized_sha256"],
|
|
2454
|
+
}:
|
|
2455
|
+
raise _v4_evidence_corrupt(
|
|
2456
|
+
"record index differs from normalized record"
|
|
2457
|
+
)
|
|
2458
|
+
if len(page_records) != item["record_count"]:
|
|
2459
|
+
raise _v4_evidence_corrupt("normalized page membership differs")
|
|
2460
|
+
if [record.to_dict() for record in page_records] != [
|
|
2461
|
+
record.to_dict() for record in reparsed.records
|
|
2462
|
+
] or tuple(skipped_ids) != reparsed.skipped_record_ids:
|
|
2463
|
+
raise _v4_evidence_corrupt(
|
|
2464
|
+
"normalized page differs from authenticated raw response"
|
|
2465
|
+
)
|
|
2466
|
+
observed_records += len(page_records)
|
|
2467
|
+
for partition, entries, conflicts in _expected_v4_page_runs(page_records):
|
|
2468
|
+
try:
|
|
2469
|
+
descriptor = next(partition_items)
|
|
2470
|
+
except StopIteration:
|
|
2471
|
+
raise _v4_evidence_corrupt(
|
|
2472
|
+
"partition index ended before page records"
|
|
2473
|
+
) from None
|
|
2474
|
+
actual_partition, actual_entries, actual_conflicts = read_partition_run(
|
|
2475
|
+
staging, descriptor
|
|
2476
|
+
)
|
|
2477
|
+
if (
|
|
2478
|
+
actual_partition != partition
|
|
2479
|
+
or actual_entries != entries
|
|
2480
|
+
or actual_conflicts != conflicts
|
|
2481
|
+
):
|
|
2482
|
+
raise _v4_evidence_corrupt(
|
|
2483
|
+
"partition run differs from normalized records"
|
|
2484
|
+
)
|
|
2485
|
+
observed_partition_runs += 1
|
|
2486
|
+
observed_pages += 1
|
|
2487
|
+
finally:
|
|
2488
|
+
connection.close()
|
|
2489
|
+
except (OSError, sqlite3.Error) as error:
|
|
2490
|
+
raise CatalogFillRefused(
|
|
2491
|
+
"DATAGOV_EXTERNAL_MERGE",
|
|
2492
|
+
"fill.staging.audit",
|
|
2493
|
+
"bounded paging evidence audit scratch failed",
|
|
2494
|
+
) from error
|
|
2495
|
+
if next(record_items, None) is not None:
|
|
2496
|
+
raise _v4_evidence_corrupt("record index extends beyond normalized records")
|
|
2497
|
+
if next(partition_items, None) is not None:
|
|
2498
|
+
raise _v4_evidence_corrupt("partition index extends beyond normalized records")
|
|
2499
|
+
if observed_pages == 0:
|
|
2500
|
+
if state["pass_number"] != 1:
|
|
2501
|
+
raise _v4_evidence_corrupt("empty history has an invalid pass number")
|
|
2502
|
+
elif state["pass_number"] == expected_pass + 1 and prior_page_terminal:
|
|
2503
|
+
pass_start_counts[state["pass_number"]] = observed_partition_runs
|
|
2504
|
+
elif state["pass_number"] != expected_pass:
|
|
2505
|
+
raise _v4_evidence_corrupt("state pass number differs from terminal boundaries")
|
|
2506
|
+
if state["pass_start_run_count"] != pass_start_counts[state["pass_number"]]:
|
|
2507
|
+
raise _v4_evidence_corrupt("pass run offset differs from terminal boundaries")
|
|
2508
|
+
if (state["completed"] or state["reason_code"] in _V4_MAP_SEALING_REASONS) and (
|
|
2509
|
+
not prior_page_terminal or state["pass_number"] != expected_pass
|
|
2510
|
+
):
|
|
2511
|
+
raise _v4_evidence_corrupt("final state is not the terminal current pass")
|
|
2512
|
+
if (state["pass_number"] == 1) != (state["previous_pass_map_sha256"] is None):
|
|
2513
|
+
raise _v4_evidence_corrupt("previous pass map differs from pass state")
|
|
2514
|
+
if (
|
|
2515
|
+
observed_pages != state["budget"]["pages"]
|
|
2516
|
+
or observed_records != state["budget"]["records"]
|
|
2517
|
+
or observed_skipped != state["budget"]["skipped_records"]
|
|
2518
|
+
or observed_normalized_bytes != state["normalized_bytes"]
|
|
2519
|
+
):
|
|
2520
|
+
raise _v4_evidence_corrupt("page, record, or skipped budget differs from evidence")
|
|
2521
|
+
if next(transitions, None) is not None:
|
|
2522
|
+
raise _v4_evidence_corrupt("cursor transition membership differs")
|
|
2523
|
+
if expected_cursor != state["next_cursor"]:
|
|
2524
|
+
raise _v4_evidence_corrupt("state cursor differs from transition tail")
|
|
2525
|
+
if (
|
|
2526
|
+
_v4_loop_count(staging, state["pass_cursor_set_sha256"], "cursor") != current_pass_members
|
|
2527
|
+
or _v4_loop_count(staging, state["pass_response_set_sha256"], "response")
|
|
2528
|
+
!= current_pass_members
|
|
2529
|
+
):
|
|
2530
|
+
raise _v4_evidence_corrupt("paging loop-set count differs from current pass")
|
|
2531
|
+
|
|
2532
|
+
manifest_sha256 = state["last_pass_manifest_sha256"]
|
|
2533
|
+
map_sha256 = state["final_map_sha256"] or state["previous_pass_map_sha256"]
|
|
2534
|
+
if (manifest_sha256 is None) != (map_sha256 is None):
|
|
2535
|
+
raise _v4_evidence_corrupt("pass manifest and map root differ")
|
|
2536
|
+
if manifest_sha256 is not None:
|
|
2537
|
+
if not _SHA256.fullmatch(manifest_sha256) or not _SHA256.fullmatch(map_sha256):
|
|
2538
|
+
raise _v4_evidence_corrupt("pass manifest coordinate is invalid")
|
|
2539
|
+
result = validate_pass_map(staging, manifest_sha256, map_sha256)
|
|
2540
|
+
if result.unique_records != state["unique_records"]:
|
|
2541
|
+
raise _v4_evidence_corrupt("pass manifest record count differs")
|
|
2542
|
+
if result.conflicting_identifiers != state["conflicting_identifiers"]:
|
|
2543
|
+
raise _v4_evidence_corrupt("pass manifest conflict count differs")
|
|
2544
|
+
if (
|
|
2545
|
+
state["completed"]
|
|
2546
|
+
and result.conflicting_identifiers > config.max_conflicting_identifiers
|
|
2547
|
+
):
|
|
2548
|
+
raise _v4_evidence_corrupt("completed map retains unadmitted identifier conflicts")
|
|
2549
|
+
run_count = state["partition_run_index"]["count"]
|
|
2550
|
+
if state["final_map_sha256"] is not None:
|
|
2551
|
+
start, stop = pass_start_counts[state["pass_number"]], run_count
|
|
2552
|
+
else:
|
|
2553
|
+
start = pass_start_counts[state["pass_number"] - 1]
|
|
2554
|
+
stop = state["pass_start_run_count"]
|
|
2555
|
+
source_runs = islice(
|
|
2556
|
+
staging.iter_index(state["partition_run_index"], stream="partition_run", start=start),
|
|
2557
|
+
stop - start,
|
|
2558
|
+
)
|
|
2559
|
+
recomputed = dry_run_external_merge_pass(staging, source_runs)
|
|
2560
|
+
if (
|
|
2561
|
+
recomputed.root_sha256 != map_sha256
|
|
2562
|
+
or recomputed.manifest_sha256 != manifest_sha256
|
|
2563
|
+
or recomputed.unique_records != state["unique_records"]
|
|
2564
|
+
):
|
|
2565
|
+
raise _v4_evidence_corrupt("pass map differs from normalized source runs")
|
|
2566
|
+
if state["final_map_sha256"] is not None:
|
|
2567
|
+
prior_start = pass_start_counts[state["pass_number"] - 1]
|
|
2568
|
+
prior_stop = pass_start_counts[state["pass_number"]]
|
|
2569
|
+
prior_runs = islice(
|
|
2570
|
+
staging.iter_index(
|
|
2571
|
+
state["partition_run_index"],
|
|
2572
|
+
stream="partition_run",
|
|
2573
|
+
start=prior_start,
|
|
2574
|
+
),
|
|
2575
|
+
prior_stop - prior_start,
|
|
2576
|
+
)
|
|
2577
|
+
prior = dry_run_external_merge_pass(staging, prior_runs)
|
|
2578
|
+
if prior.root_sha256 != state["previous_pass_map_sha256"]:
|
|
2579
|
+
raise _v4_evidence_corrupt("preceding pass map differs from normalized source runs")
|
|
2580
|
+
if state["completed"] and prior.root_sha256 != recomputed.root_sha256:
|
|
2581
|
+
raise _v4_evidence_corrupt("completed passes did not converge")
|
|
2582
|
+
|
|
2583
|
+
|
|
2584
|
+
def iter_v4_normalized_records(
|
|
2585
|
+
staging: FillStaging,
|
|
2586
|
+
manifest_sha256: str,
|
|
2587
|
+
*,
|
|
2588
|
+
raw_response_sha256: str,
|
|
2589
|
+
record_count: int,
|
|
2590
|
+
) -> Iterator[dict[str, Any]]:
|
|
2591
|
+
"""Yield one logical normalized page while authenticating its bounded ordered segments."""
|
|
2592
|
+
|
|
2593
|
+
manifest = staging.read_shard("normalized", manifest_sha256)
|
|
2594
|
+
if (
|
|
2595
|
+
set(manifest)
|
|
2596
|
+
!= {
|
|
2597
|
+
"schema_version",
|
|
2598
|
+
"raw_response_sha256",
|
|
2599
|
+
"record_count",
|
|
2600
|
+
"record_segments",
|
|
2601
|
+
"skipped_record_ids",
|
|
2602
|
+
}
|
|
2603
|
+
or manifest["schema_version"] != "harness-datagov-v4-normalized-page.v2"
|
|
2604
|
+
or manifest["raw_response_sha256"] != raw_response_sha256
|
|
2605
|
+
or manifest["record_count"] != record_count
|
|
2606
|
+
or not isinstance(manifest["record_segments"], list)
|
|
2607
|
+
or not isinstance(manifest["skipped_record_ids"], list)
|
|
2608
|
+
):
|
|
2609
|
+
raise _v4_evidence_corrupt("normalized page manifest differs")
|
|
2610
|
+
next_record = 0
|
|
2611
|
+
for descriptor in manifest["record_segments"]:
|
|
2612
|
+
if (
|
|
2613
|
+
not isinstance(descriptor, dict)
|
|
2614
|
+
or set(descriptor) != {"sha256", "first_record", "record_count"}
|
|
2615
|
+
or not _SHA256.fullmatch(descriptor["sha256"] or "")
|
|
2616
|
+
or descriptor["first_record"] != next_record
|
|
2617
|
+
or type(descriptor["record_count"]) is not int
|
|
2618
|
+
or descriptor["record_count"] < 1
|
|
2619
|
+
):
|
|
2620
|
+
raise _v4_evidence_corrupt("normalized record segment descriptor differs")
|
|
2621
|
+
segment = staging.read_shard("normalized", descriptor["sha256"])
|
|
2622
|
+
if (
|
|
2623
|
+
set(segment)
|
|
2624
|
+
!= {
|
|
2625
|
+
"schema_version",
|
|
2626
|
+
"raw_response_sha256",
|
|
2627
|
+
"first_record",
|
|
2628
|
+
"records",
|
|
2629
|
+
}
|
|
2630
|
+
or segment["schema_version"] != "harness-datagov-v4-normalized-record-segment.v1"
|
|
2631
|
+
or segment["raw_response_sha256"] != raw_response_sha256
|
|
2632
|
+
or segment["first_record"] != next_record
|
|
2633
|
+
or not isinstance(segment["records"], list)
|
|
2634
|
+
or len(segment["records"]) != descriptor["record_count"]
|
|
2635
|
+
):
|
|
2636
|
+
raise _v4_evidence_corrupt("normalized record segment differs")
|
|
2637
|
+
for record in segment["records"]:
|
|
2638
|
+
if not isinstance(record, dict):
|
|
2639
|
+
raise _v4_evidence_corrupt("normalized record is invalid")
|
|
2640
|
+
yield record
|
|
2641
|
+
next_record += descriptor["record_count"]
|
|
2642
|
+
if next_record != record_count:
|
|
2643
|
+
raise _v4_evidence_corrupt("normalized record segment count differs")
|
|
2644
|
+
|
|
2645
|
+
|
|
2646
|
+
def _v4_record_from_dict(value: Any) -> DatagovV4Record:
|
|
2647
|
+
required = {
|
|
2648
|
+
"schema_version",
|
|
2649
|
+
"protocol",
|
|
2650
|
+
"record_id",
|
|
2651
|
+
"fields",
|
|
2652
|
+
"raw_response_sha256",
|
|
2653
|
+
"normalized_sha256",
|
|
2654
|
+
}
|
|
2655
|
+
if (
|
|
2656
|
+
not isinstance(value, dict)
|
|
2657
|
+
or not required <= set(value) <= required | {"observations"}
|
|
2658
|
+
or not isinstance(value["fields"], list)
|
|
2659
|
+
or not isinstance(value.get("observations", []), list)
|
|
2660
|
+
):
|
|
2661
|
+
raise _v4_evidence_corrupt("normalized record shape differs")
|
|
2662
|
+
try:
|
|
2663
|
+
record = DatagovV4Record(
|
|
2664
|
+
schema_version=value["schema_version"],
|
|
2665
|
+
protocol=value["protocol"],
|
|
2666
|
+
record_id=value["record_id"],
|
|
2667
|
+
fields=_v4_projected_fields(value["fields"]),
|
|
2668
|
+
observations=_v4_projected_fields(value.get("observations", [])),
|
|
2669
|
+
raw_response_sha256=value["raw_response_sha256"],
|
|
2670
|
+
)
|
|
2671
|
+
except (CatalogHarvestError, CanonicalJSONError, TypeError, ValueError):
|
|
2672
|
+
raise _v4_evidence_corrupt("normalized record contract differs") from None
|
|
2673
|
+
if record.to_dict() != value:
|
|
2674
|
+
raise _v4_evidence_corrupt("normalized record digest differs")
|
|
2675
|
+
return record
|
|
2676
|
+
|
|
2677
|
+
|
|
2678
|
+
def _v4_projected_fields(values: Any) -> tuple[DatagovProjectedField, ...]:
|
|
2679
|
+
fields: list[DatagovProjectedField] = []
|
|
2680
|
+
for field in values:
|
|
2681
|
+
if not isinstance(field, dict) or not {"path", "value"} <= set(field) <= {
|
|
2682
|
+
"path",
|
|
2683
|
+
"value",
|
|
2684
|
+
"encoding",
|
|
2685
|
+
}:
|
|
2686
|
+
raise TypeError
|
|
2687
|
+
fields.append(
|
|
2688
|
+
DatagovProjectedField(
|
|
2689
|
+
path=field["path"],
|
|
2690
|
+
value=field["value"],
|
|
2691
|
+
encoding=field.get("encoding", "canonical_json"),
|
|
2692
|
+
)
|
|
2693
|
+
)
|
|
2694
|
+
return tuple(fields)
|
|
2695
|
+
|
|
2696
|
+
|
|
2697
|
+
def _expected_v4_page_runs(
|
|
2698
|
+
records: list[DatagovV4Record],
|
|
2699
|
+
) -> Iterator[tuple[str, list[tuple[str, str]], set[str]]]:
|
|
2700
|
+
yield from plan_partition_runs(records)
|
|
2701
|
+
|
|
2702
|
+
|
|
2703
|
+
def _valid_v4_request_url(
|
|
2704
|
+
value: Any, *, page_size: int | None = None, sort: str | None = None
|
|
2705
|
+
) -> bool:
|
|
2706
|
+
try:
|
|
2707
|
+
parsed_sort, parsed_page_size, _cursor = _parse_v4_request_url(value)
|
|
2708
|
+
except (CatalogHarvestError, TypeError, ValueError):
|
|
2709
|
+
return False
|
|
2710
|
+
return (page_size is None or parsed_page_size == page_size) and (
|
|
2711
|
+
sort is None or parsed_sort == sort
|
|
2712
|
+
)
|
|
2713
|
+
|
|
2714
|
+
|
|
2715
|
+
def _parse_v4_request_url(value: Any) -> tuple[str, int, HarvestCursor | None]:
|
|
2716
|
+
if not isinstance(value, str):
|
|
2717
|
+
raise ValueError("request URL must be text")
|
|
2718
|
+
try:
|
|
2719
|
+
parsed = urlsplit(value)
|
|
2720
|
+
endpoint = urlsplit(DATAGOV_V4_ENDPOINT)
|
|
2721
|
+
port = parsed.port
|
|
2722
|
+
endpoint_port = endpoint.port
|
|
2723
|
+
pairs = parse_qsl(parsed.query, keep_blank_values=True, strict_parsing=True)
|
|
2724
|
+
except ValueError:
|
|
2725
|
+
raise ValueError("request URL cannot be parsed") from None
|
|
2726
|
+
if not (
|
|
2727
|
+
parsed.scheme == endpoint.scheme
|
|
2728
|
+
and parsed.hostname == endpoint.hostname
|
|
2729
|
+
and port == endpoint_port
|
|
2730
|
+
and parsed.path == endpoint.path
|
|
2731
|
+
and parsed.fragment == ""
|
|
2732
|
+
and len(pairs) in {2, 3}
|
|
2733
|
+
and pairs[0][0] == "sort"
|
|
2734
|
+
and pairs[0][1] in DATAGOV_V4_SORT_ORDERS
|
|
2735
|
+
and pairs[1][0] == "per_page"
|
|
2736
|
+
):
|
|
2737
|
+
raise ValueError("request URL coordinate differs")
|
|
2738
|
+
per_page_text = pairs[1][1]
|
|
2739
|
+
if not re.fullmatch(r"(?:[1-9]|[1-9][0-9]{1,2}|1000)", per_page_text):
|
|
2740
|
+
raise ValueError("request page size differs")
|
|
2741
|
+
cursor: HarvestCursor | None = None
|
|
2742
|
+
if len(pairs) == 3:
|
|
2743
|
+
if pairs[2][0] != "after":
|
|
2744
|
+
raise ValueError("request cursor key differs")
|
|
2745
|
+
cursor = HarvestCursor(protocol="datagov_v4", kind="after", value=pairs[2][1])
|
|
2746
|
+
page_size = int(per_page_text)
|
|
2747
|
+
sort = pairs[0][1]
|
|
2748
|
+
canonical_query = [("sort", sort), ("per_page", per_page_text)]
|
|
2749
|
+
if cursor is not None:
|
|
2750
|
+
canonical_query.append(("after", cursor.value))
|
|
2751
|
+
canonical = f"{DATAGOV_V4_ENDPOINT}?{urlencode(canonical_query)}"
|
|
2752
|
+
if value != canonical:
|
|
2753
|
+
raise ValueError("request URL spelling is not canonical")
|
|
2754
|
+
return sort, page_size, cursor
|
|
2755
|
+
|
|
2756
|
+
|
|
2757
|
+
def _v4_evidence_corrupt(detail: str) -> CatalogFillRefused:
|
|
2758
|
+
return CatalogFillRefused("FILL_CHECKPOINT_CORRUPT", "fill.staging.indexes", detail)
|
|
2759
|
+
|
|
2760
|
+
|
|
2761
|
+
@dataclass
|
|
2762
|
+
class _V4LoopSet:
|
|
2763
|
+
staging: FillStaging
|
|
2764
|
+
set_kind: str
|
|
2765
|
+
root_sha256: str | None
|
|
2766
|
+
|
|
2767
|
+
def __contains__(self, digest: object) -> bool:
|
|
2768
|
+
return isinstance(digest, str) and _v4_loop_contains(
|
|
2769
|
+
self.staging, self.root_sha256, self.set_kind, digest
|
|
2770
|
+
)
|
|
2771
|
+
|
|
2772
|
+
def add(self, digest: str) -> None:
|
|
2773
|
+
self.root_sha256 = _v4_loop_add(self.staging, self.root_sha256, self.set_kind, digest)
|
|
2774
|
+
|
|
2775
|
+
def clear(self) -> None:
|
|
2776
|
+
self.root_sha256 = None
|
|
2777
|
+
|
|
2778
|
+
|
|
2779
|
+
def _v4_pass_loop_coordinates(
|
|
2780
|
+
staging: FillStaging, state: dict[str, Any]
|
|
2781
|
+
) -> tuple[_V4LoopSet, _V4LoopSet]:
|
|
2782
|
+
return (
|
|
2783
|
+
_V4LoopSet(staging, "cursor", state["pass_cursor_set_sha256"]),
|
|
2784
|
+
_V4LoopSet(staging, "response", state["pass_response_set_sha256"]),
|
|
2785
|
+
)
|
|
2786
|
+
|
|
2787
|
+
|
|
2788
|
+
def _v4_loop_node(
|
|
2789
|
+
staging: FillStaging,
|
|
2790
|
+
digest: str,
|
|
2791
|
+
*,
|
|
2792
|
+
set_kind: str,
|
|
2793
|
+
depth: int,
|
|
2794
|
+
prefix: str,
|
|
2795
|
+
) -> dict[str, Any]:
|
|
2796
|
+
node = staging.read_shard("loop_set", digest)
|
|
2797
|
+
base_valid = (
|
|
2798
|
+
node.get("schema_version") == DATAGOV_V4_LOOP_SET_SCHEMA
|
|
2799
|
+
and node.get("set_kind") == set_kind
|
|
2800
|
+
and node.get("depth") == depth
|
|
2801
|
+
and node.get("prefix") == prefix
|
|
2802
|
+
and set_kind in {"cursor", "response"}
|
|
2803
|
+
and 0 <= depth <= 64
|
|
2804
|
+
)
|
|
2805
|
+
if not base_valid:
|
|
2806
|
+
raise _v4_evidence_corrupt("paging loop-set node coordinate differs")
|
|
2807
|
+
if set(node) == {"schema_version", "set_kind", "depth", "prefix", "values"}:
|
|
2808
|
+
values = node["values"]
|
|
2809
|
+
if (
|
|
2810
|
+
not isinstance(values, list)
|
|
2811
|
+
or not 1 <= len(values) <= DATAGOV_V4_LOOP_SET_LEAF_ITEMS
|
|
2812
|
+
or values != sorted(set(values))
|
|
2813
|
+
or any(not _SHA256.fullmatch(value) or not value.startswith(prefix) for value in values)
|
|
2814
|
+
):
|
|
2815
|
+
raise _v4_evidence_corrupt("paging loop-set leaf is invalid")
|
|
2816
|
+
return node
|
|
2817
|
+
if set(node) == {"schema_version", "set_kind", "depth", "prefix", "children"}:
|
|
2818
|
+
children = node["children"]
|
|
2819
|
+
if (
|
|
2820
|
+
depth >= 64
|
|
2821
|
+
or not isinstance(children, list)
|
|
2822
|
+
or not 1 <= len(children) <= 16
|
|
2823
|
+
or any(
|
|
2824
|
+
not isinstance(child, dict)
|
|
2825
|
+
or set(child) != {"nibble", "sha256"}
|
|
2826
|
+
or child["nibble"] not in "0123456789abcdef"
|
|
2827
|
+
or not _SHA256.fullmatch(child["sha256"] or "")
|
|
2828
|
+
for child in children
|
|
2829
|
+
)
|
|
2830
|
+
or [child["nibble"] for child in children]
|
|
2831
|
+
!= sorted({child["nibble"] for child in children})
|
|
2832
|
+
):
|
|
2833
|
+
raise _v4_evidence_corrupt("paging loop-set branch is invalid")
|
|
2834
|
+
return node
|
|
2835
|
+
raise _v4_evidence_corrupt("paging loop-set node shape differs")
|
|
2836
|
+
|
|
2837
|
+
|
|
2838
|
+
def _v4_write_loop_leaf(
|
|
2839
|
+
staging: FillStaging,
|
|
2840
|
+
*,
|
|
2841
|
+
set_kind: str,
|
|
2842
|
+
depth: int,
|
|
2843
|
+
prefix: str,
|
|
2844
|
+
values: list[str],
|
|
2845
|
+
) -> str:
|
|
2846
|
+
return staging.write_shard(
|
|
2847
|
+
"loop_set",
|
|
2848
|
+
{
|
|
2849
|
+
"schema_version": DATAGOV_V4_LOOP_SET_SCHEMA,
|
|
2850
|
+
"set_kind": set_kind,
|
|
2851
|
+
"depth": depth,
|
|
2852
|
+
"prefix": prefix,
|
|
2853
|
+
"values": values,
|
|
2854
|
+
},
|
|
2855
|
+
)
|
|
2856
|
+
|
|
2857
|
+
|
|
2858
|
+
def _v4_write_loop_subtree(
|
|
2859
|
+
staging: FillStaging,
|
|
2860
|
+
*,
|
|
2861
|
+
set_kind: str,
|
|
2862
|
+
depth: int,
|
|
2863
|
+
prefix: str,
|
|
2864
|
+
values: list[str],
|
|
2865
|
+
) -> str:
|
|
2866
|
+
if len(values) <= DATAGOV_V4_LOOP_SET_LEAF_ITEMS:
|
|
2867
|
+
return _v4_write_loop_leaf(
|
|
2868
|
+
staging,
|
|
2869
|
+
set_kind=set_kind,
|
|
2870
|
+
depth=depth,
|
|
2871
|
+
prefix=prefix,
|
|
2872
|
+
values=values,
|
|
2873
|
+
)
|
|
2874
|
+
if depth >= 64:
|
|
2875
|
+
raise _v4_evidence_corrupt("paging loop-set leaf cannot split")
|
|
2876
|
+
children = []
|
|
2877
|
+
for nibble in sorted({value[depth] for value in values}):
|
|
2878
|
+
child_values = [value for value in values if value[depth] == nibble]
|
|
2879
|
+
children.append(
|
|
2880
|
+
{
|
|
2881
|
+
"nibble": nibble,
|
|
2882
|
+
"sha256": _v4_write_loop_subtree(
|
|
2883
|
+
staging,
|
|
2884
|
+
set_kind=set_kind,
|
|
2885
|
+
depth=depth + 1,
|
|
2886
|
+
prefix=prefix + nibble,
|
|
2887
|
+
values=child_values,
|
|
2888
|
+
),
|
|
2889
|
+
}
|
|
2890
|
+
)
|
|
2891
|
+
return staging.write_shard(
|
|
2892
|
+
"loop_set",
|
|
2893
|
+
{
|
|
2894
|
+
"schema_version": DATAGOV_V4_LOOP_SET_SCHEMA,
|
|
2895
|
+
"set_kind": set_kind,
|
|
2896
|
+
"depth": depth,
|
|
2897
|
+
"prefix": prefix,
|
|
2898
|
+
"children": children,
|
|
2899
|
+
},
|
|
2900
|
+
)
|
|
2901
|
+
|
|
2902
|
+
|
|
2903
|
+
def _v4_loop_add(
|
|
2904
|
+
staging: FillStaging,
|
|
2905
|
+
root_sha256: str | None,
|
|
2906
|
+
set_kind: str,
|
|
2907
|
+
digest: str,
|
|
2908
|
+
*,
|
|
2909
|
+
depth: int = 0,
|
|
2910
|
+
prefix: str = "",
|
|
2911
|
+
) -> str:
|
|
2912
|
+
if set_kind not in {"cursor", "response"} or not _SHA256.fullmatch(digest or ""):
|
|
2913
|
+
raise _v4_evidence_corrupt("paging loop-set member is invalid")
|
|
2914
|
+
if root_sha256 is None:
|
|
2915
|
+
return _v4_write_loop_leaf(
|
|
2916
|
+
staging,
|
|
2917
|
+
set_kind=set_kind,
|
|
2918
|
+
depth=depth,
|
|
2919
|
+
prefix=prefix,
|
|
2920
|
+
values=[digest],
|
|
2921
|
+
)
|
|
2922
|
+
node = _v4_loop_node(staging, root_sha256, set_kind=set_kind, depth=depth, prefix=prefix)
|
|
2923
|
+
if "values" in node:
|
|
2924
|
+
values = node["values"]
|
|
2925
|
+
if digest in values:
|
|
2926
|
+
return root_sha256
|
|
2927
|
+
combined = sorted([*values, digest])
|
|
2928
|
+
if len(combined) <= DATAGOV_V4_LOOP_SET_LEAF_ITEMS:
|
|
2929
|
+
return _v4_write_loop_leaf(
|
|
2930
|
+
staging,
|
|
2931
|
+
set_kind=set_kind,
|
|
2932
|
+
depth=depth,
|
|
2933
|
+
prefix=prefix,
|
|
2934
|
+
values=combined,
|
|
2935
|
+
)
|
|
2936
|
+
return _v4_write_loop_subtree(
|
|
2937
|
+
staging,
|
|
2938
|
+
set_kind=set_kind,
|
|
2939
|
+
depth=depth,
|
|
2940
|
+
prefix=prefix,
|
|
2941
|
+
values=combined,
|
|
2942
|
+
)
|
|
2943
|
+
else:
|
|
2944
|
+
children = [dict(child) for child in node["children"]]
|
|
2945
|
+
nibble = digest[depth]
|
|
2946
|
+
for child in children:
|
|
2947
|
+
if child["nibble"] == nibble:
|
|
2948
|
+
child["sha256"] = _v4_loop_add(
|
|
2949
|
+
staging,
|
|
2950
|
+
child["sha256"],
|
|
2951
|
+
set_kind,
|
|
2952
|
+
digest,
|
|
2953
|
+
depth=depth + 1,
|
|
2954
|
+
prefix=prefix + nibble,
|
|
2955
|
+
)
|
|
2956
|
+
break
|
|
2957
|
+
else:
|
|
2958
|
+
children.append(
|
|
2959
|
+
{
|
|
2960
|
+
"nibble": nibble,
|
|
2961
|
+
"sha256": _v4_write_loop_leaf(
|
|
2962
|
+
staging,
|
|
2963
|
+
set_kind=set_kind,
|
|
2964
|
+
depth=depth + 1,
|
|
2965
|
+
prefix=prefix + nibble,
|
|
2966
|
+
values=[digest],
|
|
2967
|
+
),
|
|
2968
|
+
}
|
|
2969
|
+
)
|
|
2970
|
+
children.sort(key=lambda child: child["nibble"])
|
|
2971
|
+
return staging.write_shard(
|
|
2972
|
+
"loop_set",
|
|
2973
|
+
{
|
|
2974
|
+
"schema_version": DATAGOV_V4_LOOP_SET_SCHEMA,
|
|
2975
|
+
"set_kind": set_kind,
|
|
2976
|
+
"depth": depth,
|
|
2977
|
+
"prefix": prefix,
|
|
2978
|
+
"children": children,
|
|
2979
|
+
},
|
|
2980
|
+
)
|
|
2981
|
+
|
|
2982
|
+
|
|
2983
|
+
def _v4_loop_contains(
|
|
2984
|
+
staging: FillStaging,
|
|
2985
|
+
root_sha256: str | None,
|
|
2986
|
+
set_kind: str,
|
|
2987
|
+
digest: str,
|
|
2988
|
+
) -> bool:
|
|
2989
|
+
if root_sha256 is None:
|
|
2990
|
+
return False
|
|
2991
|
+
depth = 0
|
|
2992
|
+
prefix = ""
|
|
2993
|
+
current = root_sha256
|
|
2994
|
+
while True:
|
|
2995
|
+
node = _v4_loop_node(staging, current, set_kind=set_kind, depth=depth, prefix=prefix)
|
|
2996
|
+
if "values" in node:
|
|
2997
|
+
return digest in node["values"]
|
|
2998
|
+
nibble = digest[depth]
|
|
2999
|
+
child = next((child for child in node["children"] if child["nibble"] == nibble), None)
|
|
3000
|
+
if child is None:
|
|
3001
|
+
return False
|
|
3002
|
+
current = child["sha256"]
|
|
3003
|
+
prefix += nibble
|
|
3004
|
+
depth += 1
|
|
3005
|
+
|
|
3006
|
+
|
|
3007
|
+
def _v4_loop_count(staging: FillStaging, root_sha256: str | None, set_kind: str) -> int:
|
|
3008
|
+
if root_sha256 is None:
|
|
3009
|
+
return 0
|
|
3010
|
+
pending = [(root_sha256, 0, "")]
|
|
3011
|
+
count = 0
|
|
3012
|
+
visited = 0
|
|
3013
|
+
while pending:
|
|
3014
|
+
digest, depth, prefix = pending.pop()
|
|
3015
|
+
visited += 1
|
|
3016
|
+
if visited > 2_000_000:
|
|
3017
|
+
raise _v4_evidence_corrupt("paging loop-set exceeds its node bound")
|
|
3018
|
+
node = _v4_loop_node(staging, digest, set_kind=set_kind, depth=depth, prefix=prefix)
|
|
3019
|
+
if "values" in node:
|
|
3020
|
+
count += len(node["values"])
|
|
3021
|
+
else:
|
|
3022
|
+
pending.extend(
|
|
3023
|
+
(child["sha256"], depth + 1, prefix + child["nibble"]) for child in node["children"]
|
|
3024
|
+
)
|
|
3025
|
+
return count
|
|
3026
|
+
|
|
3027
|
+
|
|
3028
|
+
def _pacing_wait(config: CatalogFillConfig, observed_epoch_seconds: int) -> int | None:
|
|
3029
|
+
if config.minimum_request_interval_ns == 0:
|
|
3030
|
+
return None
|
|
3031
|
+
seconds = (config.minimum_request_interval_ns + 999_999_999) // 1_000_000_000
|
|
3032
|
+
return _checked_epoch_add(observed_epoch_seconds, seconds)
|
|
3033
|
+
|
|
3034
|
+
|
|
3035
|
+
def _pacing_sleep_seconds(
|
|
3036
|
+
config: CatalogFillConfig,
|
|
3037
|
+
*,
|
|
3038
|
+
deadline_epoch_seconds: int,
|
|
3039
|
+
now_epoch_seconds: int,
|
|
3040
|
+
anchor_ns: int | None,
|
|
3041
|
+
now_ns: int,
|
|
3042
|
+
) -> float:
|
|
3043
|
+
"""How long an in-process pace waits, honouring the later of two independent bounds.
|
|
3044
|
+
|
|
3045
|
+
The wall-clock bound is the durable one: it is the deadline the checkpoint records, so an
|
|
3046
|
+
in-process pace can never resume earlier than a cross-process resume of the same checkpoint
|
|
3047
|
+
would have. The monotonic bound is the configured interval measured from the response this
|
|
3048
|
+
pace follows. It is needed because the durable deadline is kept in whole seconds and both
|
|
3049
|
+
ends of it are truncated, which can leave a wall-clock difference nearly a second shorter
|
|
3050
|
+
than the interval it stands for -- a rounding artefact that must never be spent as speed.
|
|
3051
|
+
``anchor_ns`` is absent only when a fresh process resumes a wait it did not itself observe,
|
|
3052
|
+
and there the durable deadline is the only bound there is evidence for.
|
|
3053
|
+
"""
|
|
3054
|
+
|
|
3055
|
+
seconds = float(max(0, deadline_epoch_seconds - now_epoch_seconds))
|
|
3056
|
+
if anchor_ns is None:
|
|
3057
|
+
return seconds
|
|
3058
|
+
remaining_ns = anchor_ns + config.minimum_request_interval_ns - now_ns
|
|
3059
|
+
return max(seconds, max(0, remaining_ns) / 1_000_000_000)
|
|
3060
|
+
|
|
3061
|
+
|
|
3062
|
+
def _checked_epoch_add(epoch_seconds: int, delta_seconds: int) -> int:
|
|
3063
|
+
if (
|
|
3064
|
+
type(epoch_seconds) is not int
|
|
3065
|
+
or type(delta_seconds) is not int
|
|
3066
|
+
or not 0 <= epoch_seconds <= _MAX_SAFE_EPOCH_SECONDS
|
|
3067
|
+
or delta_seconds < 0
|
|
3068
|
+
or epoch_seconds > _MAX_SAFE_EPOCH_SECONDS - delta_seconds
|
|
3069
|
+
):
|
|
3070
|
+
raise CatalogFillRefused(
|
|
3071
|
+
"FILL_CLOCK", "fill.wall_clock", "wall-clock deadline exceeds the safe epoch"
|
|
3072
|
+
)
|
|
3073
|
+
return epoch_seconds + delta_seconds
|
|
3074
|
+
|
|
3075
|
+
|
|
3076
|
+
def _v4_response_wait(
|
|
3077
|
+
config: CatalogFillConfig,
|
|
3078
|
+
status: int,
|
|
3079
|
+
rate: DatagovRateObservation | None,
|
|
3080
|
+
observed_epoch_seconds: int,
|
|
3081
|
+
) -> tuple[int | None, str | None]:
|
|
3082
|
+
if status == 429:
|
|
3083
|
+
retry_at = (
|
|
3084
|
+
_checked_epoch_add(observed_epoch_seconds, rate.retry_after_seconds)
|
|
3085
|
+
if rate is not None and rate.retry_after_seconds is not None
|
|
3086
|
+
else _checked_epoch_add(observed_epoch_seconds, 3_600)
|
|
3087
|
+
)
|
|
3088
|
+
if rate is not None and rate.reset_epoch_seconds is not None:
|
|
3089
|
+
retry_at = max(retry_at, rate.reset_epoch_seconds)
|
|
3090
|
+
return retry_at, "DATAGOV_RATE_LIMITED"
|
|
3091
|
+
wait_until = _pacing_wait(config, observed_epoch_seconds)
|
|
3092
|
+
return wait_until, "DATAGOV_PACING_REQUIRED" if wait_until is not None else None
|
|
3093
|
+
|
|
3094
|
+
|
|
3095
|
+
def run_catalog_fill(
|
|
3096
|
+
*,
|
|
3097
|
+
config: CatalogFillConfig,
|
|
3098
|
+
staging_root: Path,
|
|
3099
|
+
retriever: Any,
|
|
3100
|
+
expected_checkpoint_digest: str | None = None,
|
|
3101
|
+
clock_ns: Callable[[], int] = time.monotonic_ns,
|
|
3102
|
+
max_pages_this_invocation: int | None = None,
|
|
3103
|
+
authorization: DatagovV4Authorization | None = None,
|
|
3104
|
+
wall_clock_seconds: Callable[[], int] = lambda: int(time.time()),
|
|
3105
|
+
staging_descriptor: int | None = None,
|
|
3106
|
+
pacing_sleep: Callable[[float], None] | None = None,
|
|
3107
|
+
) -> CatalogFillResult:
|
|
3108
|
+
"""Run or resume one fill; only a reconciled stable state becomes eligible.
|
|
3109
|
+
|
|
3110
|
+
``pacing_sleep`` opts one invocation into waiting the provider's minimum request interval out
|
|
3111
|
+
in this process rather than returning ``DATAGOV_PACING_REQUIRED`` for the caller to wait out
|
|
3112
|
+
and resume. Absent it the return-and-resume contract is exactly what it always was, which is
|
|
3113
|
+
what the hosted one-shot job depends on. Present, the interval is unchanged and every page is
|
|
3114
|
+
still committed before the wait, so an interrupted paced sweep resumes from the same journal;
|
|
3115
|
+
what it buys is that the accumulated evidence tree is authenticated once for the invocation
|
|
3116
|
+
instead of once for every page, which is the difference between a sweep whose cost grows with
|
|
3117
|
+
the square of the records it has already staged and one that grows with the records.
|
|
3118
|
+
"""
|
|
3119
|
+
|
|
3120
|
+
if not isinstance(config, CatalogFillConfig):
|
|
3121
|
+
raise CatalogFillRefused("FILL_CONFIG", "fill.config", "strict configuration is required")
|
|
3122
|
+
if max_pages_this_invocation is not None and (
|
|
3123
|
+
type(max_pages_this_invocation) is not int or max_pages_this_invocation < 1
|
|
3124
|
+
):
|
|
3125
|
+
raise CatalogFillRefused(
|
|
3126
|
+
"FILL_INVOCATION_LIMIT", "fill.max_pages_this_invocation", "must be positive"
|
|
3127
|
+
)
|
|
3128
|
+
if pacing_sleep is not None and not callable(pacing_sleep):
|
|
3129
|
+
raise CatalogFillRefused(
|
|
3130
|
+
"FILL_PACING_SLEEP", "fill.pacing_sleep", "in-process pacing needs a callable wait"
|
|
3131
|
+
)
|
|
3132
|
+
if config.protocol == "datagov_v4":
|
|
3133
|
+
if not isinstance(authorization, DatagovV4Authorization):
|
|
3134
|
+
raise CatalogFillRefused(
|
|
3135
|
+
"DATAGOV_AUTHORIZATION", "fill.authorization", "strict authorization is required"
|
|
3136
|
+
)
|
|
3137
|
+
elif authorization is not None:
|
|
3138
|
+
raise CatalogFillRefused(
|
|
3139
|
+
"FILL_AUTHENTICATION",
|
|
3140
|
+
"fill.authorization",
|
|
3141
|
+
"legacy public protocols remain credential-free",
|
|
3142
|
+
)
|
|
3143
|
+
elif pacing_sleep is not None:
|
|
3144
|
+
raise CatalogFillRefused(
|
|
3145
|
+
"FILL_PACING_SLEEP",
|
|
3146
|
+
"fill.pacing_sleep",
|
|
3147
|
+
"only the keyed lane paces, so only it has an interval to wait out",
|
|
3148
|
+
)
|
|
3149
|
+
if config.protocol == "datagov_v4":
|
|
3150
|
+
return _run_datagov_v4_fill(
|
|
3151
|
+
config=config,
|
|
3152
|
+
staging_root=Path(staging_root),
|
|
3153
|
+
retriever=retriever,
|
|
3154
|
+
authorization=authorization,
|
|
3155
|
+
expected_checkpoint_digest=expected_checkpoint_digest,
|
|
3156
|
+
clock_ns=clock_ns,
|
|
3157
|
+
wall_clock_seconds=wall_clock_seconds,
|
|
3158
|
+
max_pages_this_invocation=max_pages_this_invocation,
|
|
3159
|
+
staging_descriptor=staging_descriptor,
|
|
3160
|
+
pacing_sleep=pacing_sleep,
|
|
3161
|
+
)
|
|
3162
|
+
_, harvester, transport_limits = _protocol_parts(config.protocol)
|
|
3163
|
+
if config.limits.max_response_bytes < transport_limits.max_response_bytes:
|
|
3164
|
+
transport_limits = replace(
|
|
3165
|
+
transport_limits, max_response_bytes=config.limits.max_response_bytes
|
|
3166
|
+
)
|
|
3167
|
+
staging = FillStaging(Path(staging_root), root_descriptor=staging_descriptor)
|
|
3168
|
+
with staging.locked():
|
|
3169
|
+
state = staging.load()
|
|
3170
|
+
if state is None:
|
|
3171
|
+
if expected_checkpoint_digest is not None:
|
|
3172
|
+
raise CatalogFillRefused(
|
|
3173
|
+
"FILL_CHECKPOINT_MISMATCH",
|
|
3174
|
+
"fill.expected_checkpoint_digest",
|
|
3175
|
+
"no live checkpoint matches the caller's expected digest",
|
|
3176
|
+
)
|
|
3177
|
+
state = staging.commit(
|
|
3178
|
+
_initial_state(config),
|
|
3179
|
+
expected_generation=-1,
|
|
3180
|
+
expected_checkpoint_digest=None,
|
|
3181
|
+
)
|
|
3182
|
+
elif (
|
|
3183
|
+
not isinstance(expected_checkpoint_digest, str)
|
|
3184
|
+
or not _SHA256.fullmatch(expected_checkpoint_digest)
|
|
3185
|
+
or expected_checkpoint_digest != staging.state_digest()
|
|
3186
|
+
):
|
|
3187
|
+
raise CatalogFillRefused(
|
|
3188
|
+
"FILL_CHECKPOINT_MISMATCH",
|
|
3189
|
+
"fill.expected_checkpoint_digest",
|
|
3190
|
+
"resume requires the exact caller-retained checkpoint digest",
|
|
3191
|
+
)
|
|
3192
|
+
_validate_state(state, config)
|
|
3193
|
+
if state["completed"]:
|
|
3194
|
+
return _result(staging, state, reason=None)
|
|
3195
|
+
if config.protocol == "datagov_v4":
|
|
3196
|
+
now = _wall_clock(wall_clock_seconds)
|
|
3197
|
+
wait_until = state["resume_not_before_epoch_seconds"]
|
|
3198
|
+
if wait_until is not None and now < wait_until:
|
|
3199
|
+
return _result(staging, state, reason="DATAGOV_RATE_LIMITED")
|
|
3200
|
+
invocation_pages = 0
|
|
3201
|
+
current_pass_map, current_pass_drift = _pass_map(staging, state)
|
|
3202
|
+
seen_cursors, seen_responses = _pass_loop_coordinates(staging, state)
|
|
3203
|
+
while True:
|
|
3204
|
+
preflight = _preflight_limit(state, config.limits)
|
|
3205
|
+
if preflight is not None:
|
|
3206
|
+
state = _stop(staging, state, preflight)
|
|
3207
|
+
return _result(staging, state, reason=preflight)
|
|
3208
|
+
|
|
3209
|
+
cursor = _cursor_from_dict(state["next_cursor"])
|
|
3210
|
+
request_url = _request_url(config, cursor)
|
|
3211
|
+
started = _clock(clock_ns)
|
|
3212
|
+
remaining_transport_limits = _remaining_transport_limits(
|
|
3213
|
+
state, config.limits, transport_limits
|
|
3214
|
+
)
|
|
3215
|
+
if config.protocol == "datagov_v4":
|
|
3216
|
+
payload, evidence, response_evidence = fetch_datagov_v4_response(
|
|
3217
|
+
retriever=retriever,
|
|
3218
|
+
url=request_url,
|
|
3219
|
+
observed_at=config.observed_at,
|
|
3220
|
+
limits=remaining_transport_limits,
|
|
3221
|
+
authorization=authorization,
|
|
3222
|
+
)
|
|
3223
|
+
else:
|
|
3224
|
+
payload, evidence, response_evidence = fetch_harvest_response(
|
|
3225
|
+
retriever=retriever,
|
|
3226
|
+
url=request_url,
|
|
3227
|
+
observed_at=config.observed_at,
|
|
3228
|
+
limits=remaining_transport_limits,
|
|
3229
|
+
)
|
|
3230
|
+
elapsed = _clock(clock_ns) - started
|
|
3231
|
+
_validate_response_endpoint(config.endpoint, request_url, response_evidence)
|
|
3232
|
+
response_digest = staging.write_shard("response", response_evidence.to_dict())
|
|
3233
|
+
state = dict(state)
|
|
3234
|
+
state["response_shards"].append(response_digest)
|
|
3235
|
+
budget = dict(state["budget"])
|
|
3236
|
+
budget["requests"] += response_evidence.requests
|
|
3237
|
+
budget["responses"] += response_evidence.responses
|
|
3238
|
+
budget["network_bytes"] += response_evidence.network_bytes
|
|
3239
|
+
budget["elapsed_ns"] += elapsed
|
|
3240
|
+
state["budget"] = budget
|
|
3241
|
+
if config.protocol == "datagov_v4":
|
|
3242
|
+
rate = response_evidence.rate_observation
|
|
3243
|
+
state["last_rate_observation"] = rate.to_dict() if rate is not None else None
|
|
3244
|
+
if response_evidence.status == 429:
|
|
3245
|
+
now = _wall_clock(wall_clock_seconds)
|
|
3246
|
+
retry_at = (
|
|
3247
|
+
now + rate.retry_after_seconds
|
|
3248
|
+
if rate is not None and rate.retry_after_seconds is not None
|
|
3249
|
+
else now + 3_600
|
|
3250
|
+
)
|
|
3251
|
+
if rate is not None and rate.reset_epoch_seconds is not None:
|
|
3252
|
+
retry_at = max(retry_at, rate.reset_epoch_seconds)
|
|
3253
|
+
state["resume_not_before_epoch_seconds"] = retry_at
|
|
3254
|
+
state["reason_code"] = "DATAGOV_RATE_LIMITED"
|
|
3255
|
+
state = _commit(staging, state)
|
|
3256
|
+
return _result(staging, state, reason="DATAGOV_RATE_LIMITED")
|
|
3257
|
+
if response_evidence.status == 403:
|
|
3258
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
3259
|
+
state["reason_code"] = "DATAGOV_AUTHENTICATION_FAILED"
|
|
3260
|
+
state = _commit(staging, state)
|
|
3261
|
+
return _result(staging, state, reason="DATAGOV_AUTHENTICATION_FAILED")
|
|
3262
|
+
if response_evidence.status != 200:
|
|
3263
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
3264
|
+
state["reason_code"] = "DATAGOV_HTTP_STATUS"
|
|
3265
|
+
state = _commit(staging, state)
|
|
3266
|
+
return _result(staging, state, reason="DATAGOV_HTTP_STATUS")
|
|
3267
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
3268
|
+
late = _late_limit(state, config.limits, response_evidence)
|
|
3269
|
+
if late is not None:
|
|
3270
|
+
state = _commit(staging, state)
|
|
3271
|
+
state = _stop(staging, state, late)
|
|
3272
|
+
return _result(staging, state, reason=late)
|
|
3273
|
+
|
|
3274
|
+
page = harvester.parse_page(
|
|
3275
|
+
payload,
|
|
3276
|
+
uri=request_url,
|
|
3277
|
+
observed_at=config.observed_at,
|
|
3278
|
+
evidence=evidence,
|
|
3279
|
+
response_evidence=response_evidence,
|
|
3280
|
+
cursor=cursor,
|
|
3281
|
+
)
|
|
3282
|
+
if page.next_cursor is not None and page.protocol == "stac":
|
|
3283
|
+
_validate_continuation(config.endpoint, page.next_cursor.value)
|
|
3284
|
+
count_drift = _page_count_drift(state, page)
|
|
3285
|
+
id_facts: dict[str, str] = {}
|
|
3286
|
+
drift = current_pass_drift or count_drift
|
|
3287
|
+
for record in page.records:
|
|
3288
|
+
digest = record.digest
|
|
3289
|
+
page_prior = id_facts.get(record.record_id)
|
|
3290
|
+
if page_prior is not None and page_prior != digest:
|
|
3291
|
+
drift = True
|
|
3292
|
+
continue
|
|
3293
|
+
id_facts[record.record_id] = digest
|
|
3294
|
+
for identifier, digest in id_facts.items():
|
|
3295
|
+
prior = current_pass_map.get(identifier)
|
|
3296
|
+
if prior is not None and prior != digest:
|
|
3297
|
+
drift = True
|
|
3298
|
+
elif prior is None:
|
|
3299
|
+
current_pass_map[identifier] = digest
|
|
3300
|
+
current_pass_drift = drift
|
|
3301
|
+
flagged = _flagged_records(page.records)
|
|
3302
|
+
projected = _projected_page_limit(state, config.limits, page)
|
|
3303
|
+
if projected is not None:
|
|
3304
|
+
state = _commit(staging, state)
|
|
3305
|
+
state = _stop(staging, state, projected)
|
|
3306
|
+
return _result(staging, state, reason=projected)
|
|
3307
|
+
page_payload = {
|
|
3308
|
+
"pass": state["pass_number"],
|
|
3309
|
+
"request_cursor": cursor.to_dict() if cursor is not None else None,
|
|
3310
|
+
"next_cursor": page.next_cursor.to_dict() if page.next_cursor is not None else None,
|
|
3311
|
+
"id_facts": id_facts,
|
|
3312
|
+
"records": [record.to_dict() for record in page.records],
|
|
3313
|
+
"provider_count": page.provider_count,
|
|
3314
|
+
"count_basis": page.count_basis,
|
|
3315
|
+
"skipped_record_ids": list(page.skipped_record_ids),
|
|
3316
|
+
"flagged_rights": flagged,
|
|
3317
|
+
"evidence": page.evidence.to_dict(),
|
|
3318
|
+
"response_evidence_digest": response_digest,
|
|
3319
|
+
}
|
|
3320
|
+
_refuse_loops(seen_cursors, seen_responses, cursor, page)
|
|
3321
|
+
page_digest = staging.write_shard("page", page_payload)
|
|
3322
|
+
state["page_shards"].append(page_digest)
|
|
3323
|
+
budget["pages"] += 1
|
|
3324
|
+
budget["records"] += len(page.records)
|
|
3325
|
+
budget["skipped_records"] += len(page.skipped_record_ids)
|
|
3326
|
+
state["budget"] = budget
|
|
3327
|
+
state["next_cursor"] = page.next_cursor.to_dict() if page.next_cursor else None
|
|
3328
|
+
state["provider_count"] = page.provider_count
|
|
3329
|
+
state["count_basis"] = page.count_basis
|
|
3330
|
+
state["pass_drift"] = drift
|
|
3331
|
+
state["reason_code"] = None
|
|
3332
|
+
state = _commit(staging, state)
|
|
3333
|
+
seen_cursors.add(canonical_sha256(cursor.to_dict() if cursor is not None else None))
|
|
3334
|
+
seen_responses.add(page.evidence.content_sha256)
|
|
3335
|
+
invocation_pages += 1
|
|
3336
|
+
if page.next_cursor is not None:
|
|
3337
|
+
after_page = _preflight_limit(state, config.limits)
|
|
3338
|
+
if after_page is not None:
|
|
3339
|
+
state = _stop(staging, state, after_page)
|
|
3340
|
+
return _result(staging, state, reason=after_page)
|
|
3341
|
+
if max_pages_this_invocation == invocation_pages:
|
|
3342
|
+
reason = (
|
|
3343
|
+
"INVOCATION_PAGE_LIMIT"
|
|
3344
|
+
if config.protocol == "datagov_v4"
|
|
3345
|
+
else "FILL_PAUSED"
|
|
3346
|
+
)
|
|
3347
|
+
state = _stop(staging, state, reason)
|
|
3348
|
+
return _result(staging, state, reason=reason)
|
|
3349
|
+
continue
|
|
3350
|
+
|
|
3351
|
+
pass_digest = canonical_sha256(current_pass_map)
|
|
3352
|
+
processed = _pass_processed_count(staging, state)
|
|
3353
|
+
count_agrees = page.provider_count is None or processed == page.provider_count
|
|
3354
|
+
one_shot = page.count_basis == "one_shot_response"
|
|
3355
|
+
stable = one_shot or state["previous_pass_map_sha256"] == pass_digest
|
|
3356
|
+
if stable and count_agrees and not current_pass_drift:
|
|
3357
|
+
state = dict(state)
|
|
3358
|
+
state["completed"] = True
|
|
3359
|
+
state["reason_code"] = None
|
|
3360
|
+
state["final_map_sha256"] = pass_digest
|
|
3361
|
+
state = _commit(staging, state)
|
|
3362
|
+
return _result(staging, state, reason=None)
|
|
3363
|
+
if state["pass_number"] >= config.max_convergence_passes:
|
|
3364
|
+
reason = "FILL_PROVIDER_COUNT" if not count_agrees else "FILL_NONCONVERGENT"
|
|
3365
|
+
state = _stop(staging, state, reason)
|
|
3366
|
+
return _result(staging, state, reason=reason)
|
|
3367
|
+
state = dict(state)
|
|
3368
|
+
state["previous_pass_map_sha256"] = pass_digest
|
|
3369
|
+
state["pass_number"] += 1
|
|
3370
|
+
state["pass_start_index"] = len(state["page_shards"])
|
|
3371
|
+
state["next_cursor"] = _initial_cursor(config)
|
|
3372
|
+
state["provider_count"] = None
|
|
3373
|
+
state["count_basis"] = None
|
|
3374
|
+
state["pass_drift"] = False
|
|
3375
|
+
state = _commit(staging, state)
|
|
3376
|
+
after_page = _preflight_limit(state, config.limits)
|
|
3377
|
+
if after_page is not None:
|
|
3378
|
+
state = _stop(staging, state, after_page)
|
|
3379
|
+
return _result(staging, state, reason=after_page)
|
|
3380
|
+
if max_pages_this_invocation == invocation_pages:
|
|
3381
|
+
reason = (
|
|
3382
|
+
"INVOCATION_PAGE_LIMIT" if config.protocol == "datagov_v4" else "FILL_PAUSED"
|
|
3383
|
+
)
|
|
3384
|
+
state = _stop(staging, state, reason)
|
|
3385
|
+
return _result(staging, state, reason=reason)
|
|
3386
|
+
current_pass_map = {}
|
|
3387
|
+
current_pass_drift = False
|
|
3388
|
+
seen_cursors = set()
|
|
3389
|
+
seen_responses = set()
|
|
3390
|
+
|
|
3391
|
+
|
|
3392
|
+
def _protocol_parts(protocol: str) -> tuple[str, SourceHarvester, Any]:
|
|
3393
|
+
if protocol == "ckan":
|
|
3394
|
+
return "ckan", CkanHarvester(), CKAN_LIMITS
|
|
3395
|
+
if protocol == "datagov_v4":
|
|
3396
|
+
return "datagov_v4", DatagovV4Harvester(), DATAGOV_V4_LIMITS
|
|
3397
|
+
if protocol == "stac":
|
|
3398
|
+
return "stac", StacHarvester(), STAC_LIMITS
|
|
3399
|
+
if protocol == "sdmx":
|
|
3400
|
+
return "sdmx", SdmxHarvester(), SDMX_LIMITS
|
|
3401
|
+
raise CatalogFillRefused("FILL_PROTOCOL", "fill.protocol", "unsupported protocol")
|
|
3402
|
+
|
|
3403
|
+
|
|
3404
|
+
def _initial_state(config: CatalogFillConfig) -> dict[str, Any]:
|
|
3405
|
+
state = {
|
|
3406
|
+
"config_sha256": config.digest,
|
|
3407
|
+
"endpoint": config.endpoint,
|
|
3408
|
+
"harvester_coordinate": config.coordinate,
|
|
3409
|
+
"predecessor_sha256": config.predecessor_sha256,
|
|
3410
|
+
"next_cursor": _initial_cursor(config),
|
|
3411
|
+
"page_shards": [],
|
|
3412
|
+
"response_shards": [],
|
|
3413
|
+
"budget": {
|
|
3414
|
+
"requests": 0,
|
|
3415
|
+
"responses": 0,
|
|
3416
|
+
"pages": 0,
|
|
3417
|
+
"records": 0,
|
|
3418
|
+
"network_bytes": 0,
|
|
3419
|
+
"elapsed_ns": 0,
|
|
3420
|
+
"skipped_records": 0,
|
|
3421
|
+
},
|
|
3422
|
+
"provider_count": None,
|
|
3423
|
+
"count_basis": None,
|
|
3424
|
+
"pass_number": 1,
|
|
3425
|
+
"pass_start_index": 0,
|
|
3426
|
+
"previous_pass_map_sha256": None,
|
|
3427
|
+
"pass_drift": False,
|
|
3428
|
+
"completed": False,
|
|
3429
|
+
"reason_code": None,
|
|
3430
|
+
"final_map_sha256": None,
|
|
3431
|
+
}
|
|
3432
|
+
if config.protocol == "datagov_v4":
|
|
3433
|
+
state["last_rate_observation"] = None
|
|
3434
|
+
state["resume_not_before_epoch_seconds"] = None
|
|
3435
|
+
return state
|
|
3436
|
+
|
|
3437
|
+
|
|
3438
|
+
def _initial_cursor(config: CatalogFillConfig) -> dict[str, str] | None:
|
|
3439
|
+
if config.protocol == "ckan":
|
|
3440
|
+
return HarvestCursor(
|
|
3441
|
+
protocol="ckan", kind="offset", value=f"0:{config.page_size}"
|
|
3442
|
+
).to_dict()
|
|
3443
|
+
return None
|
|
3444
|
+
|
|
3445
|
+
|
|
3446
|
+
def _validate_state(state: dict[str, Any], config: CatalogFillConfig) -> None:
|
|
3447
|
+
expected = (
|
|
3448
|
+
state.get("config_sha256") == config.digest
|
|
3449
|
+
and state.get("endpoint") == config.endpoint
|
|
3450
|
+
and state.get("harvester_coordinate") == config.coordinate
|
|
3451
|
+
and state.get("predecessor_sha256") == config.predecessor_sha256
|
|
3452
|
+
)
|
|
3453
|
+
if not expected:
|
|
3454
|
+
raise CatalogFillRefused(
|
|
3455
|
+
"FILL_CHECKPOINT_CONFIG", "fill.staging.state", "checkpoint coordinate differs"
|
|
3456
|
+
)
|
|
3457
|
+
required = {
|
|
3458
|
+
"config_sha256",
|
|
3459
|
+
"endpoint",
|
|
3460
|
+
"harvester_coordinate",
|
|
3461
|
+
"predecessor_sha256",
|
|
3462
|
+
"next_cursor",
|
|
3463
|
+
"page_shards",
|
|
3464
|
+
"response_shards",
|
|
3465
|
+
"budget",
|
|
3466
|
+
"provider_count",
|
|
3467
|
+
"count_basis",
|
|
3468
|
+
"pass_number",
|
|
3469
|
+
"pass_start_index",
|
|
3470
|
+
"previous_pass_map_sha256",
|
|
3471
|
+
"pass_drift",
|
|
3472
|
+
"completed",
|
|
3473
|
+
"reason_code",
|
|
3474
|
+
"final_map_sha256",
|
|
3475
|
+
"schema_version",
|
|
3476
|
+
"generation",
|
|
3477
|
+
}
|
|
3478
|
+
if config.protocol == "datagov_v4":
|
|
3479
|
+
required.update({"last_rate_observation", "resume_not_before_epoch_seconds"})
|
|
3480
|
+
if set(state) != required:
|
|
3481
|
+
raise CatalogFillRefused(
|
|
3482
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.state", "state keys differ"
|
|
3483
|
+
)
|
|
3484
|
+
if not isinstance(state["page_shards"], list) or not isinstance(state["response_shards"], list):
|
|
3485
|
+
raise CatalogFillRefused(
|
|
3486
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.state", "shard lists are invalid"
|
|
3487
|
+
)
|
|
3488
|
+
if type(state["pass_start_index"]) is not int or not 0 <= state["pass_start_index"] <= len(
|
|
3489
|
+
state["page_shards"]
|
|
3490
|
+
):
|
|
3491
|
+
raise CatalogFillRefused(
|
|
3492
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.state", "pass offset is invalid"
|
|
3493
|
+
)
|
|
3494
|
+
_cursor_from_dict(state["next_cursor"])
|
|
3495
|
+
|
|
3496
|
+
|
|
3497
|
+
def _cursor_from_dict(value: Any) -> HarvestCursor | None:
|
|
3498
|
+
if value is None:
|
|
3499
|
+
return None
|
|
3500
|
+
if not isinstance(value, dict) or set(value) != {"protocol", "kind", "value"}:
|
|
3501
|
+
raise CatalogFillRefused(
|
|
3502
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.cursor", "cursor envelope is invalid"
|
|
3503
|
+
)
|
|
3504
|
+
try:
|
|
3505
|
+
return HarvestCursor(protocol=value["protocol"], kind=value["kind"], value=value["value"])
|
|
3506
|
+
except Exception as error:
|
|
3507
|
+
raise CatalogFillRefused(
|
|
3508
|
+
"FILL_CHECKPOINT_CORRUPT", "fill.staging.cursor", "cursor is invalid"
|
|
3509
|
+
) from error
|
|
3510
|
+
|
|
3511
|
+
|
|
3512
|
+
def _request_url(config: CatalogFillConfig, cursor: HarvestCursor | None) -> str:
|
|
3513
|
+
if config.protocol == "ckan":
|
|
3514
|
+
if cursor is None:
|
|
3515
|
+
raise CatalogFillRefused("FILL_CURSOR", "fill.cursor", "CKAN cursor is required")
|
|
3516
|
+
start, rows = cursor.value.split(":", 1)
|
|
3517
|
+
parts = urlsplit(config.endpoint)
|
|
3518
|
+
query = [
|
|
3519
|
+
(key, value) for key, value in parse_qsl(parts.query) if key not in {"start", "rows"}
|
|
3520
|
+
]
|
|
3521
|
+
query.extend((("rows", rows), ("start", start)))
|
|
3522
|
+
return urlunsplit((parts.scheme, parts.netloc, parts.path, urlencode(sorted(query)), ""))
|
|
3523
|
+
if config.protocol == "stac" and cursor is not None:
|
|
3524
|
+
return cursor.value
|
|
3525
|
+
if config.protocol == "datagov_v4":
|
|
3526
|
+
if cursor is not None and cursor.protocol != "datagov_v4":
|
|
3527
|
+
raise CatalogFillRefused("FILL_CURSOR", "fill.cursor", "protocol mismatch")
|
|
3528
|
+
query = [("sort", config.sort), ("per_page", str(config.page_size))]
|
|
3529
|
+
if cursor is not None:
|
|
3530
|
+
query.append(("after", cursor.value))
|
|
3531
|
+
return f"{config.endpoint}?{urlencode(query)}"
|
|
3532
|
+
return config.endpoint
|
|
3533
|
+
|
|
3534
|
+
|
|
3535
|
+
def _validate_response_endpoint(
|
|
3536
|
+
endpoint: str, requested: str, response: HarvestResponseEvidence
|
|
3537
|
+
) -> None:
|
|
3538
|
+
if response.final_url != requested:
|
|
3539
|
+
raise CatalogFillRefused(
|
|
3540
|
+
"FILL_ENDPOINT_DRIFT",
|
|
3541
|
+
"fill.response.final_url",
|
|
3542
|
+
"zero-redirect response URL differs from the exact request",
|
|
3543
|
+
)
|
|
3544
|
+
base = urlsplit(endpoint)
|
|
3545
|
+
requested_parts = urlsplit(requested)
|
|
3546
|
+
for candidate, required_path in (
|
|
3547
|
+
(requested, base.path),
|
|
3548
|
+
(response.final_url, requested_parts.path),
|
|
3549
|
+
):
|
|
3550
|
+
parsed = urlsplit(candidate)
|
|
3551
|
+
if (
|
|
3552
|
+
parsed.scheme != "https"
|
|
3553
|
+
or parsed.hostname != base.hostname
|
|
3554
|
+
or parsed.port != base.port
|
|
3555
|
+
or parsed.path != required_path
|
|
3556
|
+
):
|
|
3557
|
+
raise CatalogFillRefused(
|
|
3558
|
+
"FILL_ENDPOINT_DRIFT",
|
|
3559
|
+
"fill.response.final_url",
|
|
3560
|
+
"continuation left endpoint origin",
|
|
3561
|
+
)
|
|
3562
|
+
|
|
3563
|
+
|
|
3564
|
+
def _validate_continuation(endpoint: str, continuation: str) -> None:
|
|
3565
|
+
base = urlsplit(endpoint)
|
|
3566
|
+
parsed = urlsplit(continuation)
|
|
3567
|
+
if (
|
|
3568
|
+
parsed.scheme != "https"
|
|
3569
|
+
or parsed.hostname != base.hostname
|
|
3570
|
+
or parsed.port != base.port
|
|
3571
|
+
or parsed.path != base.path
|
|
3572
|
+
):
|
|
3573
|
+
raise CatalogFillRefused(
|
|
3574
|
+
"FILL_ENDPOINT_DRIFT", "fill.page.next_cursor", "continuation left endpoint origin"
|
|
3575
|
+
)
|
|
3576
|
+
|
|
3577
|
+
|
|
3578
|
+
def _page_count_drift(state: dict[str, Any], page: HarvestPage) -> bool:
|
|
3579
|
+
basis = state["count_basis"]
|
|
3580
|
+
count = state["provider_count"]
|
|
3581
|
+
return (basis is not None and basis != page.count_basis) or (
|
|
3582
|
+
count is not None and page.provider_count != count
|
|
3583
|
+
)
|
|
3584
|
+
|
|
3585
|
+
|
|
3586
|
+
def _refuse_loops(
|
|
3587
|
+
seen_cursors: set[str],
|
|
3588
|
+
seen_responses: set[str],
|
|
3589
|
+
cursor: HarvestCursor | None,
|
|
3590
|
+
page: HarvestPage,
|
|
3591
|
+
) -> None:
|
|
3592
|
+
requested = cursor.to_dict() if cursor is not None else None
|
|
3593
|
+
requested_digest = canonical_sha256(requested)
|
|
3594
|
+
response_digest = page.evidence.content_sha256
|
|
3595
|
+
if requested_digest in seen_cursors:
|
|
3596
|
+
raise CatalogFillRefused("FILL_CURSOR_LOOP", "fill.cursor", "cursor repeated")
|
|
3597
|
+
if response_digest in seen_responses:
|
|
3598
|
+
raise CatalogFillRefused("FILL_RESPONSE_LOOP", "fill.response", "response repeated")
|
|
3599
|
+
if page.next_cursor is not None and page.next_cursor.to_dict() == requested:
|
|
3600
|
+
raise CatalogFillRefused("FILL_CURSOR_LOOP", "fill.cursor", "next cursor did not advance")
|
|
3601
|
+
|
|
3602
|
+
|
|
3603
|
+
def _pass_loop_coordinates(
|
|
3604
|
+
staging: FillStaging, state: dict[str, Any]
|
|
3605
|
+
) -> tuple[set[str], set[str]]:
|
|
3606
|
+
cursors: set[str] = set()
|
|
3607
|
+
responses: set[str] = set()
|
|
3608
|
+
for digest in state["page_shards"][state["pass_start_index"] :]:
|
|
3609
|
+
shard = staging.read_shard("page", digest)
|
|
3610
|
+
cursors.add(canonical_sha256(shard["request_cursor"]))
|
|
3611
|
+
responses.add(shard["evidence"]["content_sha256"])
|
|
3612
|
+
return cursors, responses
|
|
3613
|
+
|
|
3614
|
+
|
|
3615
|
+
def _pass_map(staging: FillStaging, state: dict[str, Any]) -> tuple[dict[str, str], bool]:
|
|
3616
|
+
result: dict[str, str] = {}
|
|
3617
|
+
drift = bool(state["pass_drift"])
|
|
3618
|
+
for digest in state["page_shards"][state["pass_start_index"] :]:
|
|
3619
|
+
shard = staging.read_shard("page", digest)
|
|
3620
|
+
for identifier, facts_digest in shard["id_facts"].items():
|
|
3621
|
+
prior = result.get(identifier)
|
|
3622
|
+
if prior is not None and prior != facts_digest:
|
|
3623
|
+
drift = True
|
|
3624
|
+
else:
|
|
3625
|
+
result[identifier] = facts_digest
|
|
3626
|
+
return result, drift
|
|
3627
|
+
|
|
3628
|
+
|
|
3629
|
+
def _pass_processed_count(staging: FillStaging, state: dict[str, Any]) -> int:
|
|
3630
|
+
identifiers: set[str] = set()
|
|
3631
|
+
skipped = 0
|
|
3632
|
+
for digest in state["page_shards"][state["pass_start_index"] :]:
|
|
3633
|
+
shard = staging.read_shard("page", digest)
|
|
3634
|
+
identifiers.update(shard["id_facts"])
|
|
3635
|
+
skipped += len(shard["skipped_record_ids"])
|
|
3636
|
+
return len(identifiers) + skipped
|
|
3637
|
+
|
|
3638
|
+
|
|
3639
|
+
def _flagged_records(records: tuple[HarvestedRecord, ...]) -> list[dict[str, Any]]:
|
|
3640
|
+
flagged: list[dict[str, Any]] = []
|
|
3641
|
+
for record in records:
|
|
3642
|
+
if record.protocol == "ckan":
|
|
3643
|
+
status = ckan_licence_rights(record.license_id)[0]
|
|
3644
|
+
elif record.protocol == "stac":
|
|
3645
|
+
status = stac_licence_rights(record.license_id)[0]
|
|
3646
|
+
elif record.protocol == "datagov_v4":
|
|
3647
|
+
status = "unclear"
|
|
3648
|
+
else:
|
|
3649
|
+
status = "unclear"
|
|
3650
|
+
if status in {"unclear", "prohibited"}:
|
|
3651
|
+
flagged.append(
|
|
3652
|
+
{
|
|
3653
|
+
"protocol": record.protocol,
|
|
3654
|
+
"record_id": record.record_id,
|
|
3655
|
+
"record_sha256": record.digest,
|
|
3656
|
+
"license_id": getattr(record, "license_id", None),
|
|
3657
|
+
"license_uri": getattr(record, "license_uri", None),
|
|
3658
|
+
"rights_status": status,
|
|
3659
|
+
"authority": "human_review_required",
|
|
3660
|
+
}
|
|
3661
|
+
)
|
|
3662
|
+
return sorted(flagged, key=lambda item: (item["protocol"], item["record_id"]))
|
|
3663
|
+
|
|
3664
|
+
|
|
3665
|
+
def _preflight_limit(state: dict[str, Any], limits: CatalogFillLimits) -> str | None:
|
|
3666
|
+
budget = state["budget"]
|
|
3667
|
+
checks = (
|
|
3668
|
+
("requests", limits.max_requests, "FILL_REQUEST_LIMIT"),
|
|
3669
|
+
("responses", limits.max_responses, "FILL_RESPONSE_LIMIT"),
|
|
3670
|
+
("pages", limits.max_pages, "FILL_PAGE_LIMIT"),
|
|
3671
|
+
("records", limits.max_records, "FILL_RECORD_LIMIT"),
|
|
3672
|
+
("network_bytes", limits.max_aggregate_bytes, "FILL_AGGREGATE_BYTE_LIMIT"),
|
|
3673
|
+
("elapsed_ns", limits.max_wall_time_ns, "FILL_WALL_TIME_LIMIT"),
|
|
3674
|
+
)
|
|
3675
|
+
for field, maximum, code in checks:
|
|
3676
|
+
if budget[field] >= maximum:
|
|
3677
|
+
return code
|
|
3678
|
+
return None
|
|
3679
|
+
|
|
3680
|
+
|
|
3681
|
+
def _remaining_transport_limits(state: dict[str, Any], limits: CatalogFillLimits, transport):
|
|
3682
|
+
"""Narrow the next fetch to the aggregate budget still authorized before it starts."""
|
|
3683
|
+
|
|
3684
|
+
budget = state["budget"]
|
|
3685
|
+
remaining_requests = min(
|
|
3686
|
+
limits.max_requests - budget["requests"],
|
|
3687
|
+
limits.max_responses - budget["responses"],
|
|
3688
|
+
transport.max_requests,
|
|
3689
|
+
)
|
|
3690
|
+
remaining_bytes = min(
|
|
3691
|
+
limits.max_aggregate_bytes - budget["network_bytes"],
|
|
3692
|
+
limits.max_response_bytes,
|
|
3693
|
+
transport.max_response_bytes,
|
|
3694
|
+
)
|
|
3695
|
+
remaining_ns = limits.max_wall_time_ns - budget["elapsed_ns"]
|
|
3696
|
+
if remaining_requests < 1 or remaining_bytes < 1 or remaining_ns < 1:
|
|
3697
|
+
raise CatalogFillRefused(
|
|
3698
|
+
"FILL_LIMIT", "fill.limits", "no aggregate transport budget remains"
|
|
3699
|
+
)
|
|
3700
|
+
remaining_seconds = remaining_ns / 1_000_000_000
|
|
3701
|
+
return replace(
|
|
3702
|
+
transport,
|
|
3703
|
+
max_response_bytes=remaining_bytes,
|
|
3704
|
+
max_aggregate_response_bytes=min(
|
|
3705
|
+
limits.max_aggregate_bytes - budget["network_bytes"],
|
|
3706
|
+
transport.max_aggregate_response_bytes,
|
|
3707
|
+
),
|
|
3708
|
+
max_requests=remaining_requests,
|
|
3709
|
+
max_redirects=min(transport.max_redirects, remaining_requests - 1),
|
|
3710
|
+
connect_timeout_seconds=min(transport.connect_timeout_seconds, remaining_seconds),
|
|
3711
|
+
read_timeout_seconds=min(transport.read_timeout_seconds, remaining_seconds),
|
|
3712
|
+
total_timeout_seconds=min(transport.total_timeout_seconds, remaining_seconds),
|
|
3713
|
+
)
|
|
3714
|
+
|
|
3715
|
+
|
|
3716
|
+
def _late_limit(
|
|
3717
|
+
state: dict[str, Any], limits: CatalogFillLimits, response: HarvestResponseEvidence
|
|
3718
|
+
) -> str | None:
|
|
3719
|
+
if response.content_bytes > limits.max_response_bytes:
|
|
3720
|
+
return "FILL_RESPONSE_BYTE_LIMIT"
|
|
3721
|
+
budget = state["budget"]
|
|
3722
|
+
checks = (
|
|
3723
|
+
(budget["requests"], limits.max_requests, "FILL_REQUEST_LIMIT"),
|
|
3724
|
+
(budget["responses"], limits.max_responses, "FILL_RESPONSE_LIMIT"),
|
|
3725
|
+
(budget["network_bytes"], limits.max_aggregate_bytes, "FILL_AGGREGATE_BYTE_LIMIT"),
|
|
3726
|
+
(budget["elapsed_ns"], limits.max_wall_time_ns, "FILL_WALL_TIME_LIMIT"),
|
|
3727
|
+
)
|
|
3728
|
+
for value, maximum, code in checks:
|
|
3729
|
+
if value > maximum:
|
|
3730
|
+
return code
|
|
3731
|
+
return None
|
|
3732
|
+
|
|
3733
|
+
|
|
3734
|
+
def _v4_transport_terminal_reason(
|
|
3735
|
+
transport_code: str, state: dict[str, Any], limits: CatalogFillLimits
|
|
3736
|
+
) -> str:
|
|
3737
|
+
"""Normalize real transport cap refusals to durable fill-budget reasons."""
|
|
3738
|
+
|
|
3739
|
+
budget = state["budget"]
|
|
3740
|
+
if budget["network_bytes"] > limits.max_aggregate_bytes:
|
|
3741
|
+
return "FILL_AGGREGATE_BYTE_LIMIT"
|
|
3742
|
+
if budget["elapsed_ns"] > limits.max_wall_time_ns:
|
|
3743
|
+
return "FILL_WALL_TIME_LIMIT"
|
|
3744
|
+
if (
|
|
3745
|
+
transport_code == "AGGREGATE_RESPONSE_LIMIT"
|
|
3746
|
+
and budget["network_bytes"] >= limits.max_aggregate_bytes
|
|
3747
|
+
):
|
|
3748
|
+
return "FILL_AGGREGATE_BYTE_LIMIT"
|
|
3749
|
+
if transport_code == "TOTAL_TIMEOUT" and budget["elapsed_ns"] >= limits.max_wall_time_ns:
|
|
3750
|
+
return "FILL_WALL_TIME_LIMIT"
|
|
3751
|
+
if transport_code == "RESPONSE_LIMIT":
|
|
3752
|
+
return "FILL_RESPONSE_BYTE_LIMIT"
|
|
3753
|
+
return transport_code
|
|
3754
|
+
|
|
3755
|
+
|
|
3756
|
+
def _v4_prospective_terminal_reason(
|
|
3757
|
+
staging: FillStaging,
|
|
3758
|
+
state: dict[str, Any],
|
|
3759
|
+
*,
|
|
3760
|
+
config: CatalogFillConfig,
|
|
3761
|
+
request_item: Any,
|
|
3762
|
+
response_item: Any,
|
|
3763
|
+
response: HarvestResponseEvidence | None,
|
|
3764
|
+
) -> str | None:
|
|
3765
|
+
"""Reproduce a rejected page/record admission from its authenticated raw tail."""
|
|
3766
|
+
|
|
3767
|
+
if state.get("reason_code") not in {"FILL_PAGE_LIMIT", "FILL_RECORD_LIMIT"}:
|
|
3768
|
+
return None
|
|
3769
|
+
if (
|
|
3770
|
+
response is None
|
|
3771
|
+
or response.status != 200
|
|
3772
|
+
or not isinstance(request_item, dict)
|
|
3773
|
+
or not isinstance(response_item, dict)
|
|
3774
|
+
or response_item.get("request_attempt") != state["budget"]["requests"]
|
|
3775
|
+
or response_item.get("request_url") != request_item.get("request_url")
|
|
3776
|
+
or response.final_url != request_item.get("request_url")
|
|
3777
|
+
or not _SHA256.fullmatch(response_item.get("raw_response_sha256", ""))
|
|
3778
|
+
):
|
|
3779
|
+
return None
|
|
3780
|
+
raw_digest = response_item["raw_response_sha256"]
|
|
3781
|
+
request_url = request_item["request_url"]
|
|
3782
|
+
try:
|
|
3783
|
+
payload = staging.read_blob("raw_response", raw_digest)
|
|
3784
|
+
page = DatagovV4Harvester().parse_page(
|
|
3785
|
+
payload,
|
|
3786
|
+
uri=request_url,
|
|
3787
|
+
observed_at=config.observed_at,
|
|
3788
|
+
evidence=EvidenceReference(
|
|
3789
|
+
uri=request_url,
|
|
3790
|
+
observed_at=config.observed_at,
|
|
3791
|
+
content_sha256=raw_digest,
|
|
3792
|
+
media_type="application/json",
|
|
3793
|
+
),
|
|
3794
|
+
response_evidence=response,
|
|
3795
|
+
cursor=_cursor_from_dict(request_item.get("cursor")),
|
|
3796
|
+
)
|
|
3797
|
+
except (CatalogHarvestError, CatalogFillRefused, CanonicalJSONError, TypeError, ValueError):
|
|
3798
|
+
raise _v4_evidence_corrupt("rejected page cannot be reproduced from its raw tail") from None
|
|
3799
|
+
return _projected_page_limit(state, config.limits, page)
|
|
3800
|
+
|
|
3801
|
+
|
|
3802
|
+
def _projected_page_limit(
|
|
3803
|
+
state: dict[str, Any], limits: CatalogFillLimits, page: HarvestPage
|
|
3804
|
+
) -> str | None:
|
|
3805
|
+
if state["budget"]["pages"] + 1 > limits.max_pages:
|
|
3806
|
+
return "FILL_PAGE_LIMIT"
|
|
3807
|
+
if state["budget"]["records"] + len(page.records) > limits.max_records:
|
|
3808
|
+
return "FILL_RECORD_LIMIT"
|
|
3809
|
+
return None
|
|
3810
|
+
|
|
3811
|
+
|
|
3812
|
+
def _stop(staging: FillStaging, state: dict[str, Any], reason: str) -> dict[str, Any]:
|
|
3813
|
+
if state.get("reason_code") == reason:
|
|
3814
|
+
return state
|
|
3815
|
+
next_state = dict(state)
|
|
3816
|
+
next_state["reason_code"] = reason
|
|
3817
|
+
return _commit(staging, next_state)
|
|
3818
|
+
|
|
3819
|
+
|
|
3820
|
+
def _commit(
|
|
3821
|
+
staging: FillStaging,
|
|
3822
|
+
next_state: dict[str, Any],
|
|
3823
|
+
) -> dict[str, Any]:
|
|
3824
|
+
return staging.commit(
|
|
3825
|
+
next_state,
|
|
3826
|
+
expected_generation=next_state["generation"],
|
|
3827
|
+
expected_checkpoint_digest=staging.state_digest(),
|
|
3828
|
+
)
|
|
3829
|
+
|
|
3830
|
+
|
|
3831
|
+
def _result(
|
|
3832
|
+
staging: FillStaging, state: dict[str, Any], *, reason: str | None
|
|
3833
|
+
) -> CatalogFillResult:
|
|
3834
|
+
final_map, _ = _pass_map(staging, state)
|
|
3835
|
+
flags: dict[tuple[str, str], dict[str, Any]] = {}
|
|
3836
|
+
for digest in state["page_shards"][state["pass_start_index"] :]:
|
|
3837
|
+
shard = staging.read_shard("page", digest)
|
|
3838
|
+
for item in shard["flagged_rights"]:
|
|
3839
|
+
flags[(item["protocol"], item["record_id"])] = item
|
|
3840
|
+
flagged_queue = [flags[key] for key in sorted(flags)]
|
|
3841
|
+
receipt = CatalogFillReceipt(
|
|
3842
|
+
config_sha256=state["config_sha256"],
|
|
3843
|
+
endpoint=state["endpoint"],
|
|
3844
|
+
harvester_coordinate=state["harvester_coordinate"],
|
|
3845
|
+
predecessor_sha256=state["predecessor_sha256"],
|
|
3846
|
+
status="complete" if state["completed"] else "incomplete",
|
|
3847
|
+
reason_code=reason,
|
|
3848
|
+
requests=state["budget"]["requests"],
|
|
3849
|
+
responses=state["budget"]["responses"],
|
|
3850
|
+
pages=state["budget"]["pages"],
|
|
3851
|
+
records=state["budget"]["records"],
|
|
3852
|
+
unique_records=len(final_map),
|
|
3853
|
+
network_bytes=state["budget"]["network_bytes"],
|
|
3854
|
+
elapsed_ns=state["budget"]["elapsed_ns"],
|
|
3855
|
+
provider_count=state["provider_count"],
|
|
3856
|
+
count_basis=state["count_basis"],
|
|
3857
|
+
skipped_records=state["budget"]["skipped_records"],
|
|
3858
|
+
flagged_rights=len(flagged_queue),
|
|
3859
|
+
convergence_passes=state["pass_number"],
|
|
3860
|
+
response_evidence_digests=tuple(state["response_shards"]),
|
|
3861
|
+
page_shard_digests=tuple(state["page_shards"]),
|
|
3862
|
+
flagged_queue_sha256=canonical_sha256(flagged_queue),
|
|
3863
|
+
rate_observation=state.get("last_rate_observation"),
|
|
3864
|
+
resume_not_before_epoch_seconds=state.get("resume_not_before_epoch_seconds"),
|
|
3865
|
+
)
|
|
3866
|
+
state_digest = staging.state_digest()
|
|
3867
|
+
return CatalogFillResult(
|
|
3868
|
+
status=receipt.status,
|
|
3869
|
+
reason_code=reason,
|
|
3870
|
+
receipt=receipt,
|
|
3871
|
+
checkpoint_sha256=state_digest,
|
|
3872
|
+
eligible_staging_sha256=state_digest if state["completed"] else None,
|
|
3873
|
+
)
|
|
3874
|
+
|
|
3875
|
+
|
|
3876
|
+
def _clock(clock_ns: Callable[[], int]) -> int:
|
|
3877
|
+
value = clock_ns()
|
|
3878
|
+
if type(value) is not int or value < 0:
|
|
3879
|
+
raise CatalogFillRefused("FILL_CLOCK", "fill.clock", "clock must return nonnegative int")
|
|
3880
|
+
return value
|
|
3881
|
+
|
|
3882
|
+
|
|
3883
|
+
def _wall_clock(clock_seconds: Callable[[], int]) -> int:
|
|
3884
|
+
value = clock_seconds()
|
|
3885
|
+
if type(value) is not int or not 0 <= value <= (1 << 53) - 1:
|
|
3886
|
+
raise CatalogFillRefused(
|
|
3887
|
+
"FILL_CLOCK", "fill.wall_clock", "wall clock must return a nonnegative integer"
|
|
3888
|
+
)
|
|
3889
|
+
return value
|