mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1115 @@
|
|
|
1
|
+
"""Networkless normalizer child for hosted public connectors.
|
|
2
|
+
|
|
3
|
+
Despite the name, nothing here crawls. This is the process behind the mr-data-crawler
|
|
4
|
+
console script, and it holds no network authority. The mr-data-crawler-job coordinator
|
|
5
|
+
fetches the response and passes it on a file descriptor, then spawns this process under a
|
|
6
|
+
seccomp filter that denies socket and connect. This module authenticates that framed input
|
|
7
|
+
and normalizes OpenLigaDB match JSON into the fixed CSV column set.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import csv
|
|
14
|
+
import io
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
import sys
|
|
19
|
+
from dataclasses import asdict
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
from typing import Any, Protocol
|
|
22
|
+
from urllib.parse import urlsplit
|
|
23
|
+
|
|
24
|
+
from mostlyright.data_harness.acquisition.parsing import (
|
|
25
|
+
ParseLimits,
|
|
26
|
+
parse_csv_evidence,
|
|
27
|
+
parse_tabular_bytes,
|
|
28
|
+
row_digest_for,
|
|
29
|
+
)
|
|
30
|
+
from mostlyright.data_harness.acquisition.url_policy import canonical_public_hostname
|
|
31
|
+
from mostlyright.data_harness.canonical import (
|
|
32
|
+
CanonicalJSONError,
|
|
33
|
+
canonical_json_bytes,
|
|
34
|
+
parse_canonical_json,
|
|
35
|
+
sha256_bytes,
|
|
36
|
+
)
|
|
37
|
+
from mostlyright.data_harness.formats import DATA_FORMATS
|
|
38
|
+
from mostlyright.data_harness.hosted_crawler_protocol import (
|
|
39
|
+
CRAWLER_REQUEST_SCHEMA_V3,
|
|
40
|
+
CRAWLER_RESULT_SCHEMA_V1,
|
|
41
|
+
CRAWLER_RESULT_SCHEMA_V3,
|
|
42
|
+
MAX_CRAWLER_REQUEST_BYTES,
|
|
43
|
+
MAX_CRAWLER_RESULT_BYTES,
|
|
44
|
+
MAX_NORMALIZED_SOURCE_BYTES,
|
|
45
|
+
OPENLIGADB_EGRESS_POLICY_ATTESTATION,
|
|
46
|
+
OPENLIGADB_QUERY_ALLOWLIST_DETAIL,
|
|
47
|
+
OPENLIGADB_QUERY_FIELDS_DETAIL,
|
|
48
|
+
PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
|
|
49
|
+
HostedCrawlerProtocolError,
|
|
50
|
+
HostedCrawlerRequest,
|
|
51
|
+
HostedCrawlerResult,
|
|
52
|
+
parse_crawler_request,
|
|
53
|
+
)
|
|
54
|
+
from mostlyright.data_harness.local_contracts import ContractError, validate_source_locator
|
|
55
|
+
from mostlyright.data_harness.readers.contracts import ReaderPin
|
|
56
|
+
from mostlyright.data_harness.readers.registry import TOOLBOX
|
|
57
|
+
from mostlyright.data_harness.readers.samples import DecodedFacts, warm_up
|
|
58
|
+
|
|
59
|
+
CRAWLER_EGRESS_POLICY_ENV = "MOSTLYRIGHT_CRAWLER_EGRESS_POLICY_ATTESTATION"
|
|
60
|
+
# The public-HTTPS fetch ceiling, equal to the hosted contract's max_source_bytes maximum and
|
|
61
|
+
# stated as max_response_bytes inside PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION. The OpenLigaDB
|
|
62
|
+
# retriever keeps its own reviewed 8 MiB cap, stated in its policy attestation.
|
|
63
|
+
MAX_RAW_RESPONSE_BYTES = 268_435_456
|
|
64
|
+
MAX_NORMALIZATION_HEADER_BYTES = 128 * 1024
|
|
65
|
+
MAX_NORMALIZATION_INPUT_BYTES = 4 + MAX_NORMALIZATION_HEADER_BYTES + MAX_RAW_RESPONSE_BYTES
|
|
66
|
+
NORMALIZATION_INPUT_SCHEMA = "mostlyright-hosted-crawler-normalization-input.v1"
|
|
67
|
+
WARM_UP_RESULT_SCHEMA = "mostlyright-hosted-crawler-reader-warm-up.v1"
|
|
68
|
+
MAX_ROWS = 20_000
|
|
69
|
+
MAX_TEXT_BYTES = 512
|
|
70
|
+
MAX_SAFE_INTEGER = 9_007_199_254_740_991
|
|
71
|
+
|
|
72
|
+
_COLUMNS = (
|
|
73
|
+
"match_id",
|
|
74
|
+
"match_datetime_utc",
|
|
75
|
+
"last_update_datetime",
|
|
76
|
+
"league_shortcut",
|
|
77
|
+
"league_season",
|
|
78
|
+
"group_order_id",
|
|
79
|
+
"team1_id",
|
|
80
|
+
"team1_name",
|
|
81
|
+
"team2_id",
|
|
82
|
+
"team2_name",
|
|
83
|
+
"match_finished",
|
|
84
|
+
"score1",
|
|
85
|
+
"score2",
|
|
86
|
+
)
|
|
87
|
+
_QUERY_FIELDS = frozenset({"league_shortcut", "league_season"})
|
|
88
|
+
_PUBLIC_QUERY_FIELDS = frozenset(
|
|
89
|
+
{"url", "data_format", "filename", "reader_pin", "resource_caps", "limits"}
|
|
90
|
+
)
|
|
91
|
+
_PUBLIC_LIMIT_FIELDS = frozenset(
|
|
92
|
+
{"max_source_bytes", "max_normalized_bytes", "max_rows", "max_columns"}
|
|
93
|
+
)
|
|
94
|
+
_PUBLIC_RESOURCE_CAP_FIELDS = frozenset(
|
|
95
|
+
{"max_declared_cells", "max_container_members", "max_nesting_depth"}
|
|
96
|
+
)
|
|
97
|
+
_LEAGUE = re.compile(r"^[a-z0-9][a-z0-9-]{0,15}$")
|
|
98
|
+
_DIGEST = re.compile(r"^[0-9a-f]{64}$")
|
|
99
|
+
_UTC_TIMESTAMP = re.compile(
|
|
100
|
+
r"^[0-9]{4}-(?:0[1-9]|1[0-2])-(?:0[1-9]|[12][0-9]|3[01])"
|
|
101
|
+
r"T(?:[01][0-9]|2[0-3]):[0-5][0-9]:[0-5][0-9](?:\.[0-9]{1,6})?Z$"
|
|
102
|
+
)
|
|
103
|
+
_FORBIDDEN_ENVIRONMENT = re.compile(
|
|
104
|
+
r"(?:^|_)(?:STUDIO|TOKEN|SECRET|PASSWORD|CREDENTIAL|AUTHORIZATION|"
|
|
105
|
+
r"GOOGLE_APPLICATION_CREDENTIALS|HOME)(?:_|$)",
|
|
106
|
+
re.IGNORECASE,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class HostedCrawlerError(RuntimeError):
|
|
111
|
+
"""One safe fixed-crawler execution invariant failed."""
|
|
112
|
+
|
|
113
|
+
def __init__(self, code: str, detail: str) -> None:
|
|
114
|
+
self.code = code
|
|
115
|
+
self.detail = detail
|
|
116
|
+
super().__init__(f"{detail} [{code}]")
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class RetrievedSource(Protocol):
|
|
120
|
+
source_url: str
|
|
121
|
+
final_url: str
|
|
122
|
+
media_type: str
|
|
123
|
+
content: bytes
|
|
124
|
+
content_sha256: str
|
|
125
|
+
transport_evidence_digest: str
|
|
126
|
+
hops: tuple[Any, ...]
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def validate_crawler_request(
|
|
130
|
+
request: HostedCrawlerRequest,
|
|
131
|
+
*,
|
|
132
|
+
expected_egress_policy_attestation: str,
|
|
133
|
+
) -> tuple[str, int]:
|
|
134
|
+
"""Validate the exact deployed adapter and return its bounded query.
|
|
135
|
+
|
|
136
|
+
expected_egress_policy_attestation is the attestation the calling process trusts. Only
|
|
137
|
+
crawler_main supplies an independent value, read from the environment. Every other call
|
|
138
|
+
site passes the request's own attestation, so for those the first check cannot fail and
|
|
139
|
+
the binding rests on the OPENLIGADB_EGRESS_POLICY_ATTESTATION comparison after it.
|
|
140
|
+
"""
|
|
141
|
+
|
|
142
|
+
if request.egress_policy_attestation != expected_egress_policy_attestation:
|
|
143
|
+
raise HostedCrawlerError(
|
|
144
|
+
"CRAWLER_POLICY_MISMATCH",
|
|
145
|
+
"crawler request differs from the deployed egress-policy attestation",
|
|
146
|
+
)
|
|
147
|
+
expected_policy = {
|
|
148
|
+
"external.openligadb": OPENLIGADB_EGRESS_POLICY_ATTESTATION,
|
|
149
|
+
"public.https": PUBLIC_HTTPS_EGRESS_POLICY_ATTESTATION,
|
|
150
|
+
}.get(request.adapter_id)
|
|
151
|
+
if expected_egress_policy_attestation != expected_policy:
|
|
152
|
+
raise HostedCrawlerError(
|
|
153
|
+
"CRAWLER_POLICY_MISMATCH",
|
|
154
|
+
"deployed egress-policy attestation is not the reviewed adapter policy",
|
|
155
|
+
)
|
|
156
|
+
if request.adapter_version != "1.0.0" or expected_policy is None:
|
|
157
|
+
raise HostedCrawlerError(
|
|
158
|
+
"CRAWLER_ADAPTER_UNAVAILABLE",
|
|
159
|
+
"crawler request adapter coordinate is not image-allowlisted",
|
|
160
|
+
)
|
|
161
|
+
if request.adapter_id == "public.https":
|
|
162
|
+
_public_https_query(request.query)
|
|
163
|
+
return "public.https", 1
|
|
164
|
+
league, season = _query(request.query)
|
|
165
|
+
return league, season
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def encode_normalization_input(
|
|
169
|
+
request: HostedCrawlerRequest,
|
|
170
|
+
retrieved: RetrievedSource,
|
|
171
|
+
) -> bytes:
|
|
172
|
+
"""Frame trusted retrieval evidence and raw bytes for the networkless child."""
|
|
173
|
+
|
|
174
|
+
coordinate = validate_crawler_request(
|
|
175
|
+
request,
|
|
176
|
+
expected_egress_policy_attestation=request.egress_policy_attestation,
|
|
177
|
+
)
|
|
178
|
+
if request.adapter_id == "public.https":
|
|
179
|
+
public = _public_https_query(request.query)
|
|
180
|
+
expected_url = public["url"]
|
|
181
|
+
expected_media_types = _public_expected_media_types(public)
|
|
182
|
+
final_host = _canonical_url_host(retrieved.final_url)
|
|
183
|
+
expected_host = _canonical_url_host(expected_url)
|
|
184
|
+
evidence_valid = (
|
|
185
|
+
retrieved.source_url == expected_url
|
|
186
|
+
and final_host == expected_host
|
|
187
|
+
and retrieved.media_type in expected_media_types
|
|
188
|
+
and 1 <= len(retrieved.content) <= public["limits"]["max_source_bytes"]
|
|
189
|
+
)
|
|
190
|
+
else:
|
|
191
|
+
league, season = coordinate
|
|
192
|
+
expected_url = f"https://api.openligadb.de/getmatchdata/{league}/{season}"
|
|
193
|
+
evidence_valid = (
|
|
194
|
+
retrieved.source_url == expected_url
|
|
195
|
+
and retrieved.final_url == expected_url
|
|
196
|
+
and retrieved.media_type == "application/json"
|
|
197
|
+
)
|
|
198
|
+
if not evidence_valid or (
|
|
199
|
+
not 1 <= len(retrieved.content) <= MAX_RAW_RESPONSE_BYTES
|
|
200
|
+
or sha256_bytes(retrieved.content) != retrieved.content_sha256
|
|
201
|
+
or _DIGEST.fullmatch(retrieved.transport_evidence_digest) is None
|
|
202
|
+
):
|
|
203
|
+
raise HostedCrawlerError(
|
|
204
|
+
"CRAWLER_RETRIEVAL_EVIDENCE_INVALID",
|
|
205
|
+
"trusted retrieval evidence is invalid or outside the reviewed source authority",
|
|
206
|
+
)
|
|
207
|
+
if not retrieved.hops:
|
|
208
|
+
raise HostedCrawlerError(
|
|
209
|
+
"CRAWLER_RETRIEVAL_EVIDENCE_INVALID",
|
|
210
|
+
"trusted retrieval evidence omitted its terminal HTTP hop",
|
|
211
|
+
)
|
|
212
|
+
terminal = retrieved.hops[-1]
|
|
213
|
+
if (
|
|
214
|
+
type(terminal.status) is not int
|
|
215
|
+
or not 200 <= terminal.status <= 299
|
|
216
|
+
or (terminal.etag is not None and not isinstance(terminal.etag, str))
|
|
217
|
+
or (terminal.last_modified is not None and not isinstance(terminal.last_modified, str))
|
|
218
|
+
):
|
|
219
|
+
raise HostedCrawlerError(
|
|
220
|
+
"CRAWLER_RETRIEVAL_EVIDENCE_INVALID",
|
|
221
|
+
"trusted retrieval evidence has invalid terminal HTTP facts",
|
|
222
|
+
)
|
|
223
|
+
header = canonical_json_bytes(
|
|
224
|
+
{
|
|
225
|
+
"schema_version": NORMALIZATION_INPUT_SCHEMA,
|
|
226
|
+
"request": request.to_dict(),
|
|
227
|
+
"source_url": retrieved.source_url,
|
|
228
|
+
"final_url": retrieved.final_url,
|
|
229
|
+
"media_type": retrieved.media_type,
|
|
230
|
+
"raw_content_sha256": retrieved.content_sha256,
|
|
231
|
+
"raw_size_bytes": len(retrieved.content),
|
|
232
|
+
"transport_evidence_digest": retrieved.transport_evidence_digest,
|
|
233
|
+
"http_status_code": terminal.status,
|
|
234
|
+
"etag": terminal.etag,
|
|
235
|
+
"last_modified": terminal.last_modified,
|
|
236
|
+
}
|
|
237
|
+
)
|
|
238
|
+
if len(header) > MAX_NORMALIZATION_HEADER_BYTES:
|
|
239
|
+
raise HostedCrawlerError(
|
|
240
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
241
|
+
"crawler normalization header exceeds its byte bound",
|
|
242
|
+
)
|
|
243
|
+
return len(header).to_bytes(4, "big") + header + retrieved.content
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def parse_normalization_input(
|
|
247
|
+
content: bytes,
|
|
248
|
+
) -> tuple[HostedCrawlerRequest, bytes, str, str, str, str, int, str | None, str | None]:
|
|
249
|
+
"""Parse and authenticate the exact coordinator-to-normalizer frame.
|
|
250
|
+
|
|
251
|
+
The returned tuple is, in order: the request, the raw response bytes, the source URL,
|
|
252
|
+
the media type, the raw-content digest, the transport-evidence digest, the terminal HTTP
|
|
253
|
+
status, the raw ETag, and the raw Last-Modified value.
|
|
254
|
+
"""
|
|
255
|
+
|
|
256
|
+
if not 5 <= len(content) <= MAX_NORMALIZATION_INPUT_BYTES:
|
|
257
|
+
raise HostedCrawlerError(
|
|
258
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
259
|
+
"crawler normalization input exceeds its byte bound",
|
|
260
|
+
)
|
|
261
|
+
header_size = int.from_bytes(content[:4], "big")
|
|
262
|
+
if not 1 <= header_size <= MAX_NORMALIZATION_HEADER_BYTES or 4 + header_size >= len(content):
|
|
263
|
+
raise HostedCrawlerError(
|
|
264
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
265
|
+
"crawler normalization header framing is invalid",
|
|
266
|
+
)
|
|
267
|
+
header_bytes = content[4 : 4 + header_size]
|
|
268
|
+
raw = content[4 + header_size :]
|
|
269
|
+
header = parse_canonical_json(header_bytes)
|
|
270
|
+
expected_fields = {
|
|
271
|
+
"schema_version",
|
|
272
|
+
"request",
|
|
273
|
+
"source_url",
|
|
274
|
+
"final_url",
|
|
275
|
+
"media_type",
|
|
276
|
+
"raw_content_sha256",
|
|
277
|
+
"raw_size_bytes",
|
|
278
|
+
"transport_evidence_digest",
|
|
279
|
+
"http_status_code",
|
|
280
|
+
"etag",
|
|
281
|
+
"last_modified",
|
|
282
|
+
}
|
|
283
|
+
if not isinstance(header, dict) or set(header) != expected_fields:
|
|
284
|
+
raise HostedCrawlerError(
|
|
285
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
286
|
+
"crawler normalization header fields are not exact",
|
|
287
|
+
)
|
|
288
|
+
request = parse_crawler_request(header["request"])
|
|
289
|
+
coordinate = validate_crawler_request(
|
|
290
|
+
request,
|
|
291
|
+
expected_egress_policy_attestation=request.egress_policy_attestation,
|
|
292
|
+
)
|
|
293
|
+
if request.adapter_id == "public.https":
|
|
294
|
+
public = _public_https_query(request.query)
|
|
295
|
+
requested_url = public["url"]
|
|
296
|
+
route_valid = (
|
|
297
|
+
header["source_url"] == requested_url
|
|
298
|
+
and _canonical_url_host(header["final_url"]) == _canonical_url_host(requested_url)
|
|
299
|
+
and header["media_type"] in _public_expected_media_types(public)
|
|
300
|
+
and len(raw) <= public["limits"]["max_source_bytes"]
|
|
301
|
+
)
|
|
302
|
+
else:
|
|
303
|
+
league, season = coordinate
|
|
304
|
+
requested_url = f"https://api.openligadb.de/getmatchdata/{league}/{season}"
|
|
305
|
+
route_valid = (
|
|
306
|
+
header["source_url"] == requested_url
|
|
307
|
+
and header["final_url"] == requested_url
|
|
308
|
+
and header["media_type"] == "application/json"
|
|
309
|
+
# Defense in depth: the reviewed OpenLigaDB policy caps one response at 8 MiB,
|
|
310
|
+
# so its normalization frame never carries more, whatever the public ceiling is.
|
|
311
|
+
and len(raw) <= 8 * 1024 * 1024
|
|
312
|
+
)
|
|
313
|
+
if (
|
|
314
|
+
header["schema_version"] != NORMALIZATION_INPUT_SCHEMA
|
|
315
|
+
or not route_valid
|
|
316
|
+
or type(header["raw_size_bytes"]) is not int
|
|
317
|
+
or header["raw_size_bytes"] != len(raw)
|
|
318
|
+
or not 1 <= len(raw) <= MAX_RAW_RESPONSE_BYTES
|
|
319
|
+
or not isinstance(header["raw_content_sha256"], str)
|
|
320
|
+
or _DIGEST.fullmatch(header["raw_content_sha256"]) is None
|
|
321
|
+
or header["raw_content_sha256"] != sha256_bytes(raw)
|
|
322
|
+
or not isinstance(header["transport_evidence_digest"], str)
|
|
323
|
+
or _DIGEST.fullmatch(header["transport_evidence_digest"]) is None
|
|
324
|
+
or type(header["http_status_code"]) is not int
|
|
325
|
+
or not 200 <= header["http_status_code"] <= 299
|
|
326
|
+
or (
|
|
327
|
+
header["etag"] is not None
|
|
328
|
+
and (not isinstance(header["etag"], str) or len(header["etag"]) > 1_024)
|
|
329
|
+
)
|
|
330
|
+
or (
|
|
331
|
+
header["last_modified"] is not None
|
|
332
|
+
and (
|
|
333
|
+
not isinstance(header["last_modified"], str) or len(header["last_modified"]) > 1_024
|
|
334
|
+
)
|
|
335
|
+
)
|
|
336
|
+
):
|
|
337
|
+
raise HostedCrawlerError(
|
|
338
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
339
|
+
"crawler normalization evidence binding is invalid",
|
|
340
|
+
)
|
|
341
|
+
return (
|
|
342
|
+
request,
|
|
343
|
+
raw,
|
|
344
|
+
header["final_url"],
|
|
345
|
+
header["media_type"],
|
|
346
|
+
header["raw_content_sha256"],
|
|
347
|
+
header["transport_evidence_digest"],
|
|
348
|
+
header["http_status_code"],
|
|
349
|
+
header["etag"],
|
|
350
|
+
header["last_modified"],
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def normalize_crawler_response(
|
|
355
|
+
request: HostedCrawlerRequest,
|
|
356
|
+
*,
|
|
357
|
+
raw: bytes,
|
|
358
|
+
source_url: str,
|
|
359
|
+
media_type: str,
|
|
360
|
+
raw_digest: str,
|
|
361
|
+
transport_evidence_digest: str,
|
|
362
|
+
http_status_code: int,
|
|
363
|
+
etag: str | None,
|
|
364
|
+
last_modified: str | None,
|
|
365
|
+
) -> HostedCrawlerResult:
|
|
366
|
+
"""Strictly normalize one already-fetched response without network authority."""
|
|
367
|
+
|
|
368
|
+
coordinate = validate_crawler_request(
|
|
369
|
+
request,
|
|
370
|
+
expected_egress_policy_attestation=request.egress_policy_attestation,
|
|
371
|
+
)
|
|
372
|
+
if request.adapter_id == "public.https":
|
|
373
|
+
return _normalize_public_https_response(
|
|
374
|
+
request,
|
|
375
|
+
raw=raw,
|
|
376
|
+
final_url=source_url,
|
|
377
|
+
media_type=media_type,
|
|
378
|
+
raw_digest=raw_digest,
|
|
379
|
+
transport_evidence_digest=transport_evidence_digest,
|
|
380
|
+
http_status_code=http_status_code,
|
|
381
|
+
etag=etag,
|
|
382
|
+
last_modified=last_modified,
|
|
383
|
+
)
|
|
384
|
+
league, season = coordinate
|
|
385
|
+
requested_url = f"https://api.openligadb.de/getmatchdata/{league}/{season}"
|
|
386
|
+
if (
|
|
387
|
+
source_url != requested_url
|
|
388
|
+
or media_type != "application/json"
|
|
389
|
+
or raw_digest != sha256_bytes(raw)
|
|
390
|
+
):
|
|
391
|
+
raise HostedCrawlerError(
|
|
392
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
393
|
+
"crawler normalization evidence binding is invalid",
|
|
394
|
+
)
|
|
395
|
+
rows, historical_start, historical_end = _normalize(
|
|
396
|
+
raw,
|
|
397
|
+
expected_league=league,
|
|
398
|
+
expected_season=season,
|
|
399
|
+
)
|
|
400
|
+
normalized = _encode_csv(rows)
|
|
401
|
+
normalized_digest = sha256_bytes(normalized)
|
|
402
|
+
receipt = {
|
|
403
|
+
"schema_version": "hosted-openligadb-acquisition-receipt.v1",
|
|
404
|
+
"source_id": request.source_id,
|
|
405
|
+
"adapter_id": request.adapter_id,
|
|
406
|
+
"adapter_version": request.adapter_version,
|
|
407
|
+
"request_digest": request.digest,
|
|
408
|
+
"query": {
|
|
409
|
+
"league_shortcut": league,
|
|
410
|
+
"league_season": season,
|
|
411
|
+
},
|
|
412
|
+
"source_url": requested_url,
|
|
413
|
+
"final_url": source_url,
|
|
414
|
+
"media_type": media_type,
|
|
415
|
+
"acquired_at": request.requested_at,
|
|
416
|
+
"raw_content_sha256": raw_digest,
|
|
417
|
+
"raw_size_bytes": len(raw),
|
|
418
|
+
"normalized_content_sha256": normalized_digest,
|
|
419
|
+
"normalized_size_bytes": len(normalized),
|
|
420
|
+
"row_count": len(rows),
|
|
421
|
+
"column_names": list(_COLUMNS),
|
|
422
|
+
"transport_evidence_digest": transport_evidence_digest,
|
|
423
|
+
"egress_policy_attestation": request.egress_policy_attestation,
|
|
424
|
+
}
|
|
425
|
+
observation = {
|
|
426
|
+
"schema_version": "hosted-openligadb-source-observation.v1",
|
|
427
|
+
"source_id": request.source_id,
|
|
428
|
+
"adapter_id": request.adapter_id,
|
|
429
|
+
"adapter_version": request.adapter_version,
|
|
430
|
+
"request_digest": request.digest,
|
|
431
|
+
"observed_at": request.requested_at,
|
|
432
|
+
"available_at": request.requested_at,
|
|
433
|
+
"historical_start": historical_start,
|
|
434
|
+
"historical_end": historical_end,
|
|
435
|
+
"event_time_field": "match_datetime_utc",
|
|
436
|
+
"live_status": "live",
|
|
437
|
+
"raw_content_sha256": raw_digest,
|
|
438
|
+
"normalized_content_sha256": normalized_digest,
|
|
439
|
+
"row_count": len(rows),
|
|
440
|
+
}
|
|
441
|
+
return HostedCrawlerResult(
|
|
442
|
+
request_digest=request.digest,
|
|
443
|
+
source_id=request.source_id,
|
|
444
|
+
adapter_id=request.adapter_id,
|
|
445
|
+
adapter_version=request.adapter_version,
|
|
446
|
+
normalized_content=normalized,
|
|
447
|
+
acquisition_receipt=receipt,
|
|
448
|
+
observation=observation,
|
|
449
|
+
next_watermark=None,
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _canonical_url_host(value: Any) -> str:
|
|
454
|
+
if not isinstance(value, str) or not value:
|
|
455
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS URL is invalid")
|
|
456
|
+
try:
|
|
457
|
+
validate_source_locator(value, "https_url", "crawler.query.url")
|
|
458
|
+
parsed = urlsplit(value)
|
|
459
|
+
port = parsed.port
|
|
460
|
+
host = canonical_public_hostname(parsed)
|
|
461
|
+
except (ContractError, TypeError, ValueError) as error:
|
|
462
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS URL is invalid") from error
|
|
463
|
+
if parsed.scheme != "https" or parsed.username is not None or parsed.password is not None:
|
|
464
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS URL is invalid")
|
|
465
|
+
if port not in {None, 443} or parsed.fragment:
|
|
466
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS URL is invalid")
|
|
467
|
+
return host
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def _public_https_query(value: Any) -> dict[str, Any]:
|
|
471
|
+
if not isinstance(value, dict) or set(value) != _PUBLIC_QUERY_FIELDS:
|
|
472
|
+
raise HostedCrawlerError(
|
|
473
|
+
"CRAWLER_QUERY_INVALID",
|
|
474
|
+
"public HTTPS query fields are not exact",
|
|
475
|
+
)
|
|
476
|
+
url = value["url"]
|
|
477
|
+
if not isinstance(url, str) or not 1 <= len(url) <= 2_048:
|
|
478
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS URL is invalid")
|
|
479
|
+
_canonical_url_host(url)
|
|
480
|
+
data_format = value["data_format"]
|
|
481
|
+
filename = value["filename"]
|
|
482
|
+
if data_format not in DATA_FORMATS:
|
|
483
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS format is unsupported")
|
|
484
|
+
if (
|
|
485
|
+
not isinstance(filename, str)
|
|
486
|
+
or not 1 <= len(filename.encode("utf-8")) <= 255
|
|
487
|
+
or "/" in filename
|
|
488
|
+
or "\\" in filename
|
|
489
|
+
or any(ord(character) < 0x20 for character in filename)
|
|
490
|
+
):
|
|
491
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS filename is invalid")
|
|
492
|
+
limits = value["limits"]
|
|
493
|
+
if not isinstance(limits, dict) or set(limits) != _PUBLIC_LIMIT_FIELDS:
|
|
494
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS limits are not exact")
|
|
495
|
+
bounds = {
|
|
496
|
+
"max_source_bytes": (1, MAX_RAW_RESPONSE_BYTES),
|
|
497
|
+
"max_normalized_bytes": (1, MAX_NORMALIZED_SOURCE_BYTES),
|
|
498
|
+
"max_rows": (1, 10_000_000),
|
|
499
|
+
"max_columns": (1, 1_024),
|
|
500
|
+
}
|
|
501
|
+
for name, (minimum, maximum) in bounds.items():
|
|
502
|
+
item = limits[name]
|
|
503
|
+
if type(item) is not int or not minimum <= item <= maximum:
|
|
504
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", f"public HTTPS {name} is invalid")
|
|
505
|
+
reader = value["reader_pin"]
|
|
506
|
+
resource_caps = value["resource_caps"]
|
|
507
|
+
if resource_caps is not None:
|
|
508
|
+
if not isinstance(resource_caps, dict) or set(resource_caps) != _PUBLIC_RESOURCE_CAP_FIELDS:
|
|
509
|
+
raise HostedCrawlerError(
|
|
510
|
+
"CRAWLER_QUERY_INVALID", "public HTTPS Reader resource caps are not exact"
|
|
511
|
+
)
|
|
512
|
+
for name in sorted(_PUBLIC_RESOURCE_CAP_FIELDS):
|
|
513
|
+
item = resource_caps[name]
|
|
514
|
+
if type(item) is not int or item < 1:
|
|
515
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", f"public HTTPS {name} is invalid")
|
|
516
|
+
if resource_caps["max_nesting_depth"] != 1:
|
|
517
|
+
raise HostedCrawlerError(
|
|
518
|
+
"CRAWLER_QUERY_INVALID", "public HTTPS max_nesting_depth must be exactly 1"
|
|
519
|
+
)
|
|
520
|
+
if reader is not None:
|
|
521
|
+
if not isinstance(reader, dict) or set(reader) != {
|
|
522
|
+
"family_id",
|
|
523
|
+
"family_version",
|
|
524
|
+
"decode_options",
|
|
525
|
+
}:
|
|
526
|
+
raise HostedCrawlerError("CRAWLER_QUERY_INVALID", "public HTTPS Reader pin is invalid")
|
|
527
|
+
try:
|
|
528
|
+
pin = ReaderPin(reader["family_id"], reader["family_version"], reader["decode_options"])
|
|
529
|
+
family = TOOLBOX.resolve(pin.family_id, pin.family_version)
|
|
530
|
+
admitted = family.validate_options(pin.decode_options)
|
|
531
|
+
except (TypeError, ValueError) as error:
|
|
532
|
+
raise HostedCrawlerError(
|
|
533
|
+
getattr(error, "code", "CRAWLER_READER_INVALID"),
|
|
534
|
+
"public HTTPS Reader pin is not certified",
|
|
535
|
+
) from error
|
|
536
|
+
reader = {
|
|
537
|
+
"family_id": pin.family_id,
|
|
538
|
+
"family_version": pin.family_version,
|
|
539
|
+
"decode_options": dict(admitted),
|
|
540
|
+
}
|
|
541
|
+
elif resource_caps is not None:
|
|
542
|
+
raise HostedCrawlerError(
|
|
543
|
+
"CRAWLER_QUERY_INVALID", "public HTTPS resource caps require a Reader pin"
|
|
544
|
+
)
|
|
545
|
+
return {
|
|
546
|
+
"url": url,
|
|
547
|
+
"data_format": data_format,
|
|
548
|
+
"filename": filename,
|
|
549
|
+
"reader_pin": reader,
|
|
550
|
+
"resource_caps": None if resource_caps is None else dict(resource_caps),
|
|
551
|
+
"limits": dict(limits),
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _public_expected_media_types(query: dict[str, Any]) -> tuple[str, ...]:
|
|
556
|
+
reader = query["reader_pin"]
|
|
557
|
+
if reader is None:
|
|
558
|
+
from mostlyright.data_harness.formats import FORMAT_MEDIA_TYPES
|
|
559
|
+
|
|
560
|
+
return FORMAT_MEDIA_TYPES[query["data_format"]]
|
|
561
|
+
family = TOOLBOX.resolve(reader["family_id"], reader["family_version"])
|
|
562
|
+
return family.accepted_media_types
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def _normalize_public_https_response(
|
|
566
|
+
request: HostedCrawlerRequest,
|
|
567
|
+
*,
|
|
568
|
+
raw: bytes,
|
|
569
|
+
final_url: str,
|
|
570
|
+
media_type: str,
|
|
571
|
+
raw_digest: str,
|
|
572
|
+
transport_evidence_digest: str,
|
|
573
|
+
http_status_code: int,
|
|
574
|
+
etag: str | None,
|
|
575
|
+
last_modified: str | None,
|
|
576
|
+
) -> HostedCrawlerResult:
|
|
577
|
+
query = _public_https_query(request.query)
|
|
578
|
+
if (
|
|
579
|
+
raw_digest != sha256_bytes(raw)
|
|
580
|
+
or not 200 <= http_status_code <= 299
|
|
581
|
+
or _canonical_url_host(final_url) != _canonical_url_host(query["url"])
|
|
582
|
+
or media_type not in _public_expected_media_types(query)
|
|
583
|
+
):
|
|
584
|
+
raise HostedCrawlerError(
|
|
585
|
+
"CRAWLER_NORMALIZATION_INPUT_INVALID",
|
|
586
|
+
"public HTTPS retrieval binding is invalid",
|
|
587
|
+
)
|
|
588
|
+
reader = query["reader_pin"]
|
|
589
|
+
if reader is None:
|
|
590
|
+
normalized = raw
|
|
591
|
+
normalized_format = query["data_format"]
|
|
592
|
+
normalized_media_type = media_type
|
|
593
|
+
normalized_filename = query["filename"]
|
|
594
|
+
family_id = family_version = options_digest = None
|
|
595
|
+
decode_flags: list[str] = []
|
|
596
|
+
reader_budgets = None
|
|
597
|
+
else:
|
|
598
|
+
pin = ReaderPin(reader["family_id"], reader["family_version"], reader["decode_options"])
|
|
599
|
+
family = TOOLBOX.resolve(pin.family_id, pin.family_version)
|
|
600
|
+
budget_caps = {
|
|
601
|
+
"max_input_bytes": query["limits"]["max_source_bytes"],
|
|
602
|
+
"max_output_bytes": query["limits"]["max_normalized_bytes"],
|
|
603
|
+
"max_rows": query["limits"]["max_rows"],
|
|
604
|
+
"max_columns": query["limits"]["max_columns"],
|
|
605
|
+
**(query["resource_caps"] or {}),
|
|
606
|
+
}
|
|
607
|
+
budgets = family.default_budgets.narrowed_by(budget_caps)
|
|
608
|
+
reader_budgets = asdict(budgets)
|
|
609
|
+
decoded = family.decode(raw, pin, budgets)
|
|
610
|
+
normalized = decoded.content
|
|
611
|
+
normalized_format = decoded.data_format
|
|
612
|
+
normalized_media_type = decoded.media_type
|
|
613
|
+
normalized_filename = decoded.filename
|
|
614
|
+
family_id = pin.family_id
|
|
615
|
+
family_version = pin.family_version
|
|
616
|
+
options_digest = pin.options_digest
|
|
617
|
+
decode_flags = list(decoded.flags)
|
|
618
|
+
parse_limits = ParseLimits(
|
|
619
|
+
# The parsed bytes are the normalized decode output, so its budget is the
|
|
620
|
+
# normalized-output budget, not the fetched-byte budget.
|
|
621
|
+
max_input_bytes=query["limits"]["max_normalized_bytes"],
|
|
622
|
+
max_uncompressed_bytes=max(128 * 1024 * 1024, query["limits"]["max_normalized_bytes"]),
|
|
623
|
+
max_rows=query["limits"]["max_rows"],
|
|
624
|
+
max_columns=query["limits"]["max_columns"],
|
|
625
|
+
# The cell budget follows the query's own geometry rather than the ParseLimits
|
|
626
|
+
# default: an EPA year is ~9.4M rows x 24 columns and must pass when its stated
|
|
627
|
+
# row and column limits admit it.
|
|
628
|
+
max_total_cells=min(
|
|
629
|
+
1_000_000_000,
|
|
630
|
+
query["limits"]["max_rows"] * query["limits"]["max_columns"],
|
|
631
|
+
),
|
|
632
|
+
)
|
|
633
|
+
if normalized_format == "csv":
|
|
634
|
+
# The verification pass streams: a year-scale table re-checked as rows, never held
|
|
635
|
+
# as one. The evidence values are the drained parse's own, pinned equal by test.
|
|
636
|
+
evidence = parse_csv_evidence(
|
|
637
|
+
normalized,
|
|
638
|
+
media_type=normalized_media_type,
|
|
639
|
+
filename=normalized_filename,
|
|
640
|
+
limits=parse_limits,
|
|
641
|
+
)
|
|
642
|
+
parsed_columns = evidence.columns
|
|
643
|
+
parsed_row_count = evidence.row_count
|
|
644
|
+
parsed_schema_digest = evidence.schema_digest
|
|
645
|
+
row_digest = evidence.row_digest
|
|
646
|
+
else:
|
|
647
|
+
parsed = parse_tabular_bytes(
|
|
648
|
+
normalized,
|
|
649
|
+
data_format=normalized_format,
|
|
650
|
+
media_type=normalized_media_type,
|
|
651
|
+
filename=normalized_filename,
|
|
652
|
+
limits=parse_limits,
|
|
653
|
+
)
|
|
654
|
+
parsed_columns = parsed.columns
|
|
655
|
+
parsed_row_count = len(parsed.rows)
|
|
656
|
+
parsed_schema_digest = parsed.schema_digest
|
|
657
|
+
row_digest = row_digest_for(parsed)
|
|
658
|
+
normalized_digest = sha256_bytes(normalized)
|
|
659
|
+
etag_digest = None if etag is None else sha256_bytes(etag.encode("utf-8"))
|
|
660
|
+
last_modified_digest = (
|
|
661
|
+
None if last_modified is None else sha256_bytes(last_modified.encode("utf-8"))
|
|
662
|
+
)
|
|
663
|
+
enhanced_cadence_evidence = request.schema_version == CRAWLER_REQUEST_SCHEMA_V3
|
|
664
|
+
receipt = {
|
|
665
|
+
"schema_version": (
|
|
666
|
+
"hosted-public-https-acquisition-receipt.v3"
|
|
667
|
+
if enhanced_cadence_evidence
|
|
668
|
+
else "hosted-public-https-acquisition-receipt.v1"
|
|
669
|
+
),
|
|
670
|
+
"source_id": request.source_id,
|
|
671
|
+
"adapter_id": request.adapter_id,
|
|
672
|
+
"adapter_version": request.adapter_version,
|
|
673
|
+
"request_digest": request.digest,
|
|
674
|
+
"source_url": query["url"],
|
|
675
|
+
"final_url": final_url,
|
|
676
|
+
"fetched_media_type": media_type,
|
|
677
|
+
"normalized_media_type": normalized_media_type,
|
|
678
|
+
"normalized_data_format": normalized_format,
|
|
679
|
+
"normalized_filename": normalized_filename,
|
|
680
|
+
"acquired_at": request.requested_at,
|
|
681
|
+
"raw_content_sha256": raw_digest,
|
|
682
|
+
"raw_size_bytes": len(raw),
|
|
683
|
+
"normalized_content_sha256": normalized_digest,
|
|
684
|
+
"normalized_size_bytes": len(normalized),
|
|
685
|
+
"row_count": parsed_row_count,
|
|
686
|
+
"column_names": list(parsed_columns),
|
|
687
|
+
"parsed_schema_digest": parsed_schema_digest,
|
|
688
|
+
"family_id": family_id,
|
|
689
|
+
"family_version": family_version,
|
|
690
|
+
"decode_options_digest": options_digest,
|
|
691
|
+
"decode_flags": decode_flags,
|
|
692
|
+
"resource_caps": query["resource_caps"],
|
|
693
|
+
"reader_budgets": reader_budgets,
|
|
694
|
+
"transport_evidence_digest": transport_evidence_digest,
|
|
695
|
+
"egress_policy_attestation": request.egress_policy_attestation,
|
|
696
|
+
"clean_room": {
|
|
697
|
+
"host_platform": "linux-amd64",
|
|
698
|
+
"parser_network_syscalls": "denied",
|
|
699
|
+
"parser_process_creation": "denied",
|
|
700
|
+
"no_new_privs": True,
|
|
701
|
+
},
|
|
702
|
+
}
|
|
703
|
+
if enhanced_cadence_evidence:
|
|
704
|
+
receipt.update(
|
|
705
|
+
row_digest=row_digest,
|
|
706
|
+
http_status_code=http_status_code,
|
|
707
|
+
etag_digest=etag_digest,
|
|
708
|
+
last_modified_digest=last_modified_digest,
|
|
709
|
+
)
|
|
710
|
+
observation = {
|
|
711
|
+
"schema_version": (
|
|
712
|
+
"hosted-public-https-source-observation.v3"
|
|
713
|
+
if enhanced_cadence_evidence
|
|
714
|
+
else "hosted-public-https-source-observation.v1"
|
|
715
|
+
),
|
|
716
|
+
"source_id": request.source_id,
|
|
717
|
+
"request_digest": request.digest,
|
|
718
|
+
"observed_at": request.requested_at,
|
|
719
|
+
"raw_content_sha256": raw_digest,
|
|
720
|
+
"normalized_content_sha256": normalized_digest,
|
|
721
|
+
"row_count": parsed_row_count,
|
|
722
|
+
}
|
|
723
|
+
if enhanced_cadence_evidence:
|
|
724
|
+
observation.update(
|
|
725
|
+
row_digest=row_digest,
|
|
726
|
+
http_status_code=http_status_code,
|
|
727
|
+
etag_digest=etag_digest,
|
|
728
|
+
last_modified_digest=last_modified_digest,
|
|
729
|
+
)
|
|
730
|
+
return HostedCrawlerResult(
|
|
731
|
+
request_digest=request.digest,
|
|
732
|
+
source_id=request.source_id,
|
|
733
|
+
adapter_id=request.adapter_id,
|
|
734
|
+
adapter_version=request.adapter_version,
|
|
735
|
+
normalized_content=normalized,
|
|
736
|
+
acquisition_receipt=receipt,
|
|
737
|
+
observation=observation,
|
|
738
|
+
next_watermark=None,
|
|
739
|
+
schema_version=(
|
|
740
|
+
CRAWLER_RESULT_SCHEMA_V3 if enhanced_cadence_evidence else CRAWLER_RESULT_SCHEMA_V1
|
|
741
|
+
),
|
|
742
|
+
)
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
def warm_up_public_https_reader(request: HostedCrawlerRequest) -> None:
|
|
746
|
+
"""Re-open the packaged known sample on the exact hosted Reader execution path."""
|
|
747
|
+
|
|
748
|
+
query = _public_https_query(request.query)
|
|
749
|
+
reader = query["reader_pin"]
|
|
750
|
+
if reader is None:
|
|
751
|
+
return
|
|
752
|
+
pin = ReaderPin(reader["family_id"], reader["family_version"], reader["decode_options"])
|
|
753
|
+
|
|
754
|
+
def decode(sample: Any) -> DecodedFacts:
|
|
755
|
+
sample_family = TOOLBOX.resolve(sample.family_id, sample.family_version)
|
|
756
|
+
decoded = sample_family.decode(sample.content, sample.pin, sample_family.default_budgets)
|
|
757
|
+
parsed = parse_tabular_bytes(
|
|
758
|
+
decoded.content,
|
|
759
|
+
data_format=decoded.data_format,
|
|
760
|
+
media_type=decoded.media_type,
|
|
761
|
+
filename=decoded.filename,
|
|
762
|
+
limits=ParseLimits(
|
|
763
|
+
max_input_bytes=sample_family.default_budgets.max_output_bytes,
|
|
764
|
+
max_uncompressed_bytes=sample_family.default_budgets.max_output_bytes,
|
|
765
|
+
max_rows=sample_family.default_budgets.max_rows,
|
|
766
|
+
max_columns=sample_family.default_budgets.max_columns,
|
|
767
|
+
),
|
|
768
|
+
)
|
|
769
|
+
return DecodedFacts(
|
|
770
|
+
sha256_bytes(decoded.content),
|
|
771
|
+
len(parsed.rows),
|
|
772
|
+
tuple(parsed.columns),
|
|
773
|
+
tuple(decoded.flags),
|
|
774
|
+
)
|
|
775
|
+
|
|
776
|
+
warm_up(pin.family_id, pin.family_version, decode=decode)
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
def _query(value: Any) -> tuple[str, int]:
|
|
780
|
+
if not isinstance(value, dict) or set(value) != _QUERY_FIELDS:
|
|
781
|
+
raise HostedCrawlerError(
|
|
782
|
+
"CRAWLER_QUERY_INVALID",
|
|
783
|
+
OPENLIGADB_QUERY_FIELDS_DETAIL,
|
|
784
|
+
)
|
|
785
|
+
league = value["league_shortcut"]
|
|
786
|
+
season = value["league_season"]
|
|
787
|
+
if (
|
|
788
|
+
not isinstance(league, str)
|
|
789
|
+
or _LEAGUE.fullmatch(league) is None
|
|
790
|
+
or type(season) is not int
|
|
791
|
+
or not 2000 <= season <= 2100
|
|
792
|
+
):
|
|
793
|
+
raise HostedCrawlerError(
|
|
794
|
+
"CRAWLER_QUERY_INVALID",
|
|
795
|
+
OPENLIGADB_QUERY_ALLOWLIST_DETAIL,
|
|
796
|
+
)
|
|
797
|
+
return league, season
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _normalize(
|
|
801
|
+
raw: bytes,
|
|
802
|
+
*,
|
|
803
|
+
expected_league: str,
|
|
804
|
+
expected_season: int,
|
|
805
|
+
) -> tuple[list[dict[str, str]], str, str]:
|
|
806
|
+
def unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
807
|
+
result: dict[str, Any] = {}
|
|
808
|
+
for key, value in pairs:
|
|
809
|
+
if key in result:
|
|
810
|
+
raise ValueError("duplicate JSON object key")
|
|
811
|
+
result[key] = value
|
|
812
|
+
return result
|
|
813
|
+
|
|
814
|
+
try:
|
|
815
|
+
value = json.loads(
|
|
816
|
+
raw.decode("utf-8", errors="strict"),
|
|
817
|
+
object_pairs_hook=unique_object,
|
|
818
|
+
parse_constant=lambda _value: (_ for _ in ()).throw(ValueError("non-finite number")),
|
|
819
|
+
)
|
|
820
|
+
except (UnicodeError, ValueError, json.JSONDecodeError) as error:
|
|
821
|
+
raise HostedCrawlerError(
|
|
822
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
823
|
+
"OpenLigaDB returned invalid strict JSON",
|
|
824
|
+
) from error
|
|
825
|
+
if not isinstance(value, list) or not 1 <= len(value) <= MAX_ROWS:
|
|
826
|
+
raise HostedCrawlerError(
|
|
827
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
828
|
+
"OpenLigaDB response row inventory is empty or exceeds the bound",
|
|
829
|
+
)
|
|
830
|
+
rows = [_normalize_match(item) for item in value]
|
|
831
|
+
rows.sort(key=lambda item: int(item["match_id"]))
|
|
832
|
+
if len({item["match_id"] for item in rows}) != len(rows) or any(
|
|
833
|
+
item["league_shortcut"] != expected_league or item["league_season"] != str(expected_season)
|
|
834
|
+
for item in rows
|
|
835
|
+
):
|
|
836
|
+
raise HostedCrawlerError(
|
|
837
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
838
|
+
"OpenLigaDB response identity or league/season binding mismatches",
|
|
839
|
+
)
|
|
840
|
+
event_times = [item["match_datetime_utc"] for item in rows]
|
|
841
|
+
return rows, min(event_times), max(event_times)
|
|
842
|
+
|
|
843
|
+
|
|
844
|
+
def _normalize_match(value: Any) -> dict[str, str]:
|
|
845
|
+
if not isinstance(value, dict):
|
|
846
|
+
raise HostedCrawlerError(
|
|
847
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
848
|
+
"OpenLigaDB match rows must be objects",
|
|
849
|
+
)
|
|
850
|
+
match_id = _integer(value.get("matchID"), "matchID", minimum=1)
|
|
851
|
+
event_time = _utc(value.get("matchDateTimeUTC"), "matchDateTimeUTC")
|
|
852
|
+
last_update = _text(value.get("lastUpdateDateTime"), "lastUpdateDateTime")
|
|
853
|
+
league = _text(value.get("leagueShortcut"), "leagueShortcut")
|
|
854
|
+
season = _integer(value.get("leagueSeason"), "leagueSeason", minimum=2000)
|
|
855
|
+
group = _object(value.get("group"), "group")
|
|
856
|
+
team1 = _object(value.get("team1"), "team1")
|
|
857
|
+
team2 = _object(value.get("team2"), "team2")
|
|
858
|
+
finished = value.get("matchIsFinished")
|
|
859
|
+
if type(finished) is not bool:
|
|
860
|
+
raise HostedCrawlerError(
|
|
861
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
862
|
+
"OpenLigaDB matchIsFinished must be boolean",
|
|
863
|
+
)
|
|
864
|
+
score1, score2 = _score(value.get("matchResults"), finished=finished)
|
|
865
|
+
return {
|
|
866
|
+
"match_id": str(match_id),
|
|
867
|
+
"match_datetime_utc": event_time,
|
|
868
|
+
"last_update_datetime": last_update,
|
|
869
|
+
"league_shortcut": league,
|
|
870
|
+
"league_season": str(season),
|
|
871
|
+
"group_order_id": str(_integer(group.get("groupOrderID"), "group.groupOrderID", minimum=1)),
|
|
872
|
+
"team1_id": str(_integer(team1.get("teamId"), "team1.teamId", minimum=1)),
|
|
873
|
+
"team1_name": _text(team1.get("teamName"), "team1.teamName"),
|
|
874
|
+
"team2_id": str(_integer(team2.get("teamId"), "team2.teamId", minimum=1)),
|
|
875
|
+
"team2_name": _text(team2.get("teamName"), "team2.teamName"),
|
|
876
|
+
"match_finished": "true" if finished else "false",
|
|
877
|
+
"score1": "" if score1 is None else str(score1),
|
|
878
|
+
"score2": "" if score2 is None else str(score2),
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
|
|
882
|
+
def _score(value: Any, *, finished: bool) -> tuple[int | None, int | None]:
|
|
883
|
+
if not isinstance(value, list) or len(value) > 32:
|
|
884
|
+
raise HostedCrawlerError(
|
|
885
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
886
|
+
"OpenLigaDB matchResults must be a bounded array",
|
|
887
|
+
)
|
|
888
|
+
if not value:
|
|
889
|
+
if finished:
|
|
890
|
+
raise HostedCrawlerError(
|
|
891
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
892
|
+
"finished OpenLigaDB match has no final result",
|
|
893
|
+
)
|
|
894
|
+
return None, None
|
|
895
|
+
results = []
|
|
896
|
+
for item in value:
|
|
897
|
+
result = _object(item, "matchResults[]")
|
|
898
|
+
results.append(
|
|
899
|
+
(
|
|
900
|
+
_integer(
|
|
901
|
+
result.get("resultTypeID"),
|
|
902
|
+
"matchResults[].resultTypeID",
|
|
903
|
+
minimum=1,
|
|
904
|
+
),
|
|
905
|
+
_integer(
|
|
906
|
+
result.get("resultOrderID"),
|
|
907
|
+
"matchResults[].resultOrderID",
|
|
908
|
+
minimum=1,
|
|
909
|
+
),
|
|
910
|
+
_integer(
|
|
911
|
+
result.get("resultID"),
|
|
912
|
+
"matchResults[].resultID",
|
|
913
|
+
minimum=1,
|
|
914
|
+
),
|
|
915
|
+
_integer(
|
|
916
|
+
result.get("pointsTeam1"),
|
|
917
|
+
"matchResults[].pointsTeam1",
|
|
918
|
+
minimum=0,
|
|
919
|
+
),
|
|
920
|
+
_integer(
|
|
921
|
+
result.get("pointsTeam2"),
|
|
922
|
+
"matchResults[].pointsTeam2",
|
|
923
|
+
minimum=0,
|
|
924
|
+
),
|
|
925
|
+
)
|
|
926
|
+
)
|
|
927
|
+
finals = [item for item in results if item[0] == 2]
|
|
928
|
+
selected = max(finals or results, key=lambda item: (item[1], item[2]))
|
|
929
|
+
return selected[3], selected[4]
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
def _object(value: Any, field: str) -> dict[str, Any]:
|
|
933
|
+
if not isinstance(value, dict):
|
|
934
|
+
raise HostedCrawlerError(
|
|
935
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
936
|
+
f"OpenLigaDB {field} must be an object",
|
|
937
|
+
)
|
|
938
|
+
return value
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def _integer(value: Any, field: str, *, minimum: int) -> int:
|
|
942
|
+
if type(value) is not int or not minimum <= value <= MAX_SAFE_INTEGER:
|
|
943
|
+
raise HostedCrawlerError(
|
|
944
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
945
|
+
f"OpenLigaDB {field} is outside its integer bound",
|
|
946
|
+
)
|
|
947
|
+
return value
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def _text(value: Any, field: str) -> str:
|
|
951
|
+
if (
|
|
952
|
+
not isinstance(value, str)
|
|
953
|
+
or not 1 <= len(value.encode("utf-8")) <= MAX_TEXT_BYTES
|
|
954
|
+
or any(character in value for character in ("\x00", "\r", "\n"))
|
|
955
|
+
):
|
|
956
|
+
raise HostedCrawlerError(
|
|
957
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
958
|
+
f"OpenLigaDB {field} is invalid or exceeds its text bound",
|
|
959
|
+
)
|
|
960
|
+
return value
|
|
961
|
+
|
|
962
|
+
|
|
963
|
+
def _utc(value: Any, field: str) -> str:
|
|
964
|
+
if not isinstance(value, str) or _UTC_TIMESTAMP.fullmatch(value) is None:
|
|
965
|
+
raise HostedCrawlerError(
|
|
966
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
967
|
+
f"OpenLigaDB {field} must be an explicit UTC timestamp",
|
|
968
|
+
)
|
|
969
|
+
try:
|
|
970
|
+
datetime.fromisoformat(value.removesuffix("Z") + "+00:00")
|
|
971
|
+
except ValueError as error:
|
|
972
|
+
raise HostedCrawlerError(
|
|
973
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
974
|
+
f"OpenLigaDB {field} is not a real UTC timestamp",
|
|
975
|
+
) from error
|
|
976
|
+
return value
|
|
977
|
+
|
|
978
|
+
|
|
979
|
+
def _encode_csv(rows: list[dict[str, str]]) -> bytes:
|
|
980
|
+
stream = io.StringIO(newline="")
|
|
981
|
+
writer = csv.DictWriter(
|
|
982
|
+
stream,
|
|
983
|
+
fieldnames=list(_COLUMNS),
|
|
984
|
+
lineterminator="\n",
|
|
985
|
+
extrasaction="raise",
|
|
986
|
+
)
|
|
987
|
+
writer.writeheader()
|
|
988
|
+
writer.writerows(rows)
|
|
989
|
+
content = stream.getvalue().encode("utf-8")
|
|
990
|
+
# The reviewed OpenLigaDB bound, pinned to its own 8 MiB literal so the shared
|
|
991
|
+
# public-HTTPS ceiling cannot widen this fixed adapter.
|
|
992
|
+
if not 1 <= len(content) <= 8 * 1024 * 1024:
|
|
993
|
+
raise HostedCrawlerError(
|
|
994
|
+
"CRAWLER_RESPONSE_INVALID",
|
|
995
|
+
"normalized OpenLigaDB CSV exceeds the source byte bound",
|
|
996
|
+
)
|
|
997
|
+
return content
|
|
998
|
+
|
|
999
|
+
|
|
1000
|
+
def _read_fd_once(descriptor: int, *, maximum: int) -> bytes:
|
|
1001
|
+
chunks: list[bytes] = []
|
|
1002
|
+
total = 0
|
|
1003
|
+
while True:
|
|
1004
|
+
chunk = os.read(descriptor, min(65_536, maximum + 1 - total))
|
|
1005
|
+
if not chunk:
|
|
1006
|
+
break
|
|
1007
|
+
total += len(chunk)
|
|
1008
|
+
if total > maximum:
|
|
1009
|
+
raise HostedCrawlerError(
|
|
1010
|
+
"CRAWLER_PROTOCOL_LIMIT",
|
|
1011
|
+
"crawler request exceeds its byte bound",
|
|
1012
|
+
)
|
|
1013
|
+
chunks.append(chunk)
|
|
1014
|
+
os.close(descriptor)
|
|
1015
|
+
return b"".join(chunks)
|
|
1016
|
+
|
|
1017
|
+
|
|
1018
|
+
def _write_fd_once(descriptor: int, content: bytes) -> None:
|
|
1019
|
+
if len(content) > MAX_CRAWLER_RESULT_BYTES:
|
|
1020
|
+
raise HostedCrawlerError(
|
|
1021
|
+
"CRAWLER_PROTOCOL_LIMIT",
|
|
1022
|
+
"crawler result exceeds its byte bound",
|
|
1023
|
+
)
|
|
1024
|
+
view = memoryview(content)
|
|
1025
|
+
while view:
|
|
1026
|
+
written = os.write(descriptor, view)
|
|
1027
|
+
view = view[written:]
|
|
1028
|
+
os.close(descriptor)
|
|
1029
|
+
|
|
1030
|
+
|
|
1031
|
+
def _require_secret_minimized_environment() -> None:
|
|
1032
|
+
forbidden = sorted(key for key in os.environ if _FORBIDDEN_ENVIRONMENT.search(key))
|
|
1033
|
+
if forbidden:
|
|
1034
|
+
raise HostedCrawlerError(
|
|
1035
|
+
"CRAWLER_ENVIRONMENT_FORBIDDEN",
|
|
1036
|
+
"crawler environment contains a credential, Studio, token, or home variable",
|
|
1037
|
+
)
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
def crawler_main(argv: list[str] | None = None) -> int:
|
|
1041
|
+
parser = argparse.ArgumentParser(prog="mr-data-crawler")
|
|
1042
|
+
parser.add_argument("--input-fd", type=int, default=0)
|
|
1043
|
+
parser.add_argument("--result-fd", type=int, default=1)
|
|
1044
|
+
parser.add_argument("--warm-up", action="store_true")
|
|
1045
|
+
args = parser.parse_args(argv)
|
|
1046
|
+
try:
|
|
1047
|
+
_require_secret_minimized_environment()
|
|
1048
|
+
attestation = os.environ.get(CRAWLER_EGRESS_POLICY_ENV)
|
|
1049
|
+
if attestation is None:
|
|
1050
|
+
raise HostedCrawlerError(
|
|
1051
|
+
"CRAWLER_POLICY_REQUIRED",
|
|
1052
|
+
"crawler egress-policy attestation is unavailable",
|
|
1053
|
+
)
|
|
1054
|
+
maximum = MAX_CRAWLER_REQUEST_BYTES if args.warm_up else MAX_NORMALIZATION_INPUT_BYTES
|
|
1055
|
+
content = _read_fd_once(args.input_fd, maximum=maximum)
|
|
1056
|
+
if args.warm_up:
|
|
1057
|
+
request = parse_crawler_request(parse_canonical_json(content))
|
|
1058
|
+
validate_crawler_request(
|
|
1059
|
+
request,
|
|
1060
|
+
expected_egress_policy_attestation=attestation,
|
|
1061
|
+
)
|
|
1062
|
+
warm_up_public_https_reader(request)
|
|
1063
|
+
_write_fd_once(
|
|
1064
|
+
args.result_fd,
|
|
1065
|
+
canonical_json_bytes(
|
|
1066
|
+
{
|
|
1067
|
+
"schema_version": WARM_UP_RESULT_SCHEMA,
|
|
1068
|
+
"request_digest": request.digest,
|
|
1069
|
+
"status": "passed",
|
|
1070
|
+
}
|
|
1071
|
+
),
|
|
1072
|
+
)
|
|
1073
|
+
return 0
|
|
1074
|
+
(
|
|
1075
|
+
request,
|
|
1076
|
+
raw,
|
|
1077
|
+
source_url,
|
|
1078
|
+
media_type,
|
|
1079
|
+
raw_digest,
|
|
1080
|
+
transport_evidence_digest,
|
|
1081
|
+
http_status_code,
|
|
1082
|
+
etag,
|
|
1083
|
+
last_modified,
|
|
1084
|
+
) = parse_normalization_input(content)
|
|
1085
|
+
validate_crawler_request(
|
|
1086
|
+
request,
|
|
1087
|
+
expected_egress_policy_attestation=attestation,
|
|
1088
|
+
)
|
|
1089
|
+
result = normalize_crawler_response(
|
|
1090
|
+
request,
|
|
1091
|
+
raw=raw,
|
|
1092
|
+
source_url=source_url,
|
|
1093
|
+
media_type=media_type,
|
|
1094
|
+
raw_digest=raw_digest,
|
|
1095
|
+
transport_evidence_digest=transport_evidence_digest,
|
|
1096
|
+
http_status_code=http_status_code,
|
|
1097
|
+
etag=etag,
|
|
1098
|
+
last_modified=last_modified,
|
|
1099
|
+
)
|
|
1100
|
+
_write_fd_once(args.result_fd, canonical_json_bytes(result.to_dict()))
|
|
1101
|
+
return 0
|
|
1102
|
+
except (
|
|
1103
|
+
CanonicalJSONError,
|
|
1104
|
+
HostedCrawlerError,
|
|
1105
|
+
HostedCrawlerProtocolError,
|
|
1106
|
+
ValueError,
|
|
1107
|
+
TypeError,
|
|
1108
|
+
) as error:
|
|
1109
|
+
code = getattr(error, "code", "CRAWLER_FAILED")
|
|
1110
|
+
sys.stderr.write(canonical_json_bytes({"status": "failed", "code": code}).decode() + "\n")
|
|
1111
|
+
return 1
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
if __name__ == "__main__": # pragma: no cover - exercised through the executable boundary.
|
|
1115
|
+
raise SystemExit(crawler_main())
|