mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,701 @@
|
|
|
1
|
+
"""The closed Data.gov v4 metadata authoring policy.
|
|
2
|
+
|
|
3
|
+
This module is the only place where a normalized Data.gov v4 provider record becomes a v2 catalog
|
|
4
|
+
entry, and it is deliberately a table rather than a program. Every known fact it authors is a
|
|
5
|
+
value the provider literally declared at an allowlisted JSON path, carried together with the exact
|
|
6
|
+
record digest that stated it. Everything else is ``unknown``. There is no inference here: no
|
|
7
|
+
heuristic, no model, no title-derived identity, no format guessing, and no resource fetch. The
|
|
8
|
+
module imports no transport and is given bytes-derived dictionaries, never a retriever.
|
|
9
|
+
|
|
10
|
+
Three separations are load-bearing.
|
|
11
|
+
|
|
12
|
+
*Semantic fields versus observations.* A normalized record has a semantic
|
|
13
|
+
``fields`` partition and a volatile ``observations`` partition. ``last_harvested_date`` is a
|
|
14
|
+
catalog-ingestion timestamp that Data.gov rewrites on every re-harvest, and ``popularity`` is a
|
|
15
|
+
per-sweep counter; neither is a fact about the source. The policy reads ``fields`` only, re-derives
|
|
16
|
+
the record's own semantic digest before authoring anything from it, and refuses outright any record
|
|
17
|
+
whose semantic ``fields`` carry an observation path. Locator values -- landing pages, access and
|
|
18
|
+
download URLs, harvest-record links, organization logos -- do stay in ``fields`` as inert bounded
|
|
19
|
+
text; the policy simply never reads them, so no fact and no provenance can ever cite one.
|
|
20
|
+
|
|
21
|
+
*Facts versus layer text.* A fact keeps the complete declared value. The four retrieval layer
|
|
22
|
+
texts are a rendering of those facts and are capped at :data:`MAX_LAYER_TEXT_BYTES` UTF-8 bytes
|
|
23
|
+
with a Unicode-boundary-safe truncation plus an explicit per-entry disposition. The cap is what
|
|
24
|
+
keeps a 1,000-entry range/layer aggregate inside the 8 MiB ``encode_many`` bound downstream
|
|
25
|
+
(1000 * 8192 = 8,192,000 <= 8,388,608), so exactly one encode call per range/layer/backend stays
|
|
26
|
+
exact. The pinned MiniLM tokenizer truncates far below this cap, so retrieval is unaffected.
|
|
27
|
+
|
|
28
|
+
*Admission versus evidence.* Only an exact reviewed license mapping may state rights. Declared
|
|
29
|
+
rights prose, a federal publisher, or a public access level never grants one; ambiguity is flagged
|
|
30
|
+
and queued, never guessed.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from collections.abc import Mapping
|
|
36
|
+
from dataclasses import dataclass
|
|
37
|
+
from types import MappingProxyType
|
|
38
|
+
from typing import Any
|
|
39
|
+
|
|
40
|
+
from mostlyright.data_harness.canonical import canonical_sha256, sha256_bytes
|
|
41
|
+
from mostlyright.data_harness.sources.catalog.contracts import (
|
|
42
|
+
EMBEDDING_LAYERS,
|
|
43
|
+
MAX_DECLARED_NAME,
|
|
44
|
+
MAX_DECLARED_VOCABULARY,
|
|
45
|
+
MAX_DESCRIPTION,
|
|
46
|
+
MAX_SPATIAL_SCOPE,
|
|
47
|
+
)
|
|
48
|
+
from mostlyright.data_harness.sources.catalog.entry_v2 import (
|
|
49
|
+
CATALOG_ENTRY_V2_CONTRACT_VERSION,
|
|
50
|
+
PROVIDER_CONTRACTS,
|
|
51
|
+
CatalogEntryV2,
|
|
52
|
+
CatalogFact,
|
|
53
|
+
CatalogObservation,
|
|
54
|
+
ProviderFactProvenance,
|
|
55
|
+
ProviderRecordIdentity,
|
|
56
|
+
)
|
|
57
|
+
from mostlyright.data_harness.sources.catalog.harvest.datagov_v4 import (
|
|
58
|
+
DATAGOV_V4_RECORD_VERSION,
|
|
59
|
+
)
|
|
60
|
+
from mostlyright.data_harness.sources.catalog.harvest.protocol import (
|
|
61
|
+
DATAGOV_V4_HARVESTER_COORDINATE,
|
|
62
|
+
)
|
|
63
|
+
from mostlyright.data_harness.sources.contracts import (
|
|
64
|
+
DATA_FORMATS,
|
|
65
|
+
RIGHTS_STATUSES,
|
|
66
|
+
SourceContractError,
|
|
67
|
+
_CanonicalContract,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
AUTHORING_POLICY_SCHEMA = "mr-data-catalog-authoring-policy.v1"
|
|
71
|
+
DATAGOV_V4_POLICY_ID = "datagov-v4-policy.v1"
|
|
72
|
+
|
|
73
|
+
# One <=1000-entry range holds at most 1000 * 8192 = 8,192,000 bytes per layer, inside the
|
|
74
|
+
# 8,388,608-byte ``encode_many`` bound. Widening this constant breaks that equation.
|
|
75
|
+
MAX_LAYER_TEXT_BYTES = 8_192
|
|
76
|
+
|
|
77
|
+
# ``CatalogEntryV2`` bounds a known title at 200 characters. A longer or unsafe title cannot be
|
|
78
|
+
# carried, and truncating identity prose would author something the provider never declared, so
|
|
79
|
+
# such a record is skipped rather than reshaped.
|
|
80
|
+
MAX_TITLE = 200
|
|
81
|
+
|
|
82
|
+
LAYER_TEXT_UNKNOWN = "unknown"
|
|
83
|
+
LAYER_TEXT_NOT_APPLICABLE = "not_applicable"
|
|
84
|
+
|
|
85
|
+
AUTHORING_DISPOSITIONS = ("authored", "flagged", "skipped", "failed")
|
|
86
|
+
|
|
87
|
+
AUTHORING_REASON_CODES = frozenset(
|
|
88
|
+
{
|
|
89
|
+
"AUTHOR_OK",
|
|
90
|
+
"AUTHOR_RIGHTS_ABSENT",
|
|
91
|
+
"AUTHOR_RIGHTS_CONDITIONAL",
|
|
92
|
+
"AUTHOR_RIGHTS_PROHIBITED",
|
|
93
|
+
"AUTHOR_RIGHTS_PROSE",
|
|
94
|
+
"AUTHOR_RIGHTS_UNMAPPED",
|
|
95
|
+
"AUTHOR_TITLE_ABSENT",
|
|
96
|
+
"AUTHOR_TITLE_LIMIT",
|
|
97
|
+
"AUTHOR_TITLE_UNSAFE",
|
|
98
|
+
"AUTHOR_IDENTIFIER_UNSAFE",
|
|
99
|
+
"AUTHOR_RECORD_SCHEMA",
|
|
100
|
+
"AUTHOR_RECORD_DIGEST",
|
|
101
|
+
"AUTHOR_RECORD_FIELDS",
|
|
102
|
+
"AUTHOR_ENTRY_CONTRACT",
|
|
103
|
+
}
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
_RECORD_MEMBERS = frozenset(
|
|
107
|
+
{
|
|
108
|
+
"schema_version",
|
|
109
|
+
"protocol",
|
|
110
|
+
"record_id",
|
|
111
|
+
"fields",
|
|
112
|
+
"observations",
|
|
113
|
+
"raw_response_sha256",
|
|
114
|
+
"normalized_sha256",
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
_REQUIRED_RECORD_MEMBERS = _RECORD_MEMBERS - {"observations"}
|
|
118
|
+
|
|
119
|
+
# The exact provider paths this policy may read. Adding one is a reviewed source edit.
|
|
120
|
+
_SOURCE_JSON_PATHS = (
|
|
121
|
+
"$.dcat.distribution[*].format",
|
|
122
|
+
"$.dcat.distribution[*].mediaType",
|
|
123
|
+
"$.dcat.license",
|
|
124
|
+
"$.dcat.rights",
|
|
125
|
+
"$.dcat.spatial",
|
|
126
|
+
"$.dcat.spatial[*]",
|
|
127
|
+
"$.description",
|
|
128
|
+
"$.identifier",
|
|
129
|
+
"$.keyword[*]",
|
|
130
|
+
"$.publisher",
|
|
131
|
+
"$.theme[*]",
|
|
132
|
+
"$.title",
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
# Catalog-ingestion observations belong in a separate ``observations`` partition
|
|
136
|
+
# precisely because they are not facts about a source, so a semantic ``fields`` partition that
|
|
137
|
+
# carries one has been tampered with and the record is refused outright.
|
|
138
|
+
_OBSERVATION_JSON_PATHS = ("$.last_harvested_date", "$.popularity")
|
|
139
|
+
|
|
140
|
+
# Locator and catalog-ingestion paths this policy never reads. The raw evidence retains the
|
|
141
|
+
# locator values in raw and normalized evidence as inert bounded text; they are simply never facts,
|
|
142
|
+
# never provenance, and never fetched.
|
|
143
|
+
_REFUSED_JSON_PATHS = (
|
|
144
|
+
"$.dcat.describedBy",
|
|
145
|
+
"$.dcat.distribution[*].accessURL",
|
|
146
|
+
"$.dcat.distribution[*].describedBy",
|
|
147
|
+
"$.dcat.distribution[*].downloadURL",
|
|
148
|
+
"$.dcat.landingPage",
|
|
149
|
+
"$.dcat.references",
|
|
150
|
+
"$.harvest_record",
|
|
151
|
+
"$.harvest_record_raw",
|
|
152
|
+
"$.harvest_record_transformed",
|
|
153
|
+
"$.last_harvested_date",
|
|
154
|
+
"$.organization.logo",
|
|
155
|
+
"$.popularity",
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
# Exact declared license values only. Public-domain dedications carry no obligation and may be
|
|
159
|
+
# ``approved``; attribution licenses are ``conditional`` and therefore still flagged, because v2
|
|
160
|
+
# has no obligation fact that could carry the attribution requirement.
|
|
161
|
+
_RIGHTS_MAP = (
|
|
162
|
+
("http://creativecommons.org/licenses/by/4.0/", "conditional"),
|
|
163
|
+
("http://creativecommons.org/publicdomain/zero/1.0/", "approved"),
|
|
164
|
+
("http://www.usa.gov/publicdomain/label/1.0/", "approved"),
|
|
165
|
+
("https://creativecommons.org/licenses/by/4.0/", "conditional"),
|
|
166
|
+
("https://creativecommons.org/publicdomain/zero/1.0/", "approved"),
|
|
167
|
+
("https://www.usa.gov/publicdomain/label/1.0/", "approved"),
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
# Exact declared distribution ``format``/``mediaType`` spellings that name one wire encoding the
|
|
171
|
+
# harness can actually read. Anything else -- HTML, PDF, ZIP, API, XML, GeoJSON -- is deliberately
|
|
172
|
+
# unmapped: an unrecognized format is not a format we may claim.
|
|
173
|
+
_FORMAT_MAP = (
|
|
174
|
+
("CSV", "csv"),
|
|
175
|
+
("JSON", "json"),
|
|
176
|
+
("NDJSON", "ndjson"),
|
|
177
|
+
("PARQUET", "parquet"),
|
|
178
|
+
("application/json", "json"),
|
|
179
|
+
("application/vnd.apache.parquet", "parquet"),
|
|
180
|
+
("application/x-ndjson", "ndjson"),
|
|
181
|
+
("csv", "csv"),
|
|
182
|
+
("json", "json"),
|
|
183
|
+
("ndjson", "ndjson"),
|
|
184
|
+
("parquet", "parquet"),
|
|
185
|
+
("text/csv", "csv"),
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
class AuthoringPolicyRefused(SourceContractError):
|
|
190
|
+
"""A stable refusal of an authoring policy coordinate or contract."""
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@dataclass(frozen=True)
|
|
194
|
+
class AuthoringPolicy(_CanonicalContract):
|
|
195
|
+
"""One reviewed, allowlisted, digest-bound provider authoring policy."""
|
|
196
|
+
|
|
197
|
+
policy_id: str
|
|
198
|
+
provider_id: str
|
|
199
|
+
harvester_coordinate: str
|
|
200
|
+
record_schema_version: str
|
|
201
|
+
entry_schema_version: str
|
|
202
|
+
entry_id_prefix: str
|
|
203
|
+
identifier_json_path: str
|
|
204
|
+
source_json_paths: tuple[str, ...]
|
|
205
|
+
refused_json_paths: tuple[str, ...]
|
|
206
|
+
observation_json_paths: tuple[str, ...]
|
|
207
|
+
rights_map: tuple[tuple[str, str], ...]
|
|
208
|
+
format_map: tuple[tuple[str, str], ...]
|
|
209
|
+
max_layer_text_bytes: int
|
|
210
|
+
schema_version: str = AUTHORING_POLICY_SCHEMA
|
|
211
|
+
|
|
212
|
+
def __post_init__(self) -> None:
|
|
213
|
+
contract = PROVIDER_CONTRACTS.get(self.provider_id)
|
|
214
|
+
if (
|
|
215
|
+
self.schema_version != AUTHORING_POLICY_SCHEMA
|
|
216
|
+
or contract is None
|
|
217
|
+
or self.harvester_coordinate != contract.harvester_coordinate
|
|
218
|
+
or self.identifier_json_path != contract.identifier_json_path
|
|
219
|
+
):
|
|
220
|
+
raise AuthoringPolicyRefused(
|
|
221
|
+
"AUTHOR_POLICY_CONTRACT",
|
|
222
|
+
"policy.provider_id",
|
|
223
|
+
"policy must bind one registered provider contract",
|
|
224
|
+
)
|
|
225
|
+
source = set(self.source_json_paths)
|
|
226
|
+
refused = set(self.refused_json_paths)
|
|
227
|
+
if (
|
|
228
|
+
self.source_json_paths != tuple(sorted(source))
|
|
229
|
+
or self.refused_json_paths != tuple(sorted(refused))
|
|
230
|
+
or self.observation_json_paths != tuple(sorted(set(self.observation_json_paths)))
|
|
231
|
+
or not source
|
|
232
|
+
or not refused
|
|
233
|
+
or not self.observation_json_paths
|
|
234
|
+
or not source.isdisjoint(refused)
|
|
235
|
+
or not set(self.observation_json_paths) <= refused
|
|
236
|
+
or self.identifier_json_path not in source
|
|
237
|
+
):
|
|
238
|
+
raise AuthoringPolicyRefused(
|
|
239
|
+
"AUTHOR_POLICY_CONTRACT",
|
|
240
|
+
"policy.source_json_paths",
|
|
241
|
+
"source and refused paths must be sorted, unique, and disjoint",
|
|
242
|
+
)
|
|
243
|
+
if self.rights_map != tuple(sorted(set(self.rights_map))) or any(
|
|
244
|
+
status not in RIGHTS_STATUSES for _value, status in self.rights_map
|
|
245
|
+
):
|
|
246
|
+
raise AuthoringPolicyRefused(
|
|
247
|
+
"AUTHOR_POLICY_CONTRACT",
|
|
248
|
+
"policy.rights_map",
|
|
249
|
+
"rights mappings must be sorted, unique, and closed",
|
|
250
|
+
)
|
|
251
|
+
if self.format_map != tuple(sorted(set(self.format_map))) or any(
|
|
252
|
+
data_format not in DATA_FORMATS for _value, data_format in self.format_map
|
|
253
|
+
):
|
|
254
|
+
raise AuthoringPolicyRefused(
|
|
255
|
+
"AUTHOR_POLICY_CONTRACT",
|
|
256
|
+
"policy.format_map",
|
|
257
|
+
"format mappings must be sorted, unique, and closed",
|
|
258
|
+
)
|
|
259
|
+
if type(self.max_layer_text_bytes) is not int or not 0 < self.max_layer_text_bytes <= (
|
|
260
|
+
MAX_LAYER_TEXT_BYTES
|
|
261
|
+
):
|
|
262
|
+
raise AuthoringPolicyRefused(
|
|
263
|
+
"AUTHOR_POLICY_CONTRACT",
|
|
264
|
+
"policy.max_layer_text_bytes",
|
|
265
|
+
"layer text cap must stay inside the shared encode bound",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
def to_dict(self) -> dict[str, Any]:
|
|
269
|
+
return {
|
|
270
|
+
"schema_version": self.schema_version,
|
|
271
|
+
"policy_id": self.policy_id,
|
|
272
|
+
"provider_id": self.provider_id,
|
|
273
|
+
"harvester_coordinate": self.harvester_coordinate,
|
|
274
|
+
"record_schema_version": self.record_schema_version,
|
|
275
|
+
"entry_schema_version": self.entry_schema_version,
|
|
276
|
+
"entry_id_prefix": self.entry_id_prefix,
|
|
277
|
+
"identifier_json_path": self.identifier_json_path,
|
|
278
|
+
"source_json_paths": list(self.source_json_paths),
|
|
279
|
+
"refused_json_paths": list(self.refused_json_paths),
|
|
280
|
+
"observation_json_paths": list(self.observation_json_paths),
|
|
281
|
+
"rights_map": [
|
|
282
|
+
{"declared": value, "status": status} for value, status in self.rights_map
|
|
283
|
+
],
|
|
284
|
+
"format_map": [
|
|
285
|
+
{"declared": value, "data_format": data_format}
|
|
286
|
+
for value, data_format in self.format_map
|
|
287
|
+
],
|
|
288
|
+
"max_layer_text_bytes": self.max_layer_text_bytes,
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
def rights_for(self, declared: str) -> str | None:
|
|
292
|
+
return dict(self.rights_map).get(declared)
|
|
293
|
+
|
|
294
|
+
def data_format_for(self, declared: str) -> str | None:
|
|
295
|
+
return dict(self.format_map).get(declared)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
DATAGOV_V4_AUTHORING_POLICY = AuthoringPolicy(
|
|
299
|
+
policy_id=DATAGOV_V4_POLICY_ID,
|
|
300
|
+
provider_id="datagov_v4",
|
|
301
|
+
harvester_coordinate=DATAGOV_V4_HARVESTER_COORDINATE,
|
|
302
|
+
record_schema_version=DATAGOV_V4_RECORD_VERSION,
|
|
303
|
+
entry_schema_version=CATALOG_ENTRY_V2_CONTRACT_VERSION,
|
|
304
|
+
entry_id_prefix="datagov_v4.",
|
|
305
|
+
identifier_json_path="$.identifier",
|
|
306
|
+
source_json_paths=_SOURCE_JSON_PATHS,
|
|
307
|
+
refused_json_paths=_REFUSED_JSON_PATHS,
|
|
308
|
+
observation_json_paths=_OBSERVATION_JSON_PATHS,
|
|
309
|
+
rights_map=_RIGHTS_MAP,
|
|
310
|
+
format_map=_FORMAT_MAP,
|
|
311
|
+
max_layer_text_bytes=MAX_LAYER_TEXT_BYTES,
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
AUTHORING_POLICIES: Mapping[str, AuthoringPolicy] = MappingProxyType(
|
|
315
|
+
{DATAGOV_V4_POLICY_ID: DATAGOV_V4_AUTHORING_POLICY}
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def resolve_authoring_policy(policy_id: Any) -> AuthoringPolicy:
|
|
320
|
+
"""Return the exact reviewed policy for one coordinate; never infer a near match."""
|
|
321
|
+
|
|
322
|
+
policy = AUTHORING_POLICIES.get(policy_id) if isinstance(policy_id, str) else None
|
|
323
|
+
if policy is None:
|
|
324
|
+
raise AuthoringPolicyRefused(
|
|
325
|
+
"AUTHOR_POLICY_UNKNOWN",
|
|
326
|
+
"policy_id",
|
|
327
|
+
"policy coordinate is not one reviewed allowlisted policy",
|
|
328
|
+
)
|
|
329
|
+
return policy
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
@dataclass(frozen=True)
|
|
333
|
+
class AuthoredRecord:
|
|
334
|
+
"""One record's exact authoring outcome: an entry or an explicit refusal, never a guess."""
|
|
335
|
+
|
|
336
|
+
provider_record_id: str
|
|
337
|
+
disposition: str
|
|
338
|
+
reason_code: str
|
|
339
|
+
entry: CatalogEntryV2 | None
|
|
340
|
+
layer_texts: tuple[tuple[str, str], ...]
|
|
341
|
+
truncated_layers: tuple[str, ...]
|
|
342
|
+
|
|
343
|
+
@property
|
|
344
|
+
def entry_id(self) -> str | None:
|
|
345
|
+
return None if self.entry is None else self.entry.entry_id
|
|
346
|
+
|
|
347
|
+
def to_dict(self) -> dict[str, Any]:
|
|
348
|
+
return {
|
|
349
|
+
"provider_record_id": self.provider_record_id,
|
|
350
|
+
"disposition": self.disposition,
|
|
351
|
+
"reason_code": self.reason_code,
|
|
352
|
+
"entry": None if self.entry is None else self.entry.to_dict(),
|
|
353
|
+
"layer_texts": [{"layer": layer, "text": text} for layer, text in self.layer_texts],
|
|
354
|
+
"layer_text_truncated": list(self.truncated_layers),
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def author_catalog_record(
|
|
359
|
+
record: Mapping[str, Any],
|
|
360
|
+
*,
|
|
361
|
+
policy: AuthoringPolicy,
|
|
362
|
+
observed_at: str,
|
|
363
|
+
page_evidence_sha256: str,
|
|
364
|
+
) -> AuthoredRecord:
|
|
365
|
+
"""Map one normalized provider record onto one honest v2 entry and its four layer texts."""
|
|
366
|
+
|
|
367
|
+
envelope = _record_envelope(record, policy)
|
|
368
|
+
if envelope is not None:
|
|
369
|
+
return _refusal(record, "failed", envelope)
|
|
370
|
+
record_id = record["record_id"]
|
|
371
|
+
if canonical_sha256(_semantic_payload(record)) != record["normalized_sha256"]:
|
|
372
|
+
return _refusal(record, "failed", "AUTHOR_RECORD_DIGEST")
|
|
373
|
+
observations = {_base_path(path) for path in policy.observation_json_paths}
|
|
374
|
+
values: dict[str, Any] = {}
|
|
375
|
+
for field in record["fields"]:
|
|
376
|
+
if _base_path(field["path"]) in observations:
|
|
377
|
+
return _refusal(record, "failed", "AUTHOR_RECORD_FIELDS")
|
|
378
|
+
# A ``typed_json_ast.v1`` value carries a provider number too large for exact JSON. No fact
|
|
379
|
+
# this policy authors is numeric, so such a value simply declares nothing here.
|
|
380
|
+
if field.get("encoding", "canonical_json") == "canonical_json":
|
|
381
|
+
values[field["path"]] = field["value"]
|
|
382
|
+
|
|
383
|
+
if values.get(policy.identifier_json_path) != record_id or not _safe_text(
|
|
384
|
+
record_id, maximum=512
|
|
385
|
+
):
|
|
386
|
+
return _refusal(record, "skipped", "AUTHOR_IDENTIFIER_UNSAFE")
|
|
387
|
+
|
|
388
|
+
title = values.get("$.title")
|
|
389
|
+
if not isinstance(title, str) or not title:
|
|
390
|
+
return _refusal(record, "skipped", "AUTHOR_TITLE_ABSENT")
|
|
391
|
+
if not _safe_text(title, maximum=len(title)):
|
|
392
|
+
return _refusal(record, "skipped", "AUTHOR_TITLE_UNSAFE")
|
|
393
|
+
if len(title) > MAX_TITLE:
|
|
394
|
+
return _refusal(record, "skipped", "AUTHOR_TITLE_LIMIT")
|
|
395
|
+
|
|
396
|
+
provenance = _provenance_factory(policy, record)
|
|
397
|
+
rights_fact, rights_reason = _rights(values, policy, provenance)
|
|
398
|
+
facts = {
|
|
399
|
+
"title": _known_text_fact(title, ("$.title",), provenance),
|
|
400
|
+
"publisher": _text_fact(values, "$.publisher", maximum=200, provenance=provenance),
|
|
401
|
+
"description": _text_fact(
|
|
402
|
+
values, "$.description", maximum=MAX_DESCRIPTION, provenance=provenance
|
|
403
|
+
),
|
|
404
|
+
"spatial_scope": _text_tuple_fact(
|
|
405
|
+
values,
|
|
406
|
+
("$.dcat.spatial", "$.dcat.spatial[*]"),
|
|
407
|
+
maximum=MAX_SPATIAL_SCOPE,
|
|
408
|
+
provenance=provenance,
|
|
409
|
+
),
|
|
410
|
+
"data_formats": _data_formats(values, policy, provenance),
|
|
411
|
+
"declared_vocabulary": _text_tuple_fact(
|
|
412
|
+
values,
|
|
413
|
+
("$.keyword[*]", "$.theme[*]"),
|
|
414
|
+
maximum=MAX_DECLARED_VOCABULARY,
|
|
415
|
+
provenance=provenance,
|
|
416
|
+
),
|
|
417
|
+
"rights": rights_fact,
|
|
418
|
+
# No v4 field asserts any of these, so a closed policy states exactly that. ``accessLevel``
|
|
419
|
+
# is a disclosure classification, not one of the harness access kinds; the provider
|
|
420
|
+
# declares no column list, row count, profile, or dataset authentication requirement.
|
|
421
|
+
"access_kind": CatalogFact(state="unknown"),
|
|
422
|
+
"authentication_required": CatalogFact(state="unknown"),
|
|
423
|
+
"declared_columns": CatalogFact(state="unknown"),
|
|
424
|
+
"declared_row_count": CatalogFact(state="unknown"),
|
|
425
|
+
"profiles": CatalogFact(state="unknown"),
|
|
426
|
+
}
|
|
427
|
+
try:
|
|
428
|
+
entry = CatalogEntryV2(
|
|
429
|
+
entry_id=f"{policy.entry_id_prefix}{sha256_bytes(record_id.encode('utf-8'))}",
|
|
430
|
+
entry_version=1,
|
|
431
|
+
provider_record=ProviderRecordIdentity(
|
|
432
|
+
provider_id=policy.provider_id,
|
|
433
|
+
provider_record_id=record_id,
|
|
434
|
+
provider_record_sha256=record["normalized_sha256"],
|
|
435
|
+
identifier_json_path=policy.identifier_json_path,
|
|
436
|
+
),
|
|
437
|
+
observations=(
|
|
438
|
+
CatalogObservation(
|
|
439
|
+
harvester_coordinate=policy.harvester_coordinate,
|
|
440
|
+
observed_at=observed_at,
|
|
441
|
+
response_evidence_sha256=record["raw_response_sha256"],
|
|
442
|
+
page_evidence_sha256=page_evidence_sha256,
|
|
443
|
+
),
|
|
444
|
+
),
|
|
445
|
+
**facts,
|
|
446
|
+
)
|
|
447
|
+
except SourceContractError:
|
|
448
|
+
return _refusal(record, "failed", "AUTHOR_ENTRY_CONTRACT")
|
|
449
|
+
|
|
450
|
+
texts, truncated = _layer_texts(entry, policy)
|
|
451
|
+
admitted = rights_fact.state == "known" and rights_fact.value == "approved"
|
|
452
|
+
return AuthoredRecord(
|
|
453
|
+
provider_record_id=record_id,
|
|
454
|
+
disposition="authored" if admitted else "flagged",
|
|
455
|
+
reason_code=rights_reason,
|
|
456
|
+
entry=entry,
|
|
457
|
+
layer_texts=texts,
|
|
458
|
+
truncated_layers=truncated,
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _record_envelope(record: Mapping[str, Any], policy: AuthoringPolicy) -> str | None:
|
|
463
|
+
if not isinstance(record, Mapping) or not _REQUIRED_RECORD_MEMBERS <= set(record) <= (
|
|
464
|
+
_RECORD_MEMBERS
|
|
465
|
+
):
|
|
466
|
+
return "AUTHOR_RECORD_SCHEMA"
|
|
467
|
+
if (
|
|
468
|
+
record["schema_version"] != policy.record_schema_version
|
|
469
|
+
or record["protocol"] != policy.provider_id
|
|
470
|
+
or not isinstance(record["record_id"], str)
|
|
471
|
+
or not record["record_id"]
|
|
472
|
+
or not isinstance(record["fields"], list)
|
|
473
|
+
or not record["fields"]
|
|
474
|
+
or not _is_digest(record["raw_response_sha256"])
|
|
475
|
+
or not _is_digest(record["normalized_sha256"])
|
|
476
|
+
or not isinstance(record.get("observations", []), list)
|
|
477
|
+
):
|
|
478
|
+
return "AUTHOR_RECORD_SCHEMA"
|
|
479
|
+
for field in record["fields"]:
|
|
480
|
+
if (
|
|
481
|
+
not isinstance(field, Mapping)
|
|
482
|
+
or not {"path", "value"} <= set(field) <= {"path", "value", "encoding"}
|
|
483
|
+
or not isinstance(field["path"], str)
|
|
484
|
+
):
|
|
485
|
+
return "AUTHOR_RECORD_SCHEMA"
|
|
486
|
+
return None
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _semantic_payload(record: Mapping[str, Any]) -> dict[str, Any]:
|
|
490
|
+
return {
|
|
491
|
+
"schema_version": record["schema_version"],
|
|
492
|
+
"protocol": record["protocol"],
|
|
493
|
+
"record_id": record["record_id"],
|
|
494
|
+
"fields": record["fields"],
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _refusal(record: Mapping[str, Any], disposition: str, reason_code: str) -> AuthoredRecord:
|
|
499
|
+
identifier = record.get("record_id") if isinstance(record, Mapping) else None
|
|
500
|
+
return AuthoredRecord(
|
|
501
|
+
provider_record_id=identifier if isinstance(identifier, str) else "",
|
|
502
|
+
disposition=disposition,
|
|
503
|
+
reason_code=reason_code,
|
|
504
|
+
entry=None,
|
|
505
|
+
layer_texts=(),
|
|
506
|
+
truncated_layers=(),
|
|
507
|
+
)
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def _provenance_factory(policy: AuthoringPolicy, record: Mapping[str, Any]):
|
|
511
|
+
def build(paths: tuple[str, ...]) -> ProviderFactProvenance:
|
|
512
|
+
return ProviderFactProvenance(
|
|
513
|
+
provider_id=policy.provider_id,
|
|
514
|
+
provider_record_id=record["record_id"],
|
|
515
|
+
provider_record_sha256=record["normalized_sha256"],
|
|
516
|
+
json_paths=tuple(sorted(set(paths))),
|
|
517
|
+
)
|
|
518
|
+
|
|
519
|
+
return build
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _known_text_fact(value: str, paths: tuple[str, ...], provenance) -> CatalogFact:
|
|
523
|
+
return CatalogFact(state="known", value=value, provenance=provenance(paths))
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _text_fact(values: Mapping[str, Any], path: str, *, maximum: int, provenance) -> CatalogFact:
|
|
527
|
+
value = values.get(path)
|
|
528
|
+
if not isinstance(value, str) or not _safe_text(value, maximum=maximum):
|
|
529
|
+
return CatalogFact(state="unknown")
|
|
530
|
+
return _known_text_fact(value, (path,), provenance)
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _text_tuple_fact(
|
|
534
|
+
values: Mapping[str, Any], paths: tuple[str, ...], *, maximum: int, provenance
|
|
535
|
+
) -> CatalogFact:
|
|
536
|
+
"""Admit exactly the declared values that satisfy the contract; an empty subset is unknown."""
|
|
537
|
+
|
|
538
|
+
admitted: set[str] = set()
|
|
539
|
+
cited: list[str] = []
|
|
540
|
+
for path in paths:
|
|
541
|
+
declared = _declared_strings(values.get(path))
|
|
542
|
+
safe = {item for item in declared if _safe_text(item, maximum=MAX_DECLARED_NAME)}
|
|
543
|
+
if safe:
|
|
544
|
+
admitted |= safe
|
|
545
|
+
cited.append(path)
|
|
546
|
+
if not admitted or len(admitted) > maximum:
|
|
547
|
+
return CatalogFact(state="unknown")
|
|
548
|
+
return CatalogFact(
|
|
549
|
+
state="known", value=tuple(sorted(admitted)), provenance=provenance(tuple(cited))
|
|
550
|
+
)
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _data_formats(values: Mapping[str, Any], policy: AuthoringPolicy, provenance) -> CatalogFact:
|
|
554
|
+
admitted: set[str] = set()
|
|
555
|
+
cited: list[str] = []
|
|
556
|
+
for path in ("$.dcat.distribution[*].format", "$.dcat.distribution[*].mediaType"):
|
|
557
|
+
mapped = {
|
|
558
|
+
policy.data_format_for(item)
|
|
559
|
+
for item in _declared_strings(values.get(path))
|
|
560
|
+
if policy.data_format_for(item) is not None
|
|
561
|
+
}
|
|
562
|
+
if mapped:
|
|
563
|
+
admitted |= {value for value in mapped if value is not None}
|
|
564
|
+
cited.append(path)
|
|
565
|
+
if not admitted:
|
|
566
|
+
return CatalogFact(state="unknown")
|
|
567
|
+
return CatalogFact(
|
|
568
|
+
state="known", value=tuple(sorted(admitted)), provenance=provenance(tuple(cited))
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _rights(
|
|
573
|
+
values: Mapping[str, Any], policy: AuthoringPolicy, provenance
|
|
574
|
+
) -> tuple[CatalogFact, str]:
|
|
575
|
+
"""Map an exact declared license, or flag. Declared prose never authorizes anything."""
|
|
576
|
+
|
|
577
|
+
prose = values.get("$.dcat.rights")
|
|
578
|
+
if isinstance(prose, str) and prose.strip():
|
|
579
|
+
return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_PROSE"
|
|
580
|
+
declared = values.get("$.dcat.license")
|
|
581
|
+
if not isinstance(declared, str) or not declared:
|
|
582
|
+
return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_ABSENT"
|
|
583
|
+
status = policy.rights_for(declared)
|
|
584
|
+
if status is None:
|
|
585
|
+
return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_UNMAPPED"
|
|
586
|
+
fact = CatalogFact(state="known", value=status, provenance=provenance(("$.dcat.license",)))
|
|
587
|
+
reason = {
|
|
588
|
+
"approved": "AUTHOR_OK",
|
|
589
|
+
"conditional": "AUTHOR_RIGHTS_CONDITIONAL",
|
|
590
|
+
"prohibited": "AUTHOR_RIGHTS_PROHIBITED",
|
|
591
|
+
"unclear": "AUTHOR_RIGHTS_UNMAPPED",
|
|
592
|
+
}[status]
|
|
593
|
+
return fact, reason
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
def _layer_texts(
|
|
597
|
+
entry: CatalogEntryV2, policy: AuthoringPolicy
|
|
598
|
+
) -> tuple[tuple[tuple[str, str], ...], tuple[str, ...]]:
|
|
599
|
+
"""Render exactly four layer texts, using canonical state tokens instead of invented prose."""
|
|
600
|
+
|
|
601
|
+
rendered = {
|
|
602
|
+
"description": f"{entry.title.value}\n{_one(entry.description)}",
|
|
603
|
+
"metadata": " ".join(
|
|
604
|
+
(
|
|
605
|
+
_one(entry.publisher),
|
|
606
|
+
*_many(entry.spatial_scope),
|
|
607
|
+
*_many(entry.data_formats),
|
|
608
|
+
_one(entry.access_kind),
|
|
609
|
+
_one(entry.rights),
|
|
610
|
+
)
|
|
611
|
+
),
|
|
612
|
+
"columns": " ".join((*_many(entry.declared_columns), *_many(entry.declared_vocabulary))),
|
|
613
|
+
"profiles": " ".join(
|
|
614
|
+
(
|
|
615
|
+
"rows",
|
|
616
|
+
_one(entry.declared_row_count),
|
|
617
|
+
"authentication",
|
|
618
|
+
_one(entry.authentication_required),
|
|
619
|
+
"profiles",
|
|
620
|
+
*_many(entry.profiles),
|
|
621
|
+
"formats",
|
|
622
|
+
*_many(entry.data_formats),
|
|
623
|
+
)
|
|
624
|
+
),
|
|
625
|
+
}
|
|
626
|
+
texts: list[tuple[str, str]] = []
|
|
627
|
+
truncated: list[str] = []
|
|
628
|
+
for layer in EMBEDDING_LAYERS:
|
|
629
|
+
text, was_truncated = _truncate_utf8(rendered[layer], policy.max_layer_text_bytes)
|
|
630
|
+
texts.append((layer, text))
|
|
631
|
+
if was_truncated:
|
|
632
|
+
truncated.append(layer)
|
|
633
|
+
return tuple(texts), tuple(truncated)
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
def _one(fact: CatalogFact) -> str:
|
|
637
|
+
if fact.state == "not_applicable":
|
|
638
|
+
return LAYER_TEXT_NOT_APPLICABLE
|
|
639
|
+
if fact.state != "known":
|
|
640
|
+
return LAYER_TEXT_UNKNOWN
|
|
641
|
+
if type(fact.value) is bool:
|
|
642
|
+
return "true" if fact.value else "false"
|
|
643
|
+
return str(fact.value)
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
def _many(fact: CatalogFact) -> tuple[str, ...]:
|
|
647
|
+
if fact.state == "not_applicable":
|
|
648
|
+
return (LAYER_TEXT_NOT_APPLICABLE,)
|
|
649
|
+
if fact.state != "known" or not isinstance(fact.value, tuple):
|
|
650
|
+
return (LAYER_TEXT_UNKNOWN,)
|
|
651
|
+
return fact.value
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def _truncate_utf8(value: str, maximum: int) -> tuple[str, bool]:
|
|
655
|
+
"""Cut to at most ``maximum`` UTF-8 bytes without ever splitting a code point."""
|
|
656
|
+
|
|
657
|
+
encoded = value.encode("utf-8")
|
|
658
|
+
if len(encoded) <= maximum:
|
|
659
|
+
return value, False
|
|
660
|
+
cut = encoded[:maximum]
|
|
661
|
+
while cut:
|
|
662
|
+
try:
|
|
663
|
+
return cut.decode("utf-8"), True
|
|
664
|
+
except UnicodeDecodeError:
|
|
665
|
+
cut = cut[:-1]
|
|
666
|
+
return "", True
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _declared_strings(value: Any) -> tuple[str, ...]:
|
|
670
|
+
"""Read a declared scalar or array of strings; every other shape declares nothing."""
|
|
671
|
+
|
|
672
|
+
if isinstance(value, str):
|
|
673
|
+
return (value,)
|
|
674
|
+
if isinstance(value, list):
|
|
675
|
+
return tuple(item for item in value if isinstance(item, str))
|
|
676
|
+
return ()
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
def _safe_text(value: str, *, maximum: int) -> bool:
|
|
680
|
+
if not value or len(value) > maximum:
|
|
681
|
+
return False
|
|
682
|
+
for character in value:
|
|
683
|
+
codepoint = ord(character)
|
|
684
|
+
if (
|
|
685
|
+
0xD800 <= codepoint <= 0xDFFF
|
|
686
|
+
or codepoint < 0x20
|
|
687
|
+
or 0x7F <= codepoint <= 0x9F
|
|
688
|
+
or (character.isspace() and character != " ")
|
|
689
|
+
):
|
|
690
|
+
return False
|
|
691
|
+
return True
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def _base_path(path: str) -> str:
|
|
695
|
+
return path[:-3] if path.endswith("[*]") else path
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
def _is_digest(value: Any) -> bool:
|
|
699
|
+
return (
|
|
700
|
+
isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
|
|
701
|
+
)
|