mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1428 @@
|
|
|
1
|
+
"""Canonical empirical source observations and deterministic cadence inference.
|
|
2
|
+
|
|
3
|
+
The existing :class:`SourceObservation` records researched/catalogue fitness. This module owns a
|
|
4
|
+
different fact: what one governed connector actually observed over repeated headless probes. The
|
|
5
|
+
two contracts deliberately have different names and schema coordinates.
|
|
6
|
+
|
|
7
|
+
Inference consumes only the exact digest-bound ordered history passed by the caller. This module
|
|
8
|
+
does not grant that history authority; the future Studio consumer must authenticate who recorded
|
|
9
|
+
each observation before persisting it. The pure calculation never reads a clock, network,
|
|
10
|
+
environment variable, mutable file, or model service, so every consumer can replay the same
|
|
11
|
+
history and compare its result with the packaged golden vectors.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
import unicodedata
|
|
18
|
+
from collections.abc import Mapping, Sequence
|
|
19
|
+
from copy import deepcopy
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from datetime import UTC, datetime, timedelta
|
|
22
|
+
from hashlib import sha256
|
|
23
|
+
from itertools import pairwise
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from mostlyright.data_harness import canonical as canonical_contract
|
|
28
|
+
from mostlyright.data_harness.canonical import canonical_sha256
|
|
29
|
+
|
|
30
|
+
SOURCE_CADENCE_OBSERVATION_SCHEMA = "harness-empirical-source-observation.v1"
|
|
31
|
+
SOURCE_CADENCE_INFERENCE_SCHEMA = "harness-source-cadence-inference.v1"
|
|
32
|
+
SOURCE_CADENCE_ALGORITHM_VERSION = "source-cadence.v1"
|
|
33
|
+
_IMPLEMENTATION_SOURCE_SHA256 = sha256(Path(__file__).read_bytes()).hexdigest()
|
|
34
|
+
_CANONICAL_SOURCE_SHA256 = sha256(Path(canonical_contract.__file__).read_bytes()).hexdigest()
|
|
35
|
+
_FAILURE_CODE_VALUES = (
|
|
36
|
+
"contract_refused",
|
|
37
|
+
"http_4xx",
|
|
38
|
+
"http_5xx",
|
|
39
|
+
"rate_limited",
|
|
40
|
+
"reader_failure",
|
|
41
|
+
"source_unavailable",
|
|
42
|
+
"timeout",
|
|
43
|
+
"tls_failure",
|
|
44
|
+
"transport_failure",
|
|
45
|
+
"unknown_failure",
|
|
46
|
+
)
|
|
47
|
+
_SOURCE_CADENCE_ALGORITHM_SPEC: dict[str, object] = {
|
|
48
|
+
"version": SOURCE_CADENCE_ALGORITHM_VERSION,
|
|
49
|
+
"implementation_source_sha256": _IMPLEMENTATION_SOURCE_SHA256,
|
|
50
|
+
"canonicalization_source_sha256": _CANONICAL_SOURCE_SHA256,
|
|
51
|
+
"integer_arithmetic": {
|
|
52
|
+
"time_delta_seconds": "exact_utc_timestamp_difference_as_integer_seconds",
|
|
53
|
+
"per_edition_interval": "delta_seconds_floor_divided_by_positive_edition_delta",
|
|
54
|
+
"basis_points": "integer_multiply_then_floor_divide_by_10000",
|
|
55
|
+
"median": "sorted_middle_or_floor_of_two_middle_sum_divided_by_two",
|
|
56
|
+
},
|
|
57
|
+
"history": {
|
|
58
|
+
"maximum_observations": 4_096,
|
|
59
|
+
"maximum_partitions_per_observation": 128,
|
|
60
|
+
"maximum_partitions_per_history": 8_192,
|
|
61
|
+
"identity": [
|
|
62
|
+
"source",
|
|
63
|
+
"epoch",
|
|
64
|
+
"connector_configuration",
|
|
65
|
+
"locator",
|
|
66
|
+
"recipe",
|
|
67
|
+
"source_authority",
|
|
68
|
+
"reader_coordinate",
|
|
69
|
+
"reader_options",
|
|
70
|
+
"frontier_policy",
|
|
71
|
+
"time_interpretation",
|
|
72
|
+
"algorithm",
|
|
73
|
+
"acquired_frontier_kind_and_derivation",
|
|
74
|
+
"publication_coordinate_presence",
|
|
75
|
+
],
|
|
76
|
+
"ordering": "contiguous_sequence_predecessor_digest_nonoverlapping_probe_time",
|
|
77
|
+
"material_order": "strict_publication_reference_and_edition_sequence_when_present",
|
|
78
|
+
},
|
|
79
|
+
"result_shapes": {
|
|
80
|
+
"changed": "complete_acquired_evidence_and_optional_complete_publication_coordinate",
|
|
81
|
+
"unchanged_acquired": "complete_acquired_evidence_equal_to_prior_acquired_facts",
|
|
82
|
+
"unchanged_no_body": "http_304_conditional_request_and_probe_receipt_no_parsed_facts",
|
|
83
|
+
"not_published": "probe_receipt_and_status_none_or_200_202_204_404_no_data_facts",
|
|
84
|
+
"failed": {
|
|
85
|
+
"shape": "closed_failure_code_and_probe_receipt_no_data_facts",
|
|
86
|
+
"failure_codes": list(_FAILURE_CODE_VALUES),
|
|
87
|
+
},
|
|
88
|
+
},
|
|
89
|
+
"interval": {
|
|
90
|
+
"source": "publication_reference_time_divided_by_edition_sequence_delta",
|
|
91
|
+
"minimum_changed_observations": 4,
|
|
92
|
+
"median": "sorted_integer_midpoint_floor",
|
|
93
|
+
"jitter": "maximum_absolute_distance_from_median_interval",
|
|
94
|
+
"regular_jitter_floor_seconds": 3_600,
|
|
95
|
+
"regular_jitter_basis_points": 2_000,
|
|
96
|
+
"regime_minimum_intervals": 5,
|
|
97
|
+
"regime_recent_intervals": 2,
|
|
98
|
+
"regime_change_basis_points": 4_000,
|
|
99
|
+
"regime_recent_consistency_basis_points": 2_000,
|
|
100
|
+
},
|
|
101
|
+
"dormancy": {
|
|
102
|
+
"elapsed_interval_multiplier": 4,
|
|
103
|
+
"minimum_successful_no_change_observations": 3,
|
|
104
|
+
"failures_disqualify_trailing_evidence": True,
|
|
105
|
+
},
|
|
106
|
+
"expected_window": {
|
|
107
|
+
"center": "last_publication_reference_plus_interval",
|
|
108
|
+
"minimum_margin_seconds": 300,
|
|
109
|
+
"interval_margin_basis_points": 500,
|
|
110
|
+
"include_only_if_latest_observation_not_after_window": True,
|
|
111
|
+
},
|
|
112
|
+
"confidence": {
|
|
113
|
+
"learning_per_change_basis_points": 1_000,
|
|
114
|
+
"learning_maximum_basis_points": 4_000,
|
|
115
|
+
"stable_base_basis_points": 6_000,
|
|
116
|
+
"stable_extra_interval_basis_points": 500,
|
|
117
|
+
"stable_maximum_basis_points": 9_500,
|
|
118
|
+
"dormant_base_basis_points": 5_000,
|
|
119
|
+
"dormant_per_interval_basis_points": 300,
|
|
120
|
+
"dormant_maximum_basis_points": 8_000,
|
|
121
|
+
"changed_pattern_basis_points": 2_500,
|
|
122
|
+
"failure_penalty_each_basis_points": 500,
|
|
123
|
+
"failure_penalty_maximum_basis_points": 3_000,
|
|
124
|
+
},
|
|
125
|
+
"probe_recommendation": {
|
|
126
|
+
"stable_divisor": 8,
|
|
127
|
+
"stable_maximum_seconds": 21_600,
|
|
128
|
+
"dormant_minimum_seconds": 86_400,
|
|
129
|
+
"dormant_maximum_seconds": 604_800,
|
|
130
|
+
"unstable_divisor": 4,
|
|
131
|
+
"unstable_minimum_seconds": 900,
|
|
132
|
+
"unstable_maximum_seconds": 86_400,
|
|
133
|
+
"unknown_seconds": 21_600,
|
|
134
|
+
"epoch_reset_maximum_seconds": 3_600,
|
|
135
|
+
"absolute_minimum_seconds": 300,
|
|
136
|
+
"absolute_maximum_seconds": 2_592_000,
|
|
137
|
+
},
|
|
138
|
+
"revision": {
|
|
139
|
+
"historical_revision": "changed_content_without_frontier_advance",
|
|
140
|
+
"replacement": "row_count_falls_or_prior_partition_missing_or_rewritten",
|
|
141
|
+
"append": "all_prior_partitions_unchanged_and_at_least_one_new_partition",
|
|
142
|
+
"fallback": "unknown",
|
|
143
|
+
},
|
|
144
|
+
"publication_delay": "observation_completion_minus_publication_reference_integer_seconds",
|
|
145
|
+
"evaluation_time": "last_observation_completed_at",
|
|
146
|
+
"state_precedence": [
|
|
147
|
+
"learning_until_four_coordinate_complete_material_changes",
|
|
148
|
+
"stable_if_jitter_within_threshold_else_changed_pattern_irregular",
|
|
149
|
+
"regime_change_overrides_regular_or_irregular_to_changed_pattern_and_epoch_reset",
|
|
150
|
+
"dormancy_overrides_stable_only_with_trailing_successful_no_change_evidence",
|
|
151
|
+
"missed_expected_window_overrides_remaining_stable_to_changed_pattern_regular",
|
|
152
|
+
"failure_penalty_and_reason_apply_after_state_selection",
|
|
153
|
+
],
|
|
154
|
+
"output_formulas": {
|
|
155
|
+
"interval_jitter_seconds": "max_abs_each_interval_minus_median",
|
|
156
|
+
"publication_delay_summary": "minimum_floor_median_maximum",
|
|
157
|
+
"expected_window_margin": "max_300_interval_floor_div_20_jitter",
|
|
158
|
+
"confidence": "state_base_plus_interval_increment_then_failure_penalty_clamped_at_zero",
|
|
159
|
+
"evidence_range": "first_started_at_through_last_completed_at",
|
|
160
|
+
"history_digest": "canonical_sha256_of_exact_observation_documents_in_order",
|
|
161
|
+
"reason_order": "pattern_then_regime_then_dormancy_or_missed_window_then_failures",
|
|
162
|
+
"revision_precedence": (
|
|
163
|
+
"historical_revision_then_replacement_then_all_transitions_append_else_unknown"
|
|
164
|
+
),
|
|
165
|
+
},
|
|
166
|
+
"reason_codes": [
|
|
167
|
+
"no_observations",
|
|
168
|
+
"missing_publication_coordinates",
|
|
169
|
+
"insufficient_changes",
|
|
170
|
+
"regular_intervals",
|
|
171
|
+
"irregular_intervals",
|
|
172
|
+
"cadence_regime_changed",
|
|
173
|
+
"publication_window_missed",
|
|
174
|
+
"prolonged_no_change",
|
|
175
|
+
"probe_failures_observed",
|
|
176
|
+
],
|
|
177
|
+
}
|
|
178
|
+
SOURCE_CADENCE_ALGORITHM_DIGEST = canonical_sha256(_SOURCE_CADENCE_ALGORITHM_SPEC)
|
|
179
|
+
|
|
180
|
+
_DIGEST = re.compile(r"^[0-9a-f]{64}$")
|
|
181
|
+
_IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$")
|
|
182
|
+
_READER_IDENTIFIER = re.compile(r"^[a-z][a-z0-9]*(?:[._-][a-z0-9]+)*$")
|
|
183
|
+
_SEMVER = re.compile(r"^[0-9]+\.[0-9]+\.[0-9]+$")
|
|
184
|
+
_TIMESTAMP = re.compile(
|
|
185
|
+
r"^(?P<date>[0-9]{4}-[0-9]{2}-[0-9]{2})T"
|
|
186
|
+
r"(?P<time>[0-9]{2}:[0-9]{2}:[0-9]{2})(?P<fraction>\.[0-9]{1,6})?Z$"
|
|
187
|
+
)
|
|
188
|
+
_HTTP_STATUS_CLASSES = frozenset({"none", "2xx", "3xx", "4xx", "5xx"})
|
|
189
|
+
_RESULT_KINDS = frozenset({"changed", "unchanged", "not_published", "failed"})
|
|
190
|
+
_FRONTIER_KINDS = frozenset({"none", "date", "timestamp", "integer", "opaque"})
|
|
191
|
+
_INFERENCE_STATES = frozenset({"unknown", "learning", "stable", "changed_pattern", "dormant"})
|
|
192
|
+
_CADENCE_KINDS = frozenset({"unknown", "regular", "irregular"})
|
|
193
|
+
_REVISION_STYLES = frozenset({"unknown", "append", "replacement", "historical_revision"})
|
|
194
|
+
_FAILURE_CODES = frozenset(_FAILURE_CODE_VALUES)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class CadenceContractError(ValueError):
|
|
198
|
+
"""Typed refusal raised by observation, history, or inference validation."""
|
|
199
|
+
|
|
200
|
+
def __init__(self, code: str, path: str, detail: str) -> None:
|
|
201
|
+
self.code = code
|
|
202
|
+
self.path = path
|
|
203
|
+
self.detail = detail
|
|
204
|
+
super().__init__(f"{path}: {detail} [{code}]")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def source_cadence_algorithm_spec() -> dict[str, object]:
|
|
208
|
+
"""Return an isolated copy of the complete normative algorithm manifest."""
|
|
209
|
+
|
|
210
|
+
return deepcopy(_SOURCE_CADENCE_ALGORITHM_SPEC)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _refuse(code: str, path: str, detail: str) -> None:
|
|
214
|
+
raise CadenceContractError(code, path, detail)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _digest(value: object, path: str, *, optional: bool = False) -> None:
|
|
218
|
+
if value is None and optional:
|
|
219
|
+
return
|
|
220
|
+
if not isinstance(value, str) or _DIGEST.fullmatch(value) is None:
|
|
221
|
+
_refuse("DIGEST", path, "must be a lowercase SHA-256 digest")
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _identifier(value: object, path: str) -> None:
|
|
225
|
+
if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) is None:
|
|
226
|
+
_refuse("IDENTIFIER", path, "must be a bounded source identifier")
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _reader_identifier(value: object, path: str) -> None:
|
|
230
|
+
if not isinstance(value, str) or len(value) > 64 or _READER_IDENTIFIER.fullmatch(value) is None:
|
|
231
|
+
_refuse("IDENTIFIER", path, "must be a canonical Reader identifier")
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _timestamp(value: object, path: str) -> datetime:
|
|
235
|
+
if not isinstance(value, str) or _TIMESTAMP.fullmatch(value) is None:
|
|
236
|
+
_refuse("TIMESTAMP", path, "must be a canonical UTC timestamp ending in Z")
|
|
237
|
+
try:
|
|
238
|
+
parsed = datetime.fromisoformat(value[:-1] + "+00:00")
|
|
239
|
+
except ValueError:
|
|
240
|
+
_refuse("TIMESTAMP", path, "must be a real calendar timestamp")
|
|
241
|
+
parsed = parsed.astimezone(UTC)
|
|
242
|
+
if _format_timestamp(parsed) != value:
|
|
243
|
+
_refuse("TIMESTAMP", path, "must use the one canonical UTC spelling")
|
|
244
|
+
return parsed
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _format_timestamp(value: datetime) -> str:
|
|
248
|
+
value = value.astimezone(UTC)
|
|
249
|
+
if value.microsecond:
|
|
250
|
+
return value.isoformat(timespec="microseconds").replace("+00:00", "Z")
|
|
251
|
+
return value.isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _bounded_text(value: object, path: str, *, maximum: int) -> None:
|
|
255
|
+
if not isinstance(value, str) or not 1 <= len(value) <= maximum or value != value.strip():
|
|
256
|
+
_refuse("TEXT", path, f"must be stripped text of 1..{maximum} characters")
|
|
257
|
+
if any(
|
|
258
|
+
not character.isprintable() or unicodedata.category(character) in {"Zl", "Zp"}
|
|
259
|
+
for character in value
|
|
260
|
+
):
|
|
261
|
+
_refuse("TEXT", path, "must contain only display-safe characters")
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _exact(value: Mapping[str, Any], expected: set[str], path: str) -> None:
|
|
265
|
+
if set(value) != expected:
|
|
266
|
+
_refuse("EXACT_KEYS", path, "does not have the exact versioned key set")
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _optional_int(value: object, path: str, *, minimum: int, maximum: int) -> None:
|
|
270
|
+
if value is None:
|
|
271
|
+
return
|
|
272
|
+
if type(value) is not int or not minimum <= value <= maximum:
|
|
273
|
+
_refuse("INTEGER", path, f"must be null or an integer in {minimum}..{maximum}")
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _safe_shift(value: datetime, seconds: int, path: str) -> datetime:
|
|
277
|
+
try:
|
|
278
|
+
return value + timedelta(seconds=seconds)
|
|
279
|
+
except OverflowError:
|
|
280
|
+
_refuse("TIMESTAMP_RANGE", path, "cannot represent the inferred time")
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _elapsed_whole_seconds(earlier: datetime, later: datetime) -> int:
|
|
284
|
+
"""Return exact floor whole seconds without floating-point conversion."""
|
|
285
|
+
|
|
286
|
+
delta = later - earlier
|
|
287
|
+
return delta.days * 86_400 + delta.seconds
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
@dataclass(frozen=True)
|
|
291
|
+
class SourceFrontier:
|
|
292
|
+
"""One approved, typed source frontier and its public derivation rule."""
|
|
293
|
+
|
|
294
|
+
kind: str
|
|
295
|
+
value: str | None
|
|
296
|
+
derivation_rule: str
|
|
297
|
+
|
|
298
|
+
def __post_init__(self) -> None:
|
|
299
|
+
if not isinstance(self.kind, str) or self.kind not in _FRONTIER_KINDS:
|
|
300
|
+
_refuse("FRONTIER_KIND", "frontier.kind", "is not a supported frontier kind")
|
|
301
|
+
_bounded_text(self.derivation_rule, "frontier.derivation_rule", maximum=128)
|
|
302
|
+
if self.kind == "none":
|
|
303
|
+
if self.value is not None:
|
|
304
|
+
_refuse("FRONTIER_VALUE", "frontier.value", "must be null for kind none")
|
|
305
|
+
return
|
|
306
|
+
_bounded_text(self.value, "frontier.value", maximum=256)
|
|
307
|
+
if self.kind == "date":
|
|
308
|
+
try:
|
|
309
|
+
parsed_date = datetime.strptime(self.value, "%Y-%m-%d")
|
|
310
|
+
except ValueError:
|
|
311
|
+
_refuse("FRONTIER_VALUE", "frontier.value", "must be a real ISO date")
|
|
312
|
+
if parsed_date.strftime("%Y-%m-%d") != self.value:
|
|
313
|
+
_refuse("FRONTIER_VALUE", "frontier.value", "must use canonical ISO spelling")
|
|
314
|
+
elif self.kind == "timestamp":
|
|
315
|
+
_timestamp(self.value, "frontier.value")
|
|
316
|
+
elif self.kind == "integer":
|
|
317
|
+
if not re.fullmatch(r"0|[1-9][0-9]{0,18}", self.value):
|
|
318
|
+
_refuse("FRONTIER_VALUE", "frontier.value", "must be a canonical unsigned integer")
|
|
319
|
+
|
|
320
|
+
def to_dict(self) -> dict[str, Any]:
|
|
321
|
+
return {"kind": self.kind, "value": self.value, "derivation_rule": self.derivation_rule}
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
@dataclass(frozen=True)
|
|
325
|
+
class PartitionDigest:
|
|
326
|
+
"""Credential-free identity and content digest for one approved source partition."""
|
|
327
|
+
|
|
328
|
+
partition_key_digest: str
|
|
329
|
+
content_digest: str
|
|
330
|
+
|
|
331
|
+
def __post_init__(self) -> None:
|
|
332
|
+
_digest(self.partition_key_digest, "partition.partition_key_digest")
|
|
333
|
+
_digest(self.content_digest, "partition.content_digest")
|
|
334
|
+
|
|
335
|
+
def to_dict(self) -> dict[str, str]:
|
|
336
|
+
return {
|
|
337
|
+
"partition_key_digest": self.partition_key_digest,
|
|
338
|
+
"content_digest": self.content_digest,
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
@dataclass(frozen=True)
|
|
343
|
+
class EmpiricalSourceObservation:
|
|
344
|
+
"""Immutable result of one bounded connector probe or acquisition."""
|
|
345
|
+
|
|
346
|
+
source_id: str
|
|
347
|
+
sequence: int
|
|
348
|
+
inference_epoch: str
|
|
349
|
+
previous_observation_digest: str | None
|
|
350
|
+
connector_configuration_digest: str
|
|
351
|
+
locator_digest: str
|
|
352
|
+
recipe_digest: str
|
|
353
|
+
source_authority_digest: str
|
|
354
|
+
reader_family_id: str
|
|
355
|
+
reader_family_version: str
|
|
356
|
+
reader_options_digest: str
|
|
357
|
+
frontier_policy_digest: str
|
|
358
|
+
time_interpretation_digest: str
|
|
359
|
+
inference_algorithm_digest: str
|
|
360
|
+
started_at: str
|
|
361
|
+
completed_at: str
|
|
362
|
+
result_kind: str
|
|
363
|
+
failure_code: str | None
|
|
364
|
+
http_status_class: str
|
|
365
|
+
http_status_code: int | None
|
|
366
|
+
conditional_request_digest: str | None
|
|
367
|
+
probe_receipt_digest: str
|
|
368
|
+
etag_digest: str | None
|
|
369
|
+
last_modified_digest: str | None
|
|
370
|
+
acquired_size_bytes: int | None
|
|
371
|
+
content_digest: str | None
|
|
372
|
+
row_digest: str | None
|
|
373
|
+
schema_digest: str | None
|
|
374
|
+
row_count: int | None
|
|
375
|
+
frontier: SourceFrontier
|
|
376
|
+
max_event_time: str | None
|
|
377
|
+
partition_digests: tuple[PartitionDigest, ...]
|
|
378
|
+
acquisition_receipt_digest: str | None
|
|
379
|
+
next_watermark_candidate_digest: str | None
|
|
380
|
+
publication_reference_time: str | None
|
|
381
|
+
edition_sequence: int | None
|
|
382
|
+
schema_version: str = SOURCE_CADENCE_OBSERVATION_SCHEMA
|
|
383
|
+
|
|
384
|
+
def __post_init__(self) -> None:
|
|
385
|
+
if self.schema_version != SOURCE_CADENCE_OBSERVATION_SCHEMA:
|
|
386
|
+
_refuse("SCHEMA_VERSION", "observation.schema_version", "is not supported")
|
|
387
|
+
_identifier(self.source_id, "observation.source_id")
|
|
388
|
+
if type(self.sequence) is not int or not 0 <= self.sequence <= (1 << 53) - 1:
|
|
389
|
+
_refuse("SEQUENCE", "observation.sequence", "must be a non-negative safe integer")
|
|
390
|
+
_identifier(self.inference_epoch, "observation.inference_epoch")
|
|
391
|
+
_digest(
|
|
392
|
+
self.previous_observation_digest,
|
|
393
|
+
"observation.previous_observation_digest",
|
|
394
|
+
optional=True,
|
|
395
|
+
)
|
|
396
|
+
if (self.sequence == 0) != (self.previous_observation_digest is None):
|
|
397
|
+
_refuse(
|
|
398
|
+
"PREDECESSOR",
|
|
399
|
+
"observation.previous_observation_digest",
|
|
400
|
+
"must be null exactly for sequence zero",
|
|
401
|
+
)
|
|
402
|
+
for name, value in (
|
|
403
|
+
("connector_configuration_digest", self.connector_configuration_digest),
|
|
404
|
+
("locator_digest", self.locator_digest),
|
|
405
|
+
("recipe_digest", self.recipe_digest),
|
|
406
|
+
("source_authority_digest", self.source_authority_digest),
|
|
407
|
+
("reader_options_digest", self.reader_options_digest),
|
|
408
|
+
("frontier_policy_digest", self.frontier_policy_digest),
|
|
409
|
+
("time_interpretation_digest", self.time_interpretation_digest),
|
|
410
|
+
("inference_algorithm_digest", self.inference_algorithm_digest),
|
|
411
|
+
("probe_receipt_digest", self.probe_receipt_digest),
|
|
412
|
+
):
|
|
413
|
+
_digest(value, f"observation.{name}")
|
|
414
|
+
_reader_identifier(self.reader_family_id, "observation.reader_family_id")
|
|
415
|
+
if (
|
|
416
|
+
not isinstance(self.reader_family_version, str)
|
|
417
|
+
or len(self.reader_family_version) > 32
|
|
418
|
+
or _SEMVER.fullmatch(self.reader_family_version) is None
|
|
419
|
+
):
|
|
420
|
+
_refuse("SEMVER", "observation.reader_family_version", "must be x.y.z")
|
|
421
|
+
started = _timestamp(self.started_at, "observation.started_at")
|
|
422
|
+
completed = _timestamp(self.completed_at, "observation.completed_at")
|
|
423
|
+
if completed < started:
|
|
424
|
+
_refuse("TEMPORAL_ORDER", "observation.completed_at", "precedes started_at")
|
|
425
|
+
if not isinstance(self.result_kind, str) or self.result_kind not in _RESULT_KINDS:
|
|
426
|
+
_refuse("RESULT_KIND", "observation.result_kind", "is not supported")
|
|
427
|
+
if (
|
|
428
|
+
not isinstance(self.http_status_class, str)
|
|
429
|
+
or self.http_status_class not in _HTTP_STATUS_CLASSES
|
|
430
|
+
):
|
|
431
|
+
_refuse("HTTP_STATUS_CLASS", "observation.http_status_class", "is not supported")
|
|
432
|
+
_optional_int(
|
|
433
|
+
self.http_status_code,
|
|
434
|
+
"observation.http_status_code",
|
|
435
|
+
minimum=100,
|
|
436
|
+
maximum=599,
|
|
437
|
+
)
|
|
438
|
+
if self.http_status_class == "none":
|
|
439
|
+
if self.http_status_code is not None:
|
|
440
|
+
_refuse("HTTP_STATUS", "observation.http_status_code", "must be null")
|
|
441
|
+
elif (
|
|
442
|
+
self.http_status_code is None
|
|
443
|
+
or f"{self.http_status_code // 100}xx" != self.http_status_class
|
|
444
|
+
):
|
|
445
|
+
_refuse("HTTP_STATUS", "observation.http_status_code", "does not match its class")
|
|
446
|
+
for name, value in (
|
|
447
|
+
("conditional_request_digest", self.conditional_request_digest),
|
|
448
|
+
("etag_digest", self.etag_digest),
|
|
449
|
+
("last_modified_digest", self.last_modified_digest),
|
|
450
|
+
("content_digest", self.content_digest),
|
|
451
|
+
("row_digest", self.row_digest),
|
|
452
|
+
("schema_digest", self.schema_digest),
|
|
453
|
+
("acquisition_receipt_digest", self.acquisition_receipt_digest),
|
|
454
|
+
("next_watermark_candidate_digest", self.next_watermark_candidate_digest),
|
|
455
|
+
):
|
|
456
|
+
_digest(value, f"observation.{name}", optional=True)
|
|
457
|
+
_optional_int(
|
|
458
|
+
self.acquired_size_bytes,
|
|
459
|
+
"observation.acquired_size_bytes",
|
|
460
|
+
minimum=0,
|
|
461
|
+
maximum=1 << 40,
|
|
462
|
+
)
|
|
463
|
+
_optional_int(self.row_count, "observation.row_count", minimum=0, maximum=1_000_000_000)
|
|
464
|
+
_optional_int(
|
|
465
|
+
self.edition_sequence,
|
|
466
|
+
"observation.edition_sequence",
|
|
467
|
+
minimum=0,
|
|
468
|
+
maximum=(1 << 53) - 1,
|
|
469
|
+
)
|
|
470
|
+
if self.publication_reference_time is not None:
|
|
471
|
+
publication_reference = _timestamp(
|
|
472
|
+
self.publication_reference_time,
|
|
473
|
+
"observation.publication_reference_time",
|
|
474
|
+
)
|
|
475
|
+
if publication_reference > completed:
|
|
476
|
+
_refuse(
|
|
477
|
+
"PUBLICATION_TIME",
|
|
478
|
+
"observation.publication_reference_time",
|
|
479
|
+
"follows observation completion",
|
|
480
|
+
)
|
|
481
|
+
if (self.publication_reference_time is None) != (self.edition_sequence is None):
|
|
482
|
+
_refuse(
|
|
483
|
+
"PUBLICATION_COORDINATE",
|
|
484
|
+
"observation",
|
|
485
|
+
"publication time and edition sequence must appear together",
|
|
486
|
+
)
|
|
487
|
+
if not isinstance(self.frontier, SourceFrontier):
|
|
488
|
+
_refuse("TYPE", "observation.frontier", "must be a SourceFrontier")
|
|
489
|
+
if self.max_event_time is not None:
|
|
490
|
+
_timestamp(self.max_event_time, "observation.max_event_time")
|
|
491
|
+
if not isinstance(self.partition_digests, tuple) or len(self.partition_digests) > 128:
|
|
492
|
+
_refuse("PARTITIONS", "observation.partition_digests", "must be a bounded tuple")
|
|
493
|
+
if any(not isinstance(item, PartitionDigest) for item in self.partition_digests):
|
|
494
|
+
_refuse("PARTITIONS", "observation.partition_digests", "contains an invalid item")
|
|
495
|
+
keys = tuple(item.partition_key_digest for item in self.partition_digests)
|
|
496
|
+
if keys != tuple(sorted(keys)) or len(keys) != len(set(keys)):
|
|
497
|
+
_refuse(
|
|
498
|
+
"PARTITIONS",
|
|
499
|
+
"observation.partition_digests",
|
|
500
|
+
"must be unique and sorted by partition_key_digest",
|
|
501
|
+
)
|
|
502
|
+
acquired = (
|
|
503
|
+
self.acquired_size_bytes,
|
|
504
|
+
self.content_digest,
|
|
505
|
+
self.row_digest,
|
|
506
|
+
self.schema_digest,
|
|
507
|
+
self.row_count,
|
|
508
|
+
self.acquisition_receipt_digest,
|
|
509
|
+
)
|
|
510
|
+
if self.result_kind == "changed":
|
|
511
|
+
if self.failure_code is not None or any(value is None for value in acquired):
|
|
512
|
+
_refuse(
|
|
513
|
+
"RESULT_SHAPE",
|
|
514
|
+
"observation",
|
|
515
|
+
"changed requires complete acquired evidence",
|
|
516
|
+
)
|
|
517
|
+
if self.http_status_class not in {"none", "2xx"}:
|
|
518
|
+
_refuse("RESULT_STATUS", "observation", "changed requires a successful result")
|
|
519
|
+
if self.probe_receipt_digest != self.acquisition_receipt_digest:
|
|
520
|
+
_refuse(
|
|
521
|
+
"RESULT_SHAPE",
|
|
522
|
+
"observation.probe_receipt_digest",
|
|
523
|
+
"must be the acquired Receipt digest",
|
|
524
|
+
)
|
|
525
|
+
elif self.result_kind == "unchanged":
|
|
526
|
+
no_acquired_evidence = all(value is None for value in acquired)
|
|
527
|
+
complete_acquired_evidence = all(value is not None for value in acquired)
|
|
528
|
+
if self.failure_code is not None or not (
|
|
529
|
+
no_acquired_evidence or complete_acquired_evidence
|
|
530
|
+
):
|
|
531
|
+
_refuse(
|
|
532
|
+
"RESULT_SHAPE",
|
|
533
|
+
"observation",
|
|
534
|
+
"unchanged requires complete evidence or a conditional no-body result",
|
|
535
|
+
)
|
|
536
|
+
if no_acquired_evidence and self.next_watermark_candidate_digest is not None:
|
|
537
|
+
_refuse(
|
|
538
|
+
"RESULT_SHAPE",
|
|
539
|
+
"observation.next_watermark_candidate_digest",
|
|
540
|
+
"cannot advance without acquired evidence",
|
|
541
|
+
)
|
|
542
|
+
if no_acquired_evidence:
|
|
543
|
+
if (
|
|
544
|
+
self.http_status_code != 304
|
|
545
|
+
or self.conditional_request_digest is None
|
|
546
|
+
or self.frontier.kind != "none"
|
|
547
|
+
or self.max_event_time is not None
|
|
548
|
+
or self.partition_digests
|
|
549
|
+
or self.publication_reference_time is not None
|
|
550
|
+
):
|
|
551
|
+
_refuse(
|
|
552
|
+
"RESULT_SHAPE",
|
|
553
|
+
"observation",
|
|
554
|
+
"no-body unchanged requires a conditional 304 with no parsed facts",
|
|
555
|
+
)
|
|
556
|
+
else:
|
|
557
|
+
if self.http_status_class not in {"none", "2xx"}:
|
|
558
|
+
_refuse(
|
|
559
|
+
"RESULT_STATUS",
|
|
560
|
+
"observation",
|
|
561
|
+
"acquired unchanged requires a successful result",
|
|
562
|
+
)
|
|
563
|
+
if self.probe_receipt_digest != self.acquisition_receipt_digest:
|
|
564
|
+
_refuse(
|
|
565
|
+
"RESULT_SHAPE",
|
|
566
|
+
"observation.probe_receipt_digest",
|
|
567
|
+
"must be the acquired Receipt digest",
|
|
568
|
+
)
|
|
569
|
+
elif self.result_kind in {"not_published", "failed"}:
|
|
570
|
+
if any(value is not None for value in acquired) or self.frontier.kind != "none":
|
|
571
|
+
_refuse("RESULT_SHAPE", "observation", "result cannot claim acquired evidence")
|
|
572
|
+
if self.next_watermark_candidate_digest is not None:
|
|
573
|
+
_refuse(
|
|
574
|
+
"RESULT_SHAPE",
|
|
575
|
+
"observation.next_watermark_candidate_digest",
|
|
576
|
+
"cannot advance without acquired evidence",
|
|
577
|
+
)
|
|
578
|
+
if self.partition_digests or self.max_event_time is not None:
|
|
579
|
+
_refuse("RESULT_SHAPE", "observation", "result cannot claim parsed evidence")
|
|
580
|
+
if self.publication_reference_time is not None:
|
|
581
|
+
_refuse("RESULT_SHAPE", "observation", "result cannot claim a publication")
|
|
582
|
+
if self.result_kind == "not_published" and self.http_status_code not in {
|
|
583
|
+
None,
|
|
584
|
+
200,
|
|
585
|
+
202,
|
|
586
|
+
204,
|
|
587
|
+
404,
|
|
588
|
+
}:
|
|
589
|
+
_refuse(
|
|
590
|
+
"RESULT_STATUS",
|
|
591
|
+
"observation.http_status_code",
|
|
592
|
+
"is not a successful not-published outcome",
|
|
593
|
+
)
|
|
594
|
+
if self.result_kind == "failed":
|
|
595
|
+
if not isinstance(self.failure_code, str) or self.failure_code not in _FAILURE_CODES:
|
|
596
|
+
_refuse("FAILURE_CODE", "observation.failure_code", "is not supported")
|
|
597
|
+
elif self.failure_code is not None:
|
|
598
|
+
_refuse("RESULT_SHAPE", "observation.failure_code", "must be null unless failed")
|
|
599
|
+
|
|
600
|
+
@property
|
|
601
|
+
def digest(self) -> str:
|
|
602
|
+
return canonical_sha256(self.to_dict())
|
|
603
|
+
|
|
604
|
+
def to_dict(self) -> dict[str, Any]:
|
|
605
|
+
return {
|
|
606
|
+
"schema_version": self.schema_version,
|
|
607
|
+
"source_id": self.source_id,
|
|
608
|
+
"sequence": self.sequence,
|
|
609
|
+
"inference_epoch": self.inference_epoch,
|
|
610
|
+
"previous_observation_digest": self.previous_observation_digest,
|
|
611
|
+
"connector_configuration_digest": self.connector_configuration_digest,
|
|
612
|
+
"locator_digest": self.locator_digest,
|
|
613
|
+
"recipe_digest": self.recipe_digest,
|
|
614
|
+
"source_authority_digest": self.source_authority_digest,
|
|
615
|
+
"reader_family_id": self.reader_family_id,
|
|
616
|
+
"reader_family_version": self.reader_family_version,
|
|
617
|
+
"reader_options_digest": self.reader_options_digest,
|
|
618
|
+
"frontier_policy_digest": self.frontier_policy_digest,
|
|
619
|
+
"time_interpretation_digest": self.time_interpretation_digest,
|
|
620
|
+
"inference_algorithm_digest": self.inference_algorithm_digest,
|
|
621
|
+
"started_at": self.started_at,
|
|
622
|
+
"completed_at": self.completed_at,
|
|
623
|
+
"result_kind": self.result_kind,
|
|
624
|
+
"failure_code": self.failure_code,
|
|
625
|
+
"http_status_class": self.http_status_class,
|
|
626
|
+
"http_status_code": self.http_status_code,
|
|
627
|
+
"conditional_request_digest": self.conditional_request_digest,
|
|
628
|
+
"probe_receipt_digest": self.probe_receipt_digest,
|
|
629
|
+
"etag_digest": self.etag_digest,
|
|
630
|
+
"last_modified_digest": self.last_modified_digest,
|
|
631
|
+
"acquired_size_bytes": self.acquired_size_bytes,
|
|
632
|
+
"content_digest": self.content_digest,
|
|
633
|
+
"row_digest": self.row_digest,
|
|
634
|
+
"schema_digest": self.schema_digest,
|
|
635
|
+
"row_count": self.row_count,
|
|
636
|
+
"frontier": self.frontier.to_dict(),
|
|
637
|
+
"max_event_time": self.max_event_time,
|
|
638
|
+
"partition_digests": [item.to_dict() for item in self.partition_digests],
|
|
639
|
+
"acquisition_receipt_digest": self.acquisition_receipt_digest,
|
|
640
|
+
"next_watermark_candidate_digest": self.next_watermark_candidate_digest,
|
|
641
|
+
"publication_reference_time": self.publication_reference_time,
|
|
642
|
+
"edition_sequence": self.edition_sequence,
|
|
643
|
+
}
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
@dataclass(frozen=True)
|
|
647
|
+
class CadenceInference:
|
|
648
|
+
"""Pure replay result for one exact empirical observation history."""
|
|
649
|
+
|
|
650
|
+
source_id: str
|
|
651
|
+
inference_epoch: str
|
|
652
|
+
history_digest: str
|
|
653
|
+
state: str
|
|
654
|
+
cadence_kind: str
|
|
655
|
+
estimated_interval_seconds: int | None
|
|
656
|
+
interval_jitter_seconds: int | None
|
|
657
|
+
publication_delay_min_seconds: int | None
|
|
658
|
+
publication_delay_median_seconds: int | None
|
|
659
|
+
publication_delay_max_seconds: int | None
|
|
660
|
+
expected_next_earliest: str | None
|
|
661
|
+
expected_next_latest: str | None
|
|
662
|
+
confidence_basis_points: int
|
|
663
|
+
evidence_count: int
|
|
664
|
+
evidence_started_at: str | None
|
|
665
|
+
evidence_completed_at: str | None
|
|
666
|
+
recommended_probe_interval_seconds: int
|
|
667
|
+
revision_style: str
|
|
668
|
+
recommended_epoch_reset: bool
|
|
669
|
+
reason_codes: tuple[str, ...]
|
|
670
|
+
algorithm_version: str = SOURCE_CADENCE_ALGORITHM_VERSION
|
|
671
|
+
algorithm_digest: str = SOURCE_CADENCE_ALGORITHM_DIGEST
|
|
672
|
+
schema_version: str = SOURCE_CADENCE_INFERENCE_SCHEMA
|
|
673
|
+
|
|
674
|
+
def __post_init__(self) -> None:
|
|
675
|
+
if self.schema_version != SOURCE_CADENCE_INFERENCE_SCHEMA:
|
|
676
|
+
_refuse("SCHEMA_VERSION", "inference.schema_version", "is not supported")
|
|
677
|
+
if (
|
|
678
|
+
self.algorithm_version != SOURCE_CADENCE_ALGORITHM_VERSION
|
|
679
|
+
or self.algorithm_digest != SOURCE_CADENCE_ALGORITHM_DIGEST
|
|
680
|
+
):
|
|
681
|
+
_refuse("ALGORITHM", "inference.algorithm_version", "is not the exact algorithm")
|
|
682
|
+
_identifier(self.source_id, "inference.source_id")
|
|
683
|
+
_identifier(self.inference_epoch, "inference.inference_epoch")
|
|
684
|
+
_digest(self.history_digest, "inference.history_digest")
|
|
685
|
+
if (
|
|
686
|
+
not isinstance(self.state, str)
|
|
687
|
+
or self.state not in _INFERENCE_STATES
|
|
688
|
+
or not isinstance(self.cadence_kind, str)
|
|
689
|
+
or self.cadence_kind not in _CADENCE_KINDS
|
|
690
|
+
):
|
|
691
|
+
_refuse("INFERENCE_STATE", "inference.state", "is not supported")
|
|
692
|
+
allowed_cadence = {
|
|
693
|
+
"unknown": {"unknown"},
|
|
694
|
+
"learning": {"unknown"},
|
|
695
|
+
"stable": {"regular"},
|
|
696
|
+
"dormant": {"regular"},
|
|
697
|
+
"changed_pattern": {"regular", "irregular"},
|
|
698
|
+
}
|
|
699
|
+
if self.cadence_kind not in allowed_cadence[self.state]:
|
|
700
|
+
_refuse("INFERENCE_STATE", "inference.cadence_kind", "contradicts the state")
|
|
701
|
+
if not isinstance(self.revision_style, str) or self.revision_style not in _REVISION_STYLES:
|
|
702
|
+
_refuse("REVISION_STYLE", "inference.revision_style", "is not supported")
|
|
703
|
+
_optional_int(
|
|
704
|
+
self.estimated_interval_seconds,
|
|
705
|
+
"inference.estimated_interval_seconds",
|
|
706
|
+
minimum=1,
|
|
707
|
+
maximum=1 << 40,
|
|
708
|
+
)
|
|
709
|
+
for name, value in (
|
|
710
|
+
("interval_jitter_seconds", self.interval_jitter_seconds),
|
|
711
|
+
("publication_delay_min_seconds", self.publication_delay_min_seconds),
|
|
712
|
+
("publication_delay_median_seconds", self.publication_delay_median_seconds),
|
|
713
|
+
("publication_delay_max_seconds", self.publication_delay_max_seconds),
|
|
714
|
+
):
|
|
715
|
+
_optional_int(value, f"inference.{name}", minimum=0, maximum=1 << 40)
|
|
716
|
+
if self.cadence_kind == "regular" and self.estimated_interval_seconds is None:
|
|
717
|
+
_refuse("INFERENCE_STATE", "inference.estimated_interval_seconds", "is required")
|
|
718
|
+
if self.cadence_kind == "irregular" and self.estimated_interval_seconds is not None:
|
|
719
|
+
_refuse("INFERENCE_STATE", "inference.estimated_interval_seconds", "must be null")
|
|
720
|
+
delays = (
|
|
721
|
+
self.publication_delay_min_seconds,
|
|
722
|
+
self.publication_delay_median_seconds,
|
|
723
|
+
self.publication_delay_max_seconds,
|
|
724
|
+
)
|
|
725
|
+
if any(value is None for value in delays) and not all(value is None for value in delays):
|
|
726
|
+
_refuse("PUBLICATION_DELAY", "inference", "delay values must appear together")
|
|
727
|
+
if all(value is not None for value in delays):
|
|
728
|
+
delay_min, delay_median, delay_max = delays
|
|
729
|
+
assert delay_min is not None and delay_median is not None and delay_max is not None
|
|
730
|
+
if not delay_min <= delay_median <= delay_max:
|
|
731
|
+
_refuse("PUBLICATION_DELAY", "inference", "delay values are out of order")
|
|
732
|
+
for name, value in (
|
|
733
|
+
("expected_next_earliest", self.expected_next_earliest),
|
|
734
|
+
("expected_next_latest", self.expected_next_latest),
|
|
735
|
+
("evidence_started_at", self.evidence_started_at),
|
|
736
|
+
("evidence_completed_at", self.evidence_completed_at),
|
|
737
|
+
):
|
|
738
|
+
if value is not None:
|
|
739
|
+
_timestamp(value, f"inference.{name}")
|
|
740
|
+
if (self.expected_next_earliest is None) != (self.expected_next_latest is None):
|
|
741
|
+
_refuse(
|
|
742
|
+
"EXPECTED_WINDOW",
|
|
743
|
+
"inference",
|
|
744
|
+
"expected window endpoints must appear together",
|
|
745
|
+
)
|
|
746
|
+
if self.expected_next_earliest is not None and _timestamp(
|
|
747
|
+
self.expected_next_latest, "inference.expected_next_latest"
|
|
748
|
+
) < _timestamp(self.expected_next_earliest, "inference.expected_next_earliest"):
|
|
749
|
+
_refuse("EXPECTED_WINDOW", "inference.expected_next_latest", "precedes earliest")
|
|
750
|
+
if self.expected_next_earliest is not None and self.state != "stable":
|
|
751
|
+
_refuse("EXPECTED_WINDOW", "inference", "is available only for stable cadence")
|
|
752
|
+
if (
|
|
753
|
+
type(self.confidence_basis_points) is not int
|
|
754
|
+
or not 0 <= self.confidence_basis_points <= 10_000
|
|
755
|
+
):
|
|
756
|
+
_refuse("CONFIDENCE", "inference.confidence_basis_points", "must be 0..10000")
|
|
757
|
+
if type(self.evidence_count) is not int or not 0 <= self.evidence_count <= 4_096:
|
|
758
|
+
_refuse("EVIDENCE_COUNT", "inference.evidence_count", "must be 0..4096")
|
|
759
|
+
if (self.evidence_count == 0) != (self.evidence_started_at is None):
|
|
760
|
+
_refuse("EVIDENCE_COUNT", "inference.evidence_started_at", "does not match the count")
|
|
761
|
+
if (self.evidence_count == 0) != (self.evidence_completed_at is None):
|
|
762
|
+
_refuse("EVIDENCE_COUNT", "inference.evidence_completed_at", "does not match the count")
|
|
763
|
+
if self.state == "unknown" and self.evidence_count != 0:
|
|
764
|
+
_refuse("EVIDENCE_COUNT", "inference.evidence_count", "contradicts unknown state")
|
|
765
|
+
if self.state != "unknown" and self.evidence_count == 0:
|
|
766
|
+
_refuse("EVIDENCE_COUNT", "inference.evidence_count", "contradicts observed state")
|
|
767
|
+
if self.evidence_started_at is not None and _timestamp(
|
|
768
|
+
self.evidence_completed_at,
|
|
769
|
+
"inference.evidence_completed_at",
|
|
770
|
+
) < _timestamp(self.evidence_started_at, "inference.evidence_started_at"):
|
|
771
|
+
_refuse("EVIDENCE_TIME", "inference.evidence_completed_at", "precedes start")
|
|
772
|
+
if (
|
|
773
|
+
type(self.recommended_probe_interval_seconds) is not int
|
|
774
|
+
or not 300 <= self.recommended_probe_interval_seconds <= 30 * 24 * 60 * 60
|
|
775
|
+
):
|
|
776
|
+
_refuse(
|
|
777
|
+
"PROBE_INTERVAL",
|
|
778
|
+
"inference.recommended_probe_interval_seconds",
|
|
779
|
+
"is out of bounds",
|
|
780
|
+
)
|
|
781
|
+
if not isinstance(self.recommended_epoch_reset, bool):
|
|
782
|
+
_refuse("TYPE", "inference.recommended_epoch_reset", "must be boolean")
|
|
783
|
+
if self.recommended_epoch_reset and self.state != "changed_pattern":
|
|
784
|
+
_refuse("INFERENCE_STATE", "inference.recommended_epoch_reset", "contradicts state")
|
|
785
|
+
if (
|
|
786
|
+
not isinstance(self.reason_codes, tuple)
|
|
787
|
+
or not 1 <= len(self.reason_codes) <= 32
|
|
788
|
+
or any(
|
|
789
|
+
not isinstance(item, str)
|
|
790
|
+
or len(item) > 64
|
|
791
|
+
or _READER_IDENTIFIER.fullmatch(item) is None
|
|
792
|
+
for item in self.reason_codes
|
|
793
|
+
)
|
|
794
|
+
or len(set(self.reason_codes)) != len(self.reason_codes)
|
|
795
|
+
):
|
|
796
|
+
_refuse("REASON_CODES", "inference.reason_codes", "must be unique typed codes")
|
|
797
|
+
reason_set = set(self.reason_codes)
|
|
798
|
+
reason_order = tuple(_SOURCE_CADENCE_ALGORITHM_SPEC["reason_codes"])
|
|
799
|
+
supported_reasons = set(reason_order)
|
|
800
|
+
if not reason_set <= supported_reasons:
|
|
801
|
+
_refuse("REASON_CODES", "inference.reason_codes", "contains an unsupported code")
|
|
802
|
+
if self.reason_codes != tuple(item for item in reason_order if item in reason_set):
|
|
803
|
+
_refuse("REASON_CODES", "inference.reason_codes", "is not in canonical order")
|
|
804
|
+
if self.state == "unknown":
|
|
805
|
+
if (
|
|
806
|
+
self.source_id != "unknown"
|
|
807
|
+
or self.inference_epoch != "unknown"
|
|
808
|
+
or self.history_digest != canonical_sha256([])
|
|
809
|
+
or self.estimated_interval_seconds is not None
|
|
810
|
+
or self.interval_jitter_seconds is not None
|
|
811
|
+
or any(value is not None for value in delays)
|
|
812
|
+
or self.expected_next_earliest is not None
|
|
813
|
+
or self.confidence_basis_points != 0
|
|
814
|
+
or self.revision_style != "unknown"
|
|
815
|
+
or self.recommended_epoch_reset
|
|
816
|
+
or self.reason_codes != ("no_observations",)
|
|
817
|
+
or self.recommended_probe_interval_seconds != 21_600
|
|
818
|
+
):
|
|
819
|
+
_refuse("INFERENCE_STATE", "inference", "is not the canonical unknown result")
|
|
820
|
+
elif self.state == "learning":
|
|
821
|
+
if (
|
|
822
|
+
"insufficient_changes" not in reason_set
|
|
823
|
+
or not reason_set
|
|
824
|
+
<= {
|
|
825
|
+
"missing_publication_coordinates",
|
|
826
|
+
"insufficient_changes",
|
|
827
|
+
"probe_failures_observed",
|
|
828
|
+
}
|
|
829
|
+
or self.expected_next_earliest is not None
|
|
830
|
+
or self.recommended_epoch_reset
|
|
831
|
+
):
|
|
832
|
+
_refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts learning state")
|
|
833
|
+
elif self.state == "stable":
|
|
834
|
+
if (
|
|
835
|
+
"regular_intervals" not in reason_set
|
|
836
|
+
or not reason_set <= {"regular_intervals", "probe_failures_observed"}
|
|
837
|
+
or self.expected_next_earliest is None
|
|
838
|
+
or self.confidence_basis_points < 3_000
|
|
839
|
+
or self.recommended_epoch_reset
|
|
840
|
+
):
|
|
841
|
+
_refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts stable state")
|
|
842
|
+
elif self.state == "dormant":
|
|
843
|
+
if (
|
|
844
|
+
not {"regular_intervals", "prolonged_no_change"} <= reason_set
|
|
845
|
+
or not reason_set
|
|
846
|
+
<= {"regular_intervals", "prolonged_no_change", "probe_failures_observed"}
|
|
847
|
+
or self.expected_next_earliest is not None
|
|
848
|
+
or self.confidence_basis_points < 2_900
|
|
849
|
+
or self.recommended_epoch_reset
|
|
850
|
+
):
|
|
851
|
+
_refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts dormant state")
|
|
852
|
+
elif self.cadence_kind == "irregular":
|
|
853
|
+
base_reasons = reason_set & {"regular_intervals", "irregular_intervals"}
|
|
854
|
+
allowed = {
|
|
855
|
+
"regular_intervals",
|
|
856
|
+
"irregular_intervals",
|
|
857
|
+
"cadence_regime_changed",
|
|
858
|
+
"probe_failures_observed",
|
|
859
|
+
}
|
|
860
|
+
base_is_canonical = (
|
|
861
|
+
len(base_reasons) == 1
|
|
862
|
+
if self.recommended_epoch_reset
|
|
863
|
+
else base_reasons == {"irregular_intervals"}
|
|
864
|
+
)
|
|
865
|
+
if not base_is_canonical or not reason_set <= allowed:
|
|
866
|
+
_refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts irregular state")
|
|
867
|
+
elif not {
|
|
868
|
+
"regular_intervals",
|
|
869
|
+
"publication_window_missed",
|
|
870
|
+
} <= reason_set or not reason_set <= {
|
|
871
|
+
"regular_intervals",
|
|
872
|
+
"publication_window_missed",
|
|
873
|
+
"probe_failures_observed",
|
|
874
|
+
}:
|
|
875
|
+
_refuse(
|
|
876
|
+
"INFERENCE_STATE",
|
|
877
|
+
"inference.reason_codes",
|
|
878
|
+
"contradicts changed regular state",
|
|
879
|
+
)
|
|
880
|
+
if self.recommended_epoch_reset != ("cadence_regime_changed" in reason_set):
|
|
881
|
+
_refuse(
|
|
882
|
+
"INFERENCE_STATE",
|
|
883
|
+
"inference.reason_codes",
|
|
884
|
+
"does not match the epoch-reset decision",
|
|
885
|
+
)
|
|
886
|
+
if self.cadence_kind in {"regular", "irregular"} and self.interval_jitter_seconds is None:
|
|
887
|
+
_refuse("INFERENCE_STATE", "inference.interval_jitter_seconds", "is required")
|
|
888
|
+
|
|
889
|
+
@property
|
|
890
|
+
def digest(self) -> str:
|
|
891
|
+
return canonical_sha256(self.to_dict())
|
|
892
|
+
|
|
893
|
+
def to_dict(self) -> dict[str, Any]:
|
|
894
|
+
return {
|
|
895
|
+
"schema_version": self.schema_version,
|
|
896
|
+
"algorithm_version": self.algorithm_version,
|
|
897
|
+
"algorithm_digest": self.algorithm_digest,
|
|
898
|
+
"source_id": self.source_id,
|
|
899
|
+
"inference_epoch": self.inference_epoch,
|
|
900
|
+
"history_digest": self.history_digest,
|
|
901
|
+
"state": self.state,
|
|
902
|
+
"cadence_kind": self.cadence_kind,
|
|
903
|
+
"estimated_interval_seconds": self.estimated_interval_seconds,
|
|
904
|
+
"interval_jitter_seconds": self.interval_jitter_seconds,
|
|
905
|
+
"publication_delay_min_seconds": self.publication_delay_min_seconds,
|
|
906
|
+
"publication_delay_median_seconds": self.publication_delay_median_seconds,
|
|
907
|
+
"publication_delay_max_seconds": self.publication_delay_max_seconds,
|
|
908
|
+
"expected_next_earliest": self.expected_next_earliest,
|
|
909
|
+
"expected_next_latest": self.expected_next_latest,
|
|
910
|
+
"confidence_basis_points": self.confidence_basis_points,
|
|
911
|
+
"evidence_count": self.evidence_count,
|
|
912
|
+
"evidence_started_at": self.evidence_started_at,
|
|
913
|
+
"evidence_completed_at": self.evidence_completed_at,
|
|
914
|
+
"recommended_probe_interval_seconds": self.recommended_probe_interval_seconds,
|
|
915
|
+
"revision_style": self.revision_style,
|
|
916
|
+
"recommended_epoch_reset": self.recommended_epoch_reset,
|
|
917
|
+
"reason_codes": list(self.reason_codes),
|
|
918
|
+
}
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
_OBSERVATION_KEYS = set(EmpiricalSourceObservation.__dataclass_fields__)
|
|
922
|
+
_INFERENCE_KEYS = set(CadenceInference.__dataclass_fields__)
|
|
923
|
+
|
|
924
|
+
|
|
925
|
+
def parse_empirical_source_observation(value: object) -> EmpiricalSourceObservation:
|
|
926
|
+
"""Parse one exact observation dictionary and reject unknown or missing fields."""
|
|
927
|
+
|
|
928
|
+
if not isinstance(value, Mapping):
|
|
929
|
+
_refuse("TYPE", "observation", "must be an object")
|
|
930
|
+
payload = dict(value)
|
|
931
|
+
_exact(payload, _OBSERVATION_KEYS, "observation")
|
|
932
|
+
frontier_raw = payload["frontier"]
|
|
933
|
+
if not isinstance(frontier_raw, Mapping):
|
|
934
|
+
_refuse("TYPE", "observation.frontier", "must be an object")
|
|
935
|
+
frontier = dict(frontier_raw)
|
|
936
|
+
_exact(frontier, {"kind", "value", "derivation_rule"}, "observation.frontier")
|
|
937
|
+
partitions_raw = payload["partition_digests"]
|
|
938
|
+
if not isinstance(partitions_raw, list):
|
|
939
|
+
_refuse("TYPE", "observation.partition_digests", "must be an array")
|
|
940
|
+
partitions: list[PartitionDigest] = []
|
|
941
|
+
for index, raw in enumerate(partitions_raw):
|
|
942
|
+
if not isinstance(raw, Mapping):
|
|
943
|
+
_refuse("TYPE", f"observation.partition_digests[{index}]", "must be an object")
|
|
944
|
+
item = dict(raw)
|
|
945
|
+
_exact(
|
|
946
|
+
item,
|
|
947
|
+
{"partition_key_digest", "content_digest"},
|
|
948
|
+
f"observation.partition_digests[{index}]",
|
|
949
|
+
)
|
|
950
|
+
partitions.append(PartitionDigest(**item))
|
|
951
|
+
payload["frontier"] = SourceFrontier(**frontier)
|
|
952
|
+
payload["partition_digests"] = tuple(partitions)
|
|
953
|
+
return EmpiricalSourceObservation(**payload)
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
def parse_cadence_inference(value: object) -> CadenceInference:
|
|
957
|
+
"""Parse one exact inference dictionary."""
|
|
958
|
+
|
|
959
|
+
if not isinstance(value, Mapping):
|
|
960
|
+
_refuse("TYPE", "inference", "must be an object")
|
|
961
|
+
payload = dict(value)
|
|
962
|
+
_exact(payload, _INFERENCE_KEYS, "inference")
|
|
963
|
+
reasons = payload["reason_codes"]
|
|
964
|
+
if not isinstance(reasons, list):
|
|
965
|
+
_refuse("TYPE", "inference.reason_codes", "must be an array")
|
|
966
|
+
payload["reason_codes"] = tuple(reasons)
|
|
967
|
+
return CadenceInference(**payload)
|
|
968
|
+
|
|
969
|
+
|
|
970
|
+
def validate_observation_history(
|
|
971
|
+
history: Sequence[EmpiricalSourceObservation],
|
|
972
|
+
) -> tuple[EmpiricalSourceObservation, ...]:
|
|
973
|
+
"""Validate order, predecessor links, coordinates, and change semantics."""
|
|
974
|
+
|
|
975
|
+
observations = tuple(history)
|
|
976
|
+
if len(observations) > 4_096:
|
|
977
|
+
_refuse("HISTORY_SIZE", "history", "exceeds 4096 observations")
|
|
978
|
+
if not observations:
|
|
979
|
+
return observations
|
|
980
|
+
if any(not isinstance(item, EmpiricalSourceObservation) for item in observations):
|
|
981
|
+
_refuse("TYPE", "history", "contains a non-observation item")
|
|
982
|
+
if sum(len(item.partition_digests) for item in observations) > 8_192:
|
|
983
|
+
_refuse("HISTORY_SIZE", "history", "exceeds 8192 partition records")
|
|
984
|
+
first = observations[0]
|
|
985
|
+
identity = (
|
|
986
|
+
first.source_id,
|
|
987
|
+
first.inference_epoch,
|
|
988
|
+
first.connector_configuration_digest,
|
|
989
|
+
first.locator_digest,
|
|
990
|
+
first.recipe_digest,
|
|
991
|
+
first.source_authority_digest,
|
|
992
|
+
first.reader_family_id,
|
|
993
|
+
first.reader_family_version,
|
|
994
|
+
first.reader_options_digest,
|
|
995
|
+
first.frontier_policy_digest,
|
|
996
|
+
first.time_interpretation_digest,
|
|
997
|
+
first.inference_algorithm_digest,
|
|
998
|
+
)
|
|
999
|
+
latest_acquired: EmpiricalSourceObservation | None = None
|
|
1000
|
+
latest_change_time: datetime | None = None
|
|
1001
|
+
latest_publication_time: datetime | None = None
|
|
1002
|
+
latest_edition_sequence: int | None = None
|
|
1003
|
+
publication_coordinates_present: bool | None = None
|
|
1004
|
+
previous: EmpiricalSourceObservation | None = None
|
|
1005
|
+
for index, item in enumerate(observations):
|
|
1006
|
+
if item.sequence != index:
|
|
1007
|
+
_refuse("HISTORY_SEQUENCE", f"history[{index}].sequence", "is not contiguous")
|
|
1008
|
+
current_identity = (
|
|
1009
|
+
item.source_id,
|
|
1010
|
+
item.inference_epoch,
|
|
1011
|
+
item.connector_configuration_digest,
|
|
1012
|
+
item.locator_digest,
|
|
1013
|
+
item.recipe_digest,
|
|
1014
|
+
item.source_authority_digest,
|
|
1015
|
+
item.reader_family_id,
|
|
1016
|
+
item.reader_family_version,
|
|
1017
|
+
item.reader_options_digest,
|
|
1018
|
+
item.frontier_policy_digest,
|
|
1019
|
+
item.time_interpretation_digest,
|
|
1020
|
+
item.inference_algorithm_digest,
|
|
1021
|
+
)
|
|
1022
|
+
if current_identity != identity:
|
|
1023
|
+
_refuse("HISTORY_IDENTITY", f"history[{index}]", "splices another source or epoch")
|
|
1024
|
+
if previous is None:
|
|
1025
|
+
if item.previous_observation_digest is not None:
|
|
1026
|
+
_refuse("HISTORY_PREDECESSOR", f"history[{index}]", "has an unexpected predecessor")
|
|
1027
|
+
else:
|
|
1028
|
+
if item.previous_observation_digest != previous.digest:
|
|
1029
|
+
_refuse("HISTORY_PREDECESSOR", f"history[{index}]", "does not bind its predecessor")
|
|
1030
|
+
if _timestamp(item.started_at, f"history[{index}].started_at") < _timestamp(
|
|
1031
|
+
previous.completed_at, f"history[{index - 1}].completed_at"
|
|
1032
|
+
):
|
|
1033
|
+
_refuse("HISTORY_TIME", f"history[{index}].started_at", "overlaps its predecessor")
|
|
1034
|
+
if item.result_kind in {"changed", "unchanged"} and item.content_digest is not None:
|
|
1035
|
+
if item.result_kind == "unchanged" and latest_acquired is None:
|
|
1036
|
+
_refuse(
|
|
1037
|
+
"UNCHANGED_BASELINE",
|
|
1038
|
+
f"history[{index}]",
|
|
1039
|
+
"has no prior acquired fact in this epoch",
|
|
1040
|
+
)
|
|
1041
|
+
if latest_acquired is not None:
|
|
1042
|
+
if (
|
|
1043
|
+
item.frontier.kind != latest_acquired.frontier.kind
|
|
1044
|
+
or item.frontier.derivation_rule != latest_acquired.frontier.derivation_rule
|
|
1045
|
+
):
|
|
1046
|
+
_refuse(
|
|
1047
|
+
"HISTORY_IDENTITY",
|
|
1048
|
+
f"history[{index}].frontier",
|
|
1049
|
+
"changes the frontier policy within one epoch",
|
|
1050
|
+
)
|
|
1051
|
+
same_content = item.content_digest == latest_acquired.content_digest
|
|
1052
|
+
same_rows = item.row_digest == latest_acquired.row_digest
|
|
1053
|
+
same_schema = item.schema_digest == latest_acquired.schema_digest
|
|
1054
|
+
same_frontier = item.frontier == latest_acquired.frontier
|
|
1055
|
+
same_acquired_facts = (
|
|
1056
|
+
item.acquired_size_bytes == latest_acquired.acquired_size_bytes
|
|
1057
|
+
and item.row_count == latest_acquired.row_count
|
|
1058
|
+
and item.max_event_time == latest_acquired.max_event_time
|
|
1059
|
+
and item.partition_digests == latest_acquired.partition_digests
|
|
1060
|
+
and item.next_watermark_candidate_digest
|
|
1061
|
+
== latest_acquired.next_watermark_candidate_digest
|
|
1062
|
+
and item.publication_reference_time
|
|
1063
|
+
== latest_acquired.publication_reference_time
|
|
1064
|
+
and item.edition_sequence == latest_acquired.edition_sequence
|
|
1065
|
+
)
|
|
1066
|
+
if item.result_kind == "changed" and same_content and same_rows and same_schema:
|
|
1067
|
+
_refuse(
|
|
1068
|
+
"CHANGE_SEMANTICS",
|
|
1069
|
+
f"history[{index}]",
|
|
1070
|
+
"claims changed for identical data",
|
|
1071
|
+
)
|
|
1072
|
+
if item.result_kind == "unchanged" and not (
|
|
1073
|
+
same_content
|
|
1074
|
+
and same_rows
|
|
1075
|
+
and same_schema
|
|
1076
|
+
and same_frontier
|
|
1077
|
+
and same_acquired_facts
|
|
1078
|
+
):
|
|
1079
|
+
_refuse(
|
|
1080
|
+
"CHANGE_SEMANTICS",
|
|
1081
|
+
f"history[{index}]",
|
|
1082
|
+
"claims unchanged for changed data",
|
|
1083
|
+
)
|
|
1084
|
+
comparison = _compare_frontiers(latest_acquired.frontier, item.frontier)
|
|
1085
|
+
if comparison is not None and comparison < 0:
|
|
1086
|
+
_refuse("FRONTIER_REGRESSION", f"history[{index}].frontier", "regresses")
|
|
1087
|
+
if comparison is not None and comparison > 0 and same_content:
|
|
1088
|
+
_refuse(
|
|
1089
|
+
"FRONTIER_CONTENT",
|
|
1090
|
+
f"history[{index}]",
|
|
1091
|
+
"advances with identical content",
|
|
1092
|
+
)
|
|
1093
|
+
latest_acquired = item
|
|
1094
|
+
elif item.result_kind == "unchanged" and latest_acquired is None:
|
|
1095
|
+
_refuse(
|
|
1096
|
+
"UNCHANGED_BASELINE",
|
|
1097
|
+
f"history[{index}]",
|
|
1098
|
+
"has no prior acquired fact in this epoch",
|
|
1099
|
+
)
|
|
1100
|
+
if item.result_kind == "changed":
|
|
1101
|
+
has_publication_coordinates = item.publication_reference_time is not None
|
|
1102
|
+
if publication_coordinates_present is None:
|
|
1103
|
+
publication_coordinates_present = has_publication_coordinates
|
|
1104
|
+
elif publication_coordinates_present != has_publication_coordinates:
|
|
1105
|
+
_refuse(
|
|
1106
|
+
"HISTORY_IDENTITY",
|
|
1107
|
+
f"history[{index}]",
|
|
1108
|
+
"changes publication-coordinate policy within one epoch",
|
|
1109
|
+
)
|
|
1110
|
+
change_time = _timestamp(item.completed_at, f"history[{index}].completed_at")
|
|
1111
|
+
if latest_change_time is not None and change_time <= latest_change_time:
|
|
1112
|
+
_refuse(
|
|
1113
|
+
"HISTORY_CHANGE_TIME",
|
|
1114
|
+
f"history[{index}].completed_at",
|
|
1115
|
+
"does not follow the prior material change",
|
|
1116
|
+
)
|
|
1117
|
+
latest_change_time = change_time
|
|
1118
|
+
if item.publication_reference_time is not None:
|
|
1119
|
+
publication_time = _timestamp(
|
|
1120
|
+
item.publication_reference_time,
|
|
1121
|
+
f"history[{index}].publication_reference_time",
|
|
1122
|
+
)
|
|
1123
|
+
assert item.edition_sequence is not None
|
|
1124
|
+
if (
|
|
1125
|
+
latest_publication_time is not None
|
|
1126
|
+
and publication_time <= latest_publication_time
|
|
1127
|
+
):
|
|
1128
|
+
_refuse(
|
|
1129
|
+
"PUBLICATION_ORDER",
|
|
1130
|
+
f"history[{index}].publication_reference_time",
|
|
1131
|
+
"does not follow the prior changed edition",
|
|
1132
|
+
)
|
|
1133
|
+
if (
|
|
1134
|
+
latest_edition_sequence is not None
|
|
1135
|
+
and item.edition_sequence <= latest_edition_sequence
|
|
1136
|
+
):
|
|
1137
|
+
_refuse(
|
|
1138
|
+
"PUBLICATION_ORDER",
|
|
1139
|
+
f"history[{index}].edition_sequence",
|
|
1140
|
+
"does not follow the prior changed edition",
|
|
1141
|
+
)
|
|
1142
|
+
latest_publication_time = publication_time
|
|
1143
|
+
latest_edition_sequence = item.edition_sequence
|
|
1144
|
+
previous = item
|
|
1145
|
+
return observations
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
def _frontier_value(frontier: SourceFrontier) -> object:
|
|
1149
|
+
if frontier.kind == "date":
|
|
1150
|
+
return datetime.strptime(frontier.value, "%Y-%m-%d").date()
|
|
1151
|
+
if frontier.kind == "timestamp":
|
|
1152
|
+
return _timestamp(frontier.value, "frontier.value")
|
|
1153
|
+
if frontier.kind == "integer":
|
|
1154
|
+
return int(frontier.value)
|
|
1155
|
+
return frontier.value
|
|
1156
|
+
|
|
1157
|
+
|
|
1158
|
+
def _compare_frontiers(left: SourceFrontier, right: SourceFrontier) -> int | None:
|
|
1159
|
+
if left.kind != right.kind or left.derivation_rule != right.derivation_rule:
|
|
1160
|
+
return None
|
|
1161
|
+
if left.kind in {"none", "opaque"}:
|
|
1162
|
+
return 0 if left.value == right.value else None
|
|
1163
|
+
left_value = _frontier_value(left)
|
|
1164
|
+
right_value = _frontier_value(right)
|
|
1165
|
+
return (right_value > left_value) - (right_value < left_value)
|
|
1166
|
+
|
|
1167
|
+
|
|
1168
|
+
def _median_int(values: Sequence[int]) -> int:
|
|
1169
|
+
ordered = sorted(values)
|
|
1170
|
+
middle = len(ordered) // 2
|
|
1171
|
+
if len(ordered) % 2:
|
|
1172
|
+
return ordered[middle]
|
|
1173
|
+
return (ordered[middle - 1] + ordered[middle]) // 2
|
|
1174
|
+
|
|
1175
|
+
|
|
1176
|
+
def _revision_style(changes: Sequence[EmpiricalSourceObservation]) -> str:
|
|
1177
|
+
if len(changes) < 2:
|
|
1178
|
+
return "unknown"
|
|
1179
|
+
saw_append = False
|
|
1180
|
+
saw_unclassified = False
|
|
1181
|
+
saw_replacement = False
|
|
1182
|
+
for previous, current in pairwise(changes):
|
|
1183
|
+
comparison = _compare_frontiers(previous.frontier, current.frontier)
|
|
1184
|
+
if comparison == 0 and current.content_digest != previous.content_digest:
|
|
1185
|
+
return "historical_revision"
|
|
1186
|
+
if (
|
|
1187
|
+
current.row_count is not None
|
|
1188
|
+
and previous.row_count is not None
|
|
1189
|
+
and current.row_count < previous.row_count
|
|
1190
|
+
):
|
|
1191
|
+
saw_replacement = True
|
|
1192
|
+
previous_partitions = {
|
|
1193
|
+
item.partition_key_digest: item.content_digest for item in previous.partition_digests
|
|
1194
|
+
}
|
|
1195
|
+
current_partitions = {
|
|
1196
|
+
item.partition_key_digest: item.content_digest for item in current.partition_digests
|
|
1197
|
+
}
|
|
1198
|
+
if previous_partitions:
|
|
1199
|
+
prior_rewritten = any(
|
|
1200
|
+
current_partitions.get(key) != digest for key, digest in previous_partitions.items()
|
|
1201
|
+
)
|
|
1202
|
+
if prior_rewritten:
|
|
1203
|
+
saw_replacement = True
|
|
1204
|
+
elif set(current_partitions) > set(previous_partitions):
|
|
1205
|
+
saw_append = True
|
|
1206
|
+
else:
|
|
1207
|
+
saw_unclassified = True
|
|
1208
|
+
else:
|
|
1209
|
+
saw_unclassified = True
|
|
1210
|
+
if saw_replacement:
|
|
1211
|
+
return "replacement"
|
|
1212
|
+
if saw_append and not saw_unclassified:
|
|
1213
|
+
return "append"
|
|
1214
|
+
return "unknown"
|
|
1215
|
+
|
|
1216
|
+
|
|
1217
|
+
def infer_source_cadence(
|
|
1218
|
+
history: Sequence[EmpiricalSourceObservation],
|
|
1219
|
+
) -> CadenceInference:
|
|
1220
|
+
"""Return a deterministic empirical cadence result for one validated history."""
|
|
1221
|
+
|
|
1222
|
+
observations = validate_observation_history(history)
|
|
1223
|
+
if not observations:
|
|
1224
|
+
return CadenceInference(
|
|
1225
|
+
source_id="unknown",
|
|
1226
|
+
inference_epoch="unknown",
|
|
1227
|
+
history_digest=canonical_sha256([]),
|
|
1228
|
+
state="unknown",
|
|
1229
|
+
cadence_kind="unknown",
|
|
1230
|
+
estimated_interval_seconds=None,
|
|
1231
|
+
interval_jitter_seconds=None,
|
|
1232
|
+
publication_delay_min_seconds=None,
|
|
1233
|
+
publication_delay_median_seconds=None,
|
|
1234
|
+
publication_delay_max_seconds=None,
|
|
1235
|
+
expected_next_earliest=None,
|
|
1236
|
+
expected_next_latest=None,
|
|
1237
|
+
confidence_basis_points=0,
|
|
1238
|
+
evidence_count=0,
|
|
1239
|
+
evidence_started_at=None,
|
|
1240
|
+
evidence_completed_at=None,
|
|
1241
|
+
recommended_probe_interval_seconds=21_600,
|
|
1242
|
+
revision_style="unknown",
|
|
1243
|
+
recommended_epoch_reset=False,
|
|
1244
|
+
reason_codes=("no_observations",),
|
|
1245
|
+
)
|
|
1246
|
+
first = observations[0]
|
|
1247
|
+
if first.inference_algorithm_digest != SOURCE_CADENCE_ALGORITHM_DIGEST:
|
|
1248
|
+
_refuse("ALGORITHM", "history", "requires a different inference implementation")
|
|
1249
|
+
history_digest = canonical_sha256([item.to_dict() for item in observations])
|
|
1250
|
+
changes = tuple(item for item in observations if item.result_kind == "changed")
|
|
1251
|
+
publication_coordinates_complete = bool(changes) and all(
|
|
1252
|
+
item.publication_reference_time is not None and item.edition_sequence is not None
|
|
1253
|
+
for item in changes
|
|
1254
|
+
)
|
|
1255
|
+
publication_times = (
|
|
1256
|
+
tuple(
|
|
1257
|
+
_timestamp(item.publication_reference_time, "observation.publication_reference_time")
|
|
1258
|
+
for item in changes
|
|
1259
|
+
)
|
|
1260
|
+
if publication_coordinates_complete
|
|
1261
|
+
else ()
|
|
1262
|
+
)
|
|
1263
|
+
edition_sequences = (
|
|
1264
|
+
tuple(item.edition_sequence for item in changes) if publication_coordinates_complete else ()
|
|
1265
|
+
)
|
|
1266
|
+
interval_values: list[int] = []
|
|
1267
|
+
for (left_time, right_time), (left_sequence, right_sequence) in zip(
|
|
1268
|
+
pairwise(publication_times),
|
|
1269
|
+
pairwise(edition_sequences),
|
|
1270
|
+
strict=True,
|
|
1271
|
+
):
|
|
1272
|
+
assert left_sequence is not None and right_sequence is not None
|
|
1273
|
+
editions = right_sequence - left_sequence
|
|
1274
|
+
seconds = _elapsed_whole_seconds(left_time, right_time)
|
|
1275
|
+
if editions <= 0 or seconds < editions:
|
|
1276
|
+
_refuse(
|
|
1277
|
+
"PUBLICATION_INTERVAL",
|
|
1278
|
+
"history",
|
|
1279
|
+
"cannot derive a positive whole-second interval per edition",
|
|
1280
|
+
)
|
|
1281
|
+
interval_values.append(seconds // editions)
|
|
1282
|
+
intervals = tuple(interval_values)
|
|
1283
|
+
delays_list: list[int] = []
|
|
1284
|
+
for item in changes:
|
|
1285
|
+
if item.publication_reference_time is None:
|
|
1286
|
+
continue
|
|
1287
|
+
completed = _timestamp(item.completed_at, "observation.completed_at")
|
|
1288
|
+
publication_time = _timestamp(
|
|
1289
|
+
item.publication_reference_time,
|
|
1290
|
+
"observation.publication_reference_time",
|
|
1291
|
+
)
|
|
1292
|
+
delays_list.append(_elapsed_whole_seconds(publication_time, completed))
|
|
1293
|
+
delays = tuple(delays_list)
|
|
1294
|
+
interval = _median_int(intervals) if intervals else None
|
|
1295
|
+
jitter = (
|
|
1296
|
+
max((abs(value - interval) for value in intervals), default=None)
|
|
1297
|
+
if interval is not None
|
|
1298
|
+
else None
|
|
1299
|
+
)
|
|
1300
|
+
state = "learning"
|
|
1301
|
+
cadence_kind = "unknown"
|
|
1302
|
+
reset = False
|
|
1303
|
+
reasons: list[str] = []
|
|
1304
|
+
if changes and not publication_coordinates_complete:
|
|
1305
|
+
reasons.append("missing_publication_coordinates")
|
|
1306
|
+
if len(changes) < 4 or not publication_coordinates_complete:
|
|
1307
|
+
reasons.append("insufficient_changes")
|
|
1308
|
+
else:
|
|
1309
|
+
assert interval is not None and jitter is not None
|
|
1310
|
+
allowed_spread = max(3_600, interval * 2_000 // 10_000)
|
|
1311
|
+
if jitter <= allowed_spread:
|
|
1312
|
+
state = "stable"
|
|
1313
|
+
cadence_kind = "regular"
|
|
1314
|
+
reasons.append("regular_intervals")
|
|
1315
|
+
else:
|
|
1316
|
+
state = "changed_pattern"
|
|
1317
|
+
cadence_kind = "irregular"
|
|
1318
|
+
interval = None
|
|
1319
|
+
reasons.append("irregular_intervals")
|
|
1320
|
+
if len(intervals) >= 5:
|
|
1321
|
+
baseline = _median_int(intervals[:-2])
|
|
1322
|
+
recent = intervals[-2:]
|
|
1323
|
+
materially_changed = all(
|
|
1324
|
+
abs(value - baseline) * 10_000 > baseline * 4_000 for value in recent
|
|
1325
|
+
)
|
|
1326
|
+
mutually_consistent = abs(recent[0] - recent[1]) * 10_000 <= max(recent) * 2_000
|
|
1327
|
+
if materially_changed and mutually_consistent:
|
|
1328
|
+
state = "changed_pattern"
|
|
1329
|
+
cadence_kind = "irregular"
|
|
1330
|
+
reset = True
|
|
1331
|
+
interval = None
|
|
1332
|
+
reasons.append("cadence_regime_changed")
|
|
1333
|
+
latest_time = _timestamp(observations[-1].completed_at, "observation.completed_at")
|
|
1334
|
+
regular_interval = _median_int(intervals) if intervals else None
|
|
1335
|
+
if state == "stable" and regular_interval is not None and changes:
|
|
1336
|
+
since_change = _elapsed_whole_seconds(publication_times[-1], latest_time)
|
|
1337
|
+
trailing = tuple(
|
|
1338
|
+
item.result_kind for item in observations[observations.index(changes[-1]) + 1 :]
|
|
1339
|
+
)
|
|
1340
|
+
if (
|
|
1341
|
+
since_change > regular_interval * 4
|
|
1342
|
+
and len(trailing) >= 3
|
|
1343
|
+
and all(kind in {"unchanged", "not_published"} for kind in trailing)
|
|
1344
|
+
):
|
|
1345
|
+
state = "dormant"
|
|
1346
|
+
reasons.append("prolonged_no_change")
|
|
1347
|
+
expected_earliest = expected_latest = None
|
|
1348
|
+
if regular_interval is not None and changes and cadence_kind == "regular":
|
|
1349
|
+
margin = max(300, regular_interval // 20, jitter or 0)
|
|
1350
|
+
center = _safe_shift(
|
|
1351
|
+
publication_times[-1],
|
|
1352
|
+
regular_interval,
|
|
1353
|
+
"inference.expected_window",
|
|
1354
|
+
)
|
|
1355
|
+
earliest = _safe_shift(center, -margin, "inference.expected_next_earliest")
|
|
1356
|
+
latest = _safe_shift(center, margin, "inference.expected_next_latest")
|
|
1357
|
+
if state == "stable" and latest_time <= latest:
|
|
1358
|
+
expected_earliest = _format_timestamp(earliest)
|
|
1359
|
+
expected_latest = _format_timestamp(latest)
|
|
1360
|
+
elif state == "stable":
|
|
1361
|
+
state = "changed_pattern"
|
|
1362
|
+
reasons.append("publication_window_missed")
|
|
1363
|
+
failure_count = sum(item.result_kind == "failed" for item in observations)
|
|
1364
|
+
if failure_count:
|
|
1365
|
+
reasons.append("probe_failures_observed")
|
|
1366
|
+
confidence = 0
|
|
1367
|
+
if state == "learning":
|
|
1368
|
+
confidence = min(4_000, len(changes) * 1_000)
|
|
1369
|
+
elif state == "stable":
|
|
1370
|
+
confidence = min(9_500, 6_000 + max(0, len(intervals) - 3) * 500)
|
|
1371
|
+
elif state == "dormant":
|
|
1372
|
+
confidence = min(8_000, 5_000 + len(intervals) * 300)
|
|
1373
|
+
elif state == "changed_pattern":
|
|
1374
|
+
confidence = 2_500
|
|
1375
|
+
confidence = max(0, confidence - min(3_000, failure_count * 500))
|
|
1376
|
+
if state == "stable" and regular_interval is not None:
|
|
1377
|
+
probe = max(300, min(21_600, regular_interval // 8))
|
|
1378
|
+
elif state == "dormant" and regular_interval is not None:
|
|
1379
|
+
probe = max(86_400, min(7 * 86_400, regular_interval))
|
|
1380
|
+
elif intervals:
|
|
1381
|
+
probe = max(900, min(86_400, _median_int(intervals) // 4))
|
|
1382
|
+
else:
|
|
1383
|
+
probe = 21_600
|
|
1384
|
+
if reset:
|
|
1385
|
+
probe = min(probe, 3_600)
|
|
1386
|
+
delay_min = min(delays) if delays else None
|
|
1387
|
+
delay_median = _median_int(delays) if delays else None
|
|
1388
|
+
delay_max = max(delays) if delays else None
|
|
1389
|
+
return CadenceInference(
|
|
1390
|
+
source_id=first.source_id,
|
|
1391
|
+
inference_epoch=first.inference_epoch,
|
|
1392
|
+
history_digest=history_digest,
|
|
1393
|
+
state=state,
|
|
1394
|
+
cadence_kind=cadence_kind,
|
|
1395
|
+
estimated_interval_seconds=interval,
|
|
1396
|
+
interval_jitter_seconds=jitter,
|
|
1397
|
+
publication_delay_min_seconds=delay_min,
|
|
1398
|
+
publication_delay_median_seconds=delay_median,
|
|
1399
|
+
publication_delay_max_seconds=delay_max,
|
|
1400
|
+
expected_next_earliest=expected_earliest,
|
|
1401
|
+
expected_next_latest=expected_latest,
|
|
1402
|
+
confidence_basis_points=confidence,
|
|
1403
|
+
evidence_count=len(observations),
|
|
1404
|
+
evidence_started_at=observations[0].started_at,
|
|
1405
|
+
evidence_completed_at=observations[-1].completed_at,
|
|
1406
|
+
recommended_probe_interval_seconds=probe,
|
|
1407
|
+
revision_style=_revision_style(changes),
|
|
1408
|
+
recommended_epoch_reset=reset,
|
|
1409
|
+
reason_codes=tuple(reasons),
|
|
1410
|
+
)
|
|
1411
|
+
|
|
1412
|
+
|
|
1413
|
+
__all__ = [
|
|
1414
|
+
"SOURCE_CADENCE_ALGORITHM_DIGEST",
|
|
1415
|
+
"SOURCE_CADENCE_ALGORITHM_VERSION",
|
|
1416
|
+
"SOURCE_CADENCE_INFERENCE_SCHEMA",
|
|
1417
|
+
"SOURCE_CADENCE_OBSERVATION_SCHEMA",
|
|
1418
|
+
"CadenceContractError",
|
|
1419
|
+
"CadenceInference",
|
|
1420
|
+
"EmpiricalSourceObservation",
|
|
1421
|
+
"PartitionDigest",
|
|
1422
|
+
"SourceFrontier",
|
|
1423
|
+
"infer_source_cadence",
|
|
1424
|
+
"parse_cadence_inference",
|
|
1425
|
+
"parse_empirical_source_observation",
|
|
1426
|
+
"source_cadence_algorithm_spec",
|
|
1427
|
+
"validate_observation_history",
|
|
1428
|
+
]
|