mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,1428 @@
1
+ """Canonical empirical source observations and deterministic cadence inference.
2
+
3
+ The existing :class:`SourceObservation` records researched/catalogue fitness. This module owns a
4
+ different fact: what one governed connector actually observed over repeated headless probes. The
5
+ two contracts deliberately have different names and schema coordinates.
6
+
7
+ Inference consumes only the exact digest-bound ordered history passed by the caller. This module
8
+ does not grant that history authority; the future Studio consumer must authenticate who recorded
9
+ each observation before persisting it. The pure calculation never reads a clock, network,
10
+ environment variable, mutable file, or model service, so every consumer can replay the same
11
+ history and compare its result with the packaged golden vectors.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ import unicodedata
18
+ from collections.abc import Mapping, Sequence
19
+ from copy import deepcopy
20
+ from dataclasses import dataclass
21
+ from datetime import UTC, datetime, timedelta
22
+ from hashlib import sha256
23
+ from itertools import pairwise
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from mostlyright.data_harness import canonical as canonical_contract
28
+ from mostlyright.data_harness.canonical import canonical_sha256
29
+
30
+ SOURCE_CADENCE_OBSERVATION_SCHEMA = "harness-empirical-source-observation.v1"
31
+ SOURCE_CADENCE_INFERENCE_SCHEMA = "harness-source-cadence-inference.v1"
32
+ SOURCE_CADENCE_ALGORITHM_VERSION = "source-cadence.v1"
33
+ _IMPLEMENTATION_SOURCE_SHA256 = sha256(Path(__file__).read_bytes()).hexdigest()
34
+ _CANONICAL_SOURCE_SHA256 = sha256(Path(canonical_contract.__file__).read_bytes()).hexdigest()
35
+ _FAILURE_CODE_VALUES = (
36
+ "contract_refused",
37
+ "http_4xx",
38
+ "http_5xx",
39
+ "rate_limited",
40
+ "reader_failure",
41
+ "source_unavailable",
42
+ "timeout",
43
+ "tls_failure",
44
+ "transport_failure",
45
+ "unknown_failure",
46
+ )
47
+ _SOURCE_CADENCE_ALGORITHM_SPEC: dict[str, object] = {
48
+ "version": SOURCE_CADENCE_ALGORITHM_VERSION,
49
+ "implementation_source_sha256": _IMPLEMENTATION_SOURCE_SHA256,
50
+ "canonicalization_source_sha256": _CANONICAL_SOURCE_SHA256,
51
+ "integer_arithmetic": {
52
+ "time_delta_seconds": "exact_utc_timestamp_difference_as_integer_seconds",
53
+ "per_edition_interval": "delta_seconds_floor_divided_by_positive_edition_delta",
54
+ "basis_points": "integer_multiply_then_floor_divide_by_10000",
55
+ "median": "sorted_middle_or_floor_of_two_middle_sum_divided_by_two",
56
+ },
57
+ "history": {
58
+ "maximum_observations": 4_096,
59
+ "maximum_partitions_per_observation": 128,
60
+ "maximum_partitions_per_history": 8_192,
61
+ "identity": [
62
+ "source",
63
+ "epoch",
64
+ "connector_configuration",
65
+ "locator",
66
+ "recipe",
67
+ "source_authority",
68
+ "reader_coordinate",
69
+ "reader_options",
70
+ "frontier_policy",
71
+ "time_interpretation",
72
+ "algorithm",
73
+ "acquired_frontier_kind_and_derivation",
74
+ "publication_coordinate_presence",
75
+ ],
76
+ "ordering": "contiguous_sequence_predecessor_digest_nonoverlapping_probe_time",
77
+ "material_order": "strict_publication_reference_and_edition_sequence_when_present",
78
+ },
79
+ "result_shapes": {
80
+ "changed": "complete_acquired_evidence_and_optional_complete_publication_coordinate",
81
+ "unchanged_acquired": "complete_acquired_evidence_equal_to_prior_acquired_facts",
82
+ "unchanged_no_body": "http_304_conditional_request_and_probe_receipt_no_parsed_facts",
83
+ "not_published": "probe_receipt_and_status_none_or_200_202_204_404_no_data_facts",
84
+ "failed": {
85
+ "shape": "closed_failure_code_and_probe_receipt_no_data_facts",
86
+ "failure_codes": list(_FAILURE_CODE_VALUES),
87
+ },
88
+ },
89
+ "interval": {
90
+ "source": "publication_reference_time_divided_by_edition_sequence_delta",
91
+ "minimum_changed_observations": 4,
92
+ "median": "sorted_integer_midpoint_floor",
93
+ "jitter": "maximum_absolute_distance_from_median_interval",
94
+ "regular_jitter_floor_seconds": 3_600,
95
+ "regular_jitter_basis_points": 2_000,
96
+ "regime_minimum_intervals": 5,
97
+ "regime_recent_intervals": 2,
98
+ "regime_change_basis_points": 4_000,
99
+ "regime_recent_consistency_basis_points": 2_000,
100
+ },
101
+ "dormancy": {
102
+ "elapsed_interval_multiplier": 4,
103
+ "minimum_successful_no_change_observations": 3,
104
+ "failures_disqualify_trailing_evidence": True,
105
+ },
106
+ "expected_window": {
107
+ "center": "last_publication_reference_plus_interval",
108
+ "minimum_margin_seconds": 300,
109
+ "interval_margin_basis_points": 500,
110
+ "include_only_if_latest_observation_not_after_window": True,
111
+ },
112
+ "confidence": {
113
+ "learning_per_change_basis_points": 1_000,
114
+ "learning_maximum_basis_points": 4_000,
115
+ "stable_base_basis_points": 6_000,
116
+ "stable_extra_interval_basis_points": 500,
117
+ "stable_maximum_basis_points": 9_500,
118
+ "dormant_base_basis_points": 5_000,
119
+ "dormant_per_interval_basis_points": 300,
120
+ "dormant_maximum_basis_points": 8_000,
121
+ "changed_pattern_basis_points": 2_500,
122
+ "failure_penalty_each_basis_points": 500,
123
+ "failure_penalty_maximum_basis_points": 3_000,
124
+ },
125
+ "probe_recommendation": {
126
+ "stable_divisor": 8,
127
+ "stable_maximum_seconds": 21_600,
128
+ "dormant_minimum_seconds": 86_400,
129
+ "dormant_maximum_seconds": 604_800,
130
+ "unstable_divisor": 4,
131
+ "unstable_minimum_seconds": 900,
132
+ "unstable_maximum_seconds": 86_400,
133
+ "unknown_seconds": 21_600,
134
+ "epoch_reset_maximum_seconds": 3_600,
135
+ "absolute_minimum_seconds": 300,
136
+ "absolute_maximum_seconds": 2_592_000,
137
+ },
138
+ "revision": {
139
+ "historical_revision": "changed_content_without_frontier_advance",
140
+ "replacement": "row_count_falls_or_prior_partition_missing_or_rewritten",
141
+ "append": "all_prior_partitions_unchanged_and_at_least_one_new_partition",
142
+ "fallback": "unknown",
143
+ },
144
+ "publication_delay": "observation_completion_minus_publication_reference_integer_seconds",
145
+ "evaluation_time": "last_observation_completed_at",
146
+ "state_precedence": [
147
+ "learning_until_four_coordinate_complete_material_changes",
148
+ "stable_if_jitter_within_threshold_else_changed_pattern_irregular",
149
+ "regime_change_overrides_regular_or_irregular_to_changed_pattern_and_epoch_reset",
150
+ "dormancy_overrides_stable_only_with_trailing_successful_no_change_evidence",
151
+ "missed_expected_window_overrides_remaining_stable_to_changed_pattern_regular",
152
+ "failure_penalty_and_reason_apply_after_state_selection",
153
+ ],
154
+ "output_formulas": {
155
+ "interval_jitter_seconds": "max_abs_each_interval_minus_median",
156
+ "publication_delay_summary": "minimum_floor_median_maximum",
157
+ "expected_window_margin": "max_300_interval_floor_div_20_jitter",
158
+ "confidence": "state_base_plus_interval_increment_then_failure_penalty_clamped_at_zero",
159
+ "evidence_range": "first_started_at_through_last_completed_at",
160
+ "history_digest": "canonical_sha256_of_exact_observation_documents_in_order",
161
+ "reason_order": "pattern_then_regime_then_dormancy_or_missed_window_then_failures",
162
+ "revision_precedence": (
163
+ "historical_revision_then_replacement_then_all_transitions_append_else_unknown"
164
+ ),
165
+ },
166
+ "reason_codes": [
167
+ "no_observations",
168
+ "missing_publication_coordinates",
169
+ "insufficient_changes",
170
+ "regular_intervals",
171
+ "irregular_intervals",
172
+ "cadence_regime_changed",
173
+ "publication_window_missed",
174
+ "prolonged_no_change",
175
+ "probe_failures_observed",
176
+ ],
177
+ }
178
+ SOURCE_CADENCE_ALGORITHM_DIGEST = canonical_sha256(_SOURCE_CADENCE_ALGORITHM_SPEC)
179
+
180
+ _DIGEST = re.compile(r"^[0-9a-f]{64}$")
181
+ _IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$")
182
+ _READER_IDENTIFIER = re.compile(r"^[a-z][a-z0-9]*(?:[._-][a-z0-9]+)*$")
183
+ _SEMVER = re.compile(r"^[0-9]+\.[0-9]+\.[0-9]+$")
184
+ _TIMESTAMP = re.compile(
185
+ r"^(?P<date>[0-9]{4}-[0-9]{2}-[0-9]{2})T"
186
+ r"(?P<time>[0-9]{2}:[0-9]{2}:[0-9]{2})(?P<fraction>\.[0-9]{1,6})?Z$"
187
+ )
188
+ _HTTP_STATUS_CLASSES = frozenset({"none", "2xx", "3xx", "4xx", "5xx"})
189
+ _RESULT_KINDS = frozenset({"changed", "unchanged", "not_published", "failed"})
190
+ _FRONTIER_KINDS = frozenset({"none", "date", "timestamp", "integer", "opaque"})
191
+ _INFERENCE_STATES = frozenset({"unknown", "learning", "stable", "changed_pattern", "dormant"})
192
+ _CADENCE_KINDS = frozenset({"unknown", "regular", "irregular"})
193
+ _REVISION_STYLES = frozenset({"unknown", "append", "replacement", "historical_revision"})
194
+ _FAILURE_CODES = frozenset(_FAILURE_CODE_VALUES)
195
+
196
+
197
+ class CadenceContractError(ValueError):
198
+ """Typed refusal raised by observation, history, or inference validation."""
199
+
200
+ def __init__(self, code: str, path: str, detail: str) -> None:
201
+ self.code = code
202
+ self.path = path
203
+ self.detail = detail
204
+ super().__init__(f"{path}: {detail} [{code}]")
205
+
206
+
207
+ def source_cadence_algorithm_spec() -> dict[str, object]:
208
+ """Return an isolated copy of the complete normative algorithm manifest."""
209
+
210
+ return deepcopy(_SOURCE_CADENCE_ALGORITHM_SPEC)
211
+
212
+
213
+ def _refuse(code: str, path: str, detail: str) -> None:
214
+ raise CadenceContractError(code, path, detail)
215
+
216
+
217
+ def _digest(value: object, path: str, *, optional: bool = False) -> None:
218
+ if value is None and optional:
219
+ return
220
+ if not isinstance(value, str) or _DIGEST.fullmatch(value) is None:
221
+ _refuse("DIGEST", path, "must be a lowercase SHA-256 digest")
222
+
223
+
224
+ def _identifier(value: object, path: str) -> None:
225
+ if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) is None:
226
+ _refuse("IDENTIFIER", path, "must be a bounded source identifier")
227
+
228
+
229
+ def _reader_identifier(value: object, path: str) -> None:
230
+ if not isinstance(value, str) or len(value) > 64 or _READER_IDENTIFIER.fullmatch(value) is None:
231
+ _refuse("IDENTIFIER", path, "must be a canonical Reader identifier")
232
+
233
+
234
+ def _timestamp(value: object, path: str) -> datetime:
235
+ if not isinstance(value, str) or _TIMESTAMP.fullmatch(value) is None:
236
+ _refuse("TIMESTAMP", path, "must be a canonical UTC timestamp ending in Z")
237
+ try:
238
+ parsed = datetime.fromisoformat(value[:-1] + "+00:00")
239
+ except ValueError:
240
+ _refuse("TIMESTAMP", path, "must be a real calendar timestamp")
241
+ parsed = parsed.astimezone(UTC)
242
+ if _format_timestamp(parsed) != value:
243
+ _refuse("TIMESTAMP", path, "must use the one canonical UTC spelling")
244
+ return parsed
245
+
246
+
247
+ def _format_timestamp(value: datetime) -> str:
248
+ value = value.astimezone(UTC)
249
+ if value.microsecond:
250
+ return value.isoformat(timespec="microseconds").replace("+00:00", "Z")
251
+ return value.isoformat(timespec="seconds").replace("+00:00", "Z")
252
+
253
+
254
+ def _bounded_text(value: object, path: str, *, maximum: int) -> None:
255
+ if not isinstance(value, str) or not 1 <= len(value) <= maximum or value != value.strip():
256
+ _refuse("TEXT", path, f"must be stripped text of 1..{maximum} characters")
257
+ if any(
258
+ not character.isprintable() or unicodedata.category(character) in {"Zl", "Zp"}
259
+ for character in value
260
+ ):
261
+ _refuse("TEXT", path, "must contain only display-safe characters")
262
+
263
+
264
+ def _exact(value: Mapping[str, Any], expected: set[str], path: str) -> None:
265
+ if set(value) != expected:
266
+ _refuse("EXACT_KEYS", path, "does not have the exact versioned key set")
267
+
268
+
269
+ def _optional_int(value: object, path: str, *, minimum: int, maximum: int) -> None:
270
+ if value is None:
271
+ return
272
+ if type(value) is not int or not minimum <= value <= maximum:
273
+ _refuse("INTEGER", path, f"must be null or an integer in {minimum}..{maximum}")
274
+
275
+
276
+ def _safe_shift(value: datetime, seconds: int, path: str) -> datetime:
277
+ try:
278
+ return value + timedelta(seconds=seconds)
279
+ except OverflowError:
280
+ _refuse("TIMESTAMP_RANGE", path, "cannot represent the inferred time")
281
+
282
+
283
+ def _elapsed_whole_seconds(earlier: datetime, later: datetime) -> int:
284
+ """Return exact floor whole seconds without floating-point conversion."""
285
+
286
+ delta = later - earlier
287
+ return delta.days * 86_400 + delta.seconds
288
+
289
+
290
+ @dataclass(frozen=True)
291
+ class SourceFrontier:
292
+ """One approved, typed source frontier and its public derivation rule."""
293
+
294
+ kind: str
295
+ value: str | None
296
+ derivation_rule: str
297
+
298
+ def __post_init__(self) -> None:
299
+ if not isinstance(self.kind, str) or self.kind not in _FRONTIER_KINDS:
300
+ _refuse("FRONTIER_KIND", "frontier.kind", "is not a supported frontier kind")
301
+ _bounded_text(self.derivation_rule, "frontier.derivation_rule", maximum=128)
302
+ if self.kind == "none":
303
+ if self.value is not None:
304
+ _refuse("FRONTIER_VALUE", "frontier.value", "must be null for kind none")
305
+ return
306
+ _bounded_text(self.value, "frontier.value", maximum=256)
307
+ if self.kind == "date":
308
+ try:
309
+ parsed_date = datetime.strptime(self.value, "%Y-%m-%d")
310
+ except ValueError:
311
+ _refuse("FRONTIER_VALUE", "frontier.value", "must be a real ISO date")
312
+ if parsed_date.strftime("%Y-%m-%d") != self.value:
313
+ _refuse("FRONTIER_VALUE", "frontier.value", "must use canonical ISO spelling")
314
+ elif self.kind == "timestamp":
315
+ _timestamp(self.value, "frontier.value")
316
+ elif self.kind == "integer":
317
+ if not re.fullmatch(r"0|[1-9][0-9]{0,18}", self.value):
318
+ _refuse("FRONTIER_VALUE", "frontier.value", "must be a canonical unsigned integer")
319
+
320
+ def to_dict(self) -> dict[str, Any]:
321
+ return {"kind": self.kind, "value": self.value, "derivation_rule": self.derivation_rule}
322
+
323
+
324
+ @dataclass(frozen=True)
325
+ class PartitionDigest:
326
+ """Credential-free identity and content digest for one approved source partition."""
327
+
328
+ partition_key_digest: str
329
+ content_digest: str
330
+
331
+ def __post_init__(self) -> None:
332
+ _digest(self.partition_key_digest, "partition.partition_key_digest")
333
+ _digest(self.content_digest, "partition.content_digest")
334
+
335
+ def to_dict(self) -> dict[str, str]:
336
+ return {
337
+ "partition_key_digest": self.partition_key_digest,
338
+ "content_digest": self.content_digest,
339
+ }
340
+
341
+
342
+ @dataclass(frozen=True)
343
+ class EmpiricalSourceObservation:
344
+ """Immutable result of one bounded connector probe or acquisition."""
345
+
346
+ source_id: str
347
+ sequence: int
348
+ inference_epoch: str
349
+ previous_observation_digest: str | None
350
+ connector_configuration_digest: str
351
+ locator_digest: str
352
+ recipe_digest: str
353
+ source_authority_digest: str
354
+ reader_family_id: str
355
+ reader_family_version: str
356
+ reader_options_digest: str
357
+ frontier_policy_digest: str
358
+ time_interpretation_digest: str
359
+ inference_algorithm_digest: str
360
+ started_at: str
361
+ completed_at: str
362
+ result_kind: str
363
+ failure_code: str | None
364
+ http_status_class: str
365
+ http_status_code: int | None
366
+ conditional_request_digest: str | None
367
+ probe_receipt_digest: str
368
+ etag_digest: str | None
369
+ last_modified_digest: str | None
370
+ acquired_size_bytes: int | None
371
+ content_digest: str | None
372
+ row_digest: str | None
373
+ schema_digest: str | None
374
+ row_count: int | None
375
+ frontier: SourceFrontier
376
+ max_event_time: str | None
377
+ partition_digests: tuple[PartitionDigest, ...]
378
+ acquisition_receipt_digest: str | None
379
+ next_watermark_candidate_digest: str | None
380
+ publication_reference_time: str | None
381
+ edition_sequence: int | None
382
+ schema_version: str = SOURCE_CADENCE_OBSERVATION_SCHEMA
383
+
384
+ def __post_init__(self) -> None:
385
+ if self.schema_version != SOURCE_CADENCE_OBSERVATION_SCHEMA:
386
+ _refuse("SCHEMA_VERSION", "observation.schema_version", "is not supported")
387
+ _identifier(self.source_id, "observation.source_id")
388
+ if type(self.sequence) is not int or not 0 <= self.sequence <= (1 << 53) - 1:
389
+ _refuse("SEQUENCE", "observation.sequence", "must be a non-negative safe integer")
390
+ _identifier(self.inference_epoch, "observation.inference_epoch")
391
+ _digest(
392
+ self.previous_observation_digest,
393
+ "observation.previous_observation_digest",
394
+ optional=True,
395
+ )
396
+ if (self.sequence == 0) != (self.previous_observation_digest is None):
397
+ _refuse(
398
+ "PREDECESSOR",
399
+ "observation.previous_observation_digest",
400
+ "must be null exactly for sequence zero",
401
+ )
402
+ for name, value in (
403
+ ("connector_configuration_digest", self.connector_configuration_digest),
404
+ ("locator_digest", self.locator_digest),
405
+ ("recipe_digest", self.recipe_digest),
406
+ ("source_authority_digest", self.source_authority_digest),
407
+ ("reader_options_digest", self.reader_options_digest),
408
+ ("frontier_policy_digest", self.frontier_policy_digest),
409
+ ("time_interpretation_digest", self.time_interpretation_digest),
410
+ ("inference_algorithm_digest", self.inference_algorithm_digest),
411
+ ("probe_receipt_digest", self.probe_receipt_digest),
412
+ ):
413
+ _digest(value, f"observation.{name}")
414
+ _reader_identifier(self.reader_family_id, "observation.reader_family_id")
415
+ if (
416
+ not isinstance(self.reader_family_version, str)
417
+ or len(self.reader_family_version) > 32
418
+ or _SEMVER.fullmatch(self.reader_family_version) is None
419
+ ):
420
+ _refuse("SEMVER", "observation.reader_family_version", "must be x.y.z")
421
+ started = _timestamp(self.started_at, "observation.started_at")
422
+ completed = _timestamp(self.completed_at, "observation.completed_at")
423
+ if completed < started:
424
+ _refuse("TEMPORAL_ORDER", "observation.completed_at", "precedes started_at")
425
+ if not isinstance(self.result_kind, str) or self.result_kind not in _RESULT_KINDS:
426
+ _refuse("RESULT_KIND", "observation.result_kind", "is not supported")
427
+ if (
428
+ not isinstance(self.http_status_class, str)
429
+ or self.http_status_class not in _HTTP_STATUS_CLASSES
430
+ ):
431
+ _refuse("HTTP_STATUS_CLASS", "observation.http_status_class", "is not supported")
432
+ _optional_int(
433
+ self.http_status_code,
434
+ "observation.http_status_code",
435
+ minimum=100,
436
+ maximum=599,
437
+ )
438
+ if self.http_status_class == "none":
439
+ if self.http_status_code is not None:
440
+ _refuse("HTTP_STATUS", "observation.http_status_code", "must be null")
441
+ elif (
442
+ self.http_status_code is None
443
+ or f"{self.http_status_code // 100}xx" != self.http_status_class
444
+ ):
445
+ _refuse("HTTP_STATUS", "observation.http_status_code", "does not match its class")
446
+ for name, value in (
447
+ ("conditional_request_digest", self.conditional_request_digest),
448
+ ("etag_digest", self.etag_digest),
449
+ ("last_modified_digest", self.last_modified_digest),
450
+ ("content_digest", self.content_digest),
451
+ ("row_digest", self.row_digest),
452
+ ("schema_digest", self.schema_digest),
453
+ ("acquisition_receipt_digest", self.acquisition_receipt_digest),
454
+ ("next_watermark_candidate_digest", self.next_watermark_candidate_digest),
455
+ ):
456
+ _digest(value, f"observation.{name}", optional=True)
457
+ _optional_int(
458
+ self.acquired_size_bytes,
459
+ "observation.acquired_size_bytes",
460
+ minimum=0,
461
+ maximum=1 << 40,
462
+ )
463
+ _optional_int(self.row_count, "observation.row_count", minimum=0, maximum=1_000_000_000)
464
+ _optional_int(
465
+ self.edition_sequence,
466
+ "observation.edition_sequence",
467
+ minimum=0,
468
+ maximum=(1 << 53) - 1,
469
+ )
470
+ if self.publication_reference_time is not None:
471
+ publication_reference = _timestamp(
472
+ self.publication_reference_time,
473
+ "observation.publication_reference_time",
474
+ )
475
+ if publication_reference > completed:
476
+ _refuse(
477
+ "PUBLICATION_TIME",
478
+ "observation.publication_reference_time",
479
+ "follows observation completion",
480
+ )
481
+ if (self.publication_reference_time is None) != (self.edition_sequence is None):
482
+ _refuse(
483
+ "PUBLICATION_COORDINATE",
484
+ "observation",
485
+ "publication time and edition sequence must appear together",
486
+ )
487
+ if not isinstance(self.frontier, SourceFrontier):
488
+ _refuse("TYPE", "observation.frontier", "must be a SourceFrontier")
489
+ if self.max_event_time is not None:
490
+ _timestamp(self.max_event_time, "observation.max_event_time")
491
+ if not isinstance(self.partition_digests, tuple) or len(self.partition_digests) > 128:
492
+ _refuse("PARTITIONS", "observation.partition_digests", "must be a bounded tuple")
493
+ if any(not isinstance(item, PartitionDigest) for item in self.partition_digests):
494
+ _refuse("PARTITIONS", "observation.partition_digests", "contains an invalid item")
495
+ keys = tuple(item.partition_key_digest for item in self.partition_digests)
496
+ if keys != tuple(sorted(keys)) or len(keys) != len(set(keys)):
497
+ _refuse(
498
+ "PARTITIONS",
499
+ "observation.partition_digests",
500
+ "must be unique and sorted by partition_key_digest",
501
+ )
502
+ acquired = (
503
+ self.acquired_size_bytes,
504
+ self.content_digest,
505
+ self.row_digest,
506
+ self.schema_digest,
507
+ self.row_count,
508
+ self.acquisition_receipt_digest,
509
+ )
510
+ if self.result_kind == "changed":
511
+ if self.failure_code is not None or any(value is None for value in acquired):
512
+ _refuse(
513
+ "RESULT_SHAPE",
514
+ "observation",
515
+ "changed requires complete acquired evidence",
516
+ )
517
+ if self.http_status_class not in {"none", "2xx"}:
518
+ _refuse("RESULT_STATUS", "observation", "changed requires a successful result")
519
+ if self.probe_receipt_digest != self.acquisition_receipt_digest:
520
+ _refuse(
521
+ "RESULT_SHAPE",
522
+ "observation.probe_receipt_digest",
523
+ "must be the acquired Receipt digest",
524
+ )
525
+ elif self.result_kind == "unchanged":
526
+ no_acquired_evidence = all(value is None for value in acquired)
527
+ complete_acquired_evidence = all(value is not None for value in acquired)
528
+ if self.failure_code is not None or not (
529
+ no_acquired_evidence or complete_acquired_evidence
530
+ ):
531
+ _refuse(
532
+ "RESULT_SHAPE",
533
+ "observation",
534
+ "unchanged requires complete evidence or a conditional no-body result",
535
+ )
536
+ if no_acquired_evidence and self.next_watermark_candidate_digest is not None:
537
+ _refuse(
538
+ "RESULT_SHAPE",
539
+ "observation.next_watermark_candidate_digest",
540
+ "cannot advance without acquired evidence",
541
+ )
542
+ if no_acquired_evidence:
543
+ if (
544
+ self.http_status_code != 304
545
+ or self.conditional_request_digest is None
546
+ or self.frontier.kind != "none"
547
+ or self.max_event_time is not None
548
+ or self.partition_digests
549
+ or self.publication_reference_time is not None
550
+ ):
551
+ _refuse(
552
+ "RESULT_SHAPE",
553
+ "observation",
554
+ "no-body unchanged requires a conditional 304 with no parsed facts",
555
+ )
556
+ else:
557
+ if self.http_status_class not in {"none", "2xx"}:
558
+ _refuse(
559
+ "RESULT_STATUS",
560
+ "observation",
561
+ "acquired unchanged requires a successful result",
562
+ )
563
+ if self.probe_receipt_digest != self.acquisition_receipt_digest:
564
+ _refuse(
565
+ "RESULT_SHAPE",
566
+ "observation.probe_receipt_digest",
567
+ "must be the acquired Receipt digest",
568
+ )
569
+ elif self.result_kind in {"not_published", "failed"}:
570
+ if any(value is not None for value in acquired) or self.frontier.kind != "none":
571
+ _refuse("RESULT_SHAPE", "observation", "result cannot claim acquired evidence")
572
+ if self.next_watermark_candidate_digest is not None:
573
+ _refuse(
574
+ "RESULT_SHAPE",
575
+ "observation.next_watermark_candidate_digest",
576
+ "cannot advance without acquired evidence",
577
+ )
578
+ if self.partition_digests or self.max_event_time is not None:
579
+ _refuse("RESULT_SHAPE", "observation", "result cannot claim parsed evidence")
580
+ if self.publication_reference_time is not None:
581
+ _refuse("RESULT_SHAPE", "observation", "result cannot claim a publication")
582
+ if self.result_kind == "not_published" and self.http_status_code not in {
583
+ None,
584
+ 200,
585
+ 202,
586
+ 204,
587
+ 404,
588
+ }:
589
+ _refuse(
590
+ "RESULT_STATUS",
591
+ "observation.http_status_code",
592
+ "is not a successful not-published outcome",
593
+ )
594
+ if self.result_kind == "failed":
595
+ if not isinstance(self.failure_code, str) or self.failure_code not in _FAILURE_CODES:
596
+ _refuse("FAILURE_CODE", "observation.failure_code", "is not supported")
597
+ elif self.failure_code is not None:
598
+ _refuse("RESULT_SHAPE", "observation.failure_code", "must be null unless failed")
599
+
600
+ @property
601
+ def digest(self) -> str:
602
+ return canonical_sha256(self.to_dict())
603
+
604
+ def to_dict(self) -> dict[str, Any]:
605
+ return {
606
+ "schema_version": self.schema_version,
607
+ "source_id": self.source_id,
608
+ "sequence": self.sequence,
609
+ "inference_epoch": self.inference_epoch,
610
+ "previous_observation_digest": self.previous_observation_digest,
611
+ "connector_configuration_digest": self.connector_configuration_digest,
612
+ "locator_digest": self.locator_digest,
613
+ "recipe_digest": self.recipe_digest,
614
+ "source_authority_digest": self.source_authority_digest,
615
+ "reader_family_id": self.reader_family_id,
616
+ "reader_family_version": self.reader_family_version,
617
+ "reader_options_digest": self.reader_options_digest,
618
+ "frontier_policy_digest": self.frontier_policy_digest,
619
+ "time_interpretation_digest": self.time_interpretation_digest,
620
+ "inference_algorithm_digest": self.inference_algorithm_digest,
621
+ "started_at": self.started_at,
622
+ "completed_at": self.completed_at,
623
+ "result_kind": self.result_kind,
624
+ "failure_code": self.failure_code,
625
+ "http_status_class": self.http_status_class,
626
+ "http_status_code": self.http_status_code,
627
+ "conditional_request_digest": self.conditional_request_digest,
628
+ "probe_receipt_digest": self.probe_receipt_digest,
629
+ "etag_digest": self.etag_digest,
630
+ "last_modified_digest": self.last_modified_digest,
631
+ "acquired_size_bytes": self.acquired_size_bytes,
632
+ "content_digest": self.content_digest,
633
+ "row_digest": self.row_digest,
634
+ "schema_digest": self.schema_digest,
635
+ "row_count": self.row_count,
636
+ "frontier": self.frontier.to_dict(),
637
+ "max_event_time": self.max_event_time,
638
+ "partition_digests": [item.to_dict() for item in self.partition_digests],
639
+ "acquisition_receipt_digest": self.acquisition_receipt_digest,
640
+ "next_watermark_candidate_digest": self.next_watermark_candidate_digest,
641
+ "publication_reference_time": self.publication_reference_time,
642
+ "edition_sequence": self.edition_sequence,
643
+ }
644
+
645
+
646
+ @dataclass(frozen=True)
647
+ class CadenceInference:
648
+ """Pure replay result for one exact empirical observation history."""
649
+
650
+ source_id: str
651
+ inference_epoch: str
652
+ history_digest: str
653
+ state: str
654
+ cadence_kind: str
655
+ estimated_interval_seconds: int | None
656
+ interval_jitter_seconds: int | None
657
+ publication_delay_min_seconds: int | None
658
+ publication_delay_median_seconds: int | None
659
+ publication_delay_max_seconds: int | None
660
+ expected_next_earliest: str | None
661
+ expected_next_latest: str | None
662
+ confidence_basis_points: int
663
+ evidence_count: int
664
+ evidence_started_at: str | None
665
+ evidence_completed_at: str | None
666
+ recommended_probe_interval_seconds: int
667
+ revision_style: str
668
+ recommended_epoch_reset: bool
669
+ reason_codes: tuple[str, ...]
670
+ algorithm_version: str = SOURCE_CADENCE_ALGORITHM_VERSION
671
+ algorithm_digest: str = SOURCE_CADENCE_ALGORITHM_DIGEST
672
+ schema_version: str = SOURCE_CADENCE_INFERENCE_SCHEMA
673
+
674
+ def __post_init__(self) -> None:
675
+ if self.schema_version != SOURCE_CADENCE_INFERENCE_SCHEMA:
676
+ _refuse("SCHEMA_VERSION", "inference.schema_version", "is not supported")
677
+ if (
678
+ self.algorithm_version != SOURCE_CADENCE_ALGORITHM_VERSION
679
+ or self.algorithm_digest != SOURCE_CADENCE_ALGORITHM_DIGEST
680
+ ):
681
+ _refuse("ALGORITHM", "inference.algorithm_version", "is not the exact algorithm")
682
+ _identifier(self.source_id, "inference.source_id")
683
+ _identifier(self.inference_epoch, "inference.inference_epoch")
684
+ _digest(self.history_digest, "inference.history_digest")
685
+ if (
686
+ not isinstance(self.state, str)
687
+ or self.state not in _INFERENCE_STATES
688
+ or not isinstance(self.cadence_kind, str)
689
+ or self.cadence_kind not in _CADENCE_KINDS
690
+ ):
691
+ _refuse("INFERENCE_STATE", "inference.state", "is not supported")
692
+ allowed_cadence = {
693
+ "unknown": {"unknown"},
694
+ "learning": {"unknown"},
695
+ "stable": {"regular"},
696
+ "dormant": {"regular"},
697
+ "changed_pattern": {"regular", "irregular"},
698
+ }
699
+ if self.cadence_kind not in allowed_cadence[self.state]:
700
+ _refuse("INFERENCE_STATE", "inference.cadence_kind", "contradicts the state")
701
+ if not isinstance(self.revision_style, str) or self.revision_style not in _REVISION_STYLES:
702
+ _refuse("REVISION_STYLE", "inference.revision_style", "is not supported")
703
+ _optional_int(
704
+ self.estimated_interval_seconds,
705
+ "inference.estimated_interval_seconds",
706
+ minimum=1,
707
+ maximum=1 << 40,
708
+ )
709
+ for name, value in (
710
+ ("interval_jitter_seconds", self.interval_jitter_seconds),
711
+ ("publication_delay_min_seconds", self.publication_delay_min_seconds),
712
+ ("publication_delay_median_seconds", self.publication_delay_median_seconds),
713
+ ("publication_delay_max_seconds", self.publication_delay_max_seconds),
714
+ ):
715
+ _optional_int(value, f"inference.{name}", minimum=0, maximum=1 << 40)
716
+ if self.cadence_kind == "regular" and self.estimated_interval_seconds is None:
717
+ _refuse("INFERENCE_STATE", "inference.estimated_interval_seconds", "is required")
718
+ if self.cadence_kind == "irregular" and self.estimated_interval_seconds is not None:
719
+ _refuse("INFERENCE_STATE", "inference.estimated_interval_seconds", "must be null")
720
+ delays = (
721
+ self.publication_delay_min_seconds,
722
+ self.publication_delay_median_seconds,
723
+ self.publication_delay_max_seconds,
724
+ )
725
+ if any(value is None for value in delays) and not all(value is None for value in delays):
726
+ _refuse("PUBLICATION_DELAY", "inference", "delay values must appear together")
727
+ if all(value is not None for value in delays):
728
+ delay_min, delay_median, delay_max = delays
729
+ assert delay_min is not None and delay_median is not None and delay_max is not None
730
+ if not delay_min <= delay_median <= delay_max:
731
+ _refuse("PUBLICATION_DELAY", "inference", "delay values are out of order")
732
+ for name, value in (
733
+ ("expected_next_earliest", self.expected_next_earliest),
734
+ ("expected_next_latest", self.expected_next_latest),
735
+ ("evidence_started_at", self.evidence_started_at),
736
+ ("evidence_completed_at", self.evidence_completed_at),
737
+ ):
738
+ if value is not None:
739
+ _timestamp(value, f"inference.{name}")
740
+ if (self.expected_next_earliest is None) != (self.expected_next_latest is None):
741
+ _refuse(
742
+ "EXPECTED_WINDOW",
743
+ "inference",
744
+ "expected window endpoints must appear together",
745
+ )
746
+ if self.expected_next_earliest is not None and _timestamp(
747
+ self.expected_next_latest, "inference.expected_next_latest"
748
+ ) < _timestamp(self.expected_next_earliest, "inference.expected_next_earliest"):
749
+ _refuse("EXPECTED_WINDOW", "inference.expected_next_latest", "precedes earliest")
750
+ if self.expected_next_earliest is not None and self.state != "stable":
751
+ _refuse("EXPECTED_WINDOW", "inference", "is available only for stable cadence")
752
+ if (
753
+ type(self.confidence_basis_points) is not int
754
+ or not 0 <= self.confidence_basis_points <= 10_000
755
+ ):
756
+ _refuse("CONFIDENCE", "inference.confidence_basis_points", "must be 0..10000")
757
+ if type(self.evidence_count) is not int or not 0 <= self.evidence_count <= 4_096:
758
+ _refuse("EVIDENCE_COUNT", "inference.evidence_count", "must be 0..4096")
759
+ if (self.evidence_count == 0) != (self.evidence_started_at is None):
760
+ _refuse("EVIDENCE_COUNT", "inference.evidence_started_at", "does not match the count")
761
+ if (self.evidence_count == 0) != (self.evidence_completed_at is None):
762
+ _refuse("EVIDENCE_COUNT", "inference.evidence_completed_at", "does not match the count")
763
+ if self.state == "unknown" and self.evidence_count != 0:
764
+ _refuse("EVIDENCE_COUNT", "inference.evidence_count", "contradicts unknown state")
765
+ if self.state != "unknown" and self.evidence_count == 0:
766
+ _refuse("EVIDENCE_COUNT", "inference.evidence_count", "contradicts observed state")
767
+ if self.evidence_started_at is not None and _timestamp(
768
+ self.evidence_completed_at,
769
+ "inference.evidence_completed_at",
770
+ ) < _timestamp(self.evidence_started_at, "inference.evidence_started_at"):
771
+ _refuse("EVIDENCE_TIME", "inference.evidence_completed_at", "precedes start")
772
+ if (
773
+ type(self.recommended_probe_interval_seconds) is not int
774
+ or not 300 <= self.recommended_probe_interval_seconds <= 30 * 24 * 60 * 60
775
+ ):
776
+ _refuse(
777
+ "PROBE_INTERVAL",
778
+ "inference.recommended_probe_interval_seconds",
779
+ "is out of bounds",
780
+ )
781
+ if not isinstance(self.recommended_epoch_reset, bool):
782
+ _refuse("TYPE", "inference.recommended_epoch_reset", "must be boolean")
783
+ if self.recommended_epoch_reset and self.state != "changed_pattern":
784
+ _refuse("INFERENCE_STATE", "inference.recommended_epoch_reset", "contradicts state")
785
+ if (
786
+ not isinstance(self.reason_codes, tuple)
787
+ or not 1 <= len(self.reason_codes) <= 32
788
+ or any(
789
+ not isinstance(item, str)
790
+ or len(item) > 64
791
+ or _READER_IDENTIFIER.fullmatch(item) is None
792
+ for item in self.reason_codes
793
+ )
794
+ or len(set(self.reason_codes)) != len(self.reason_codes)
795
+ ):
796
+ _refuse("REASON_CODES", "inference.reason_codes", "must be unique typed codes")
797
+ reason_set = set(self.reason_codes)
798
+ reason_order = tuple(_SOURCE_CADENCE_ALGORITHM_SPEC["reason_codes"])
799
+ supported_reasons = set(reason_order)
800
+ if not reason_set <= supported_reasons:
801
+ _refuse("REASON_CODES", "inference.reason_codes", "contains an unsupported code")
802
+ if self.reason_codes != tuple(item for item in reason_order if item in reason_set):
803
+ _refuse("REASON_CODES", "inference.reason_codes", "is not in canonical order")
804
+ if self.state == "unknown":
805
+ if (
806
+ self.source_id != "unknown"
807
+ or self.inference_epoch != "unknown"
808
+ or self.history_digest != canonical_sha256([])
809
+ or self.estimated_interval_seconds is not None
810
+ or self.interval_jitter_seconds is not None
811
+ or any(value is not None for value in delays)
812
+ or self.expected_next_earliest is not None
813
+ or self.confidence_basis_points != 0
814
+ or self.revision_style != "unknown"
815
+ or self.recommended_epoch_reset
816
+ or self.reason_codes != ("no_observations",)
817
+ or self.recommended_probe_interval_seconds != 21_600
818
+ ):
819
+ _refuse("INFERENCE_STATE", "inference", "is not the canonical unknown result")
820
+ elif self.state == "learning":
821
+ if (
822
+ "insufficient_changes" not in reason_set
823
+ or not reason_set
824
+ <= {
825
+ "missing_publication_coordinates",
826
+ "insufficient_changes",
827
+ "probe_failures_observed",
828
+ }
829
+ or self.expected_next_earliest is not None
830
+ or self.recommended_epoch_reset
831
+ ):
832
+ _refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts learning state")
833
+ elif self.state == "stable":
834
+ if (
835
+ "regular_intervals" not in reason_set
836
+ or not reason_set <= {"regular_intervals", "probe_failures_observed"}
837
+ or self.expected_next_earliest is None
838
+ or self.confidence_basis_points < 3_000
839
+ or self.recommended_epoch_reset
840
+ ):
841
+ _refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts stable state")
842
+ elif self.state == "dormant":
843
+ if (
844
+ not {"regular_intervals", "prolonged_no_change"} <= reason_set
845
+ or not reason_set
846
+ <= {"regular_intervals", "prolonged_no_change", "probe_failures_observed"}
847
+ or self.expected_next_earliest is not None
848
+ or self.confidence_basis_points < 2_900
849
+ or self.recommended_epoch_reset
850
+ ):
851
+ _refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts dormant state")
852
+ elif self.cadence_kind == "irregular":
853
+ base_reasons = reason_set & {"regular_intervals", "irregular_intervals"}
854
+ allowed = {
855
+ "regular_intervals",
856
+ "irregular_intervals",
857
+ "cadence_regime_changed",
858
+ "probe_failures_observed",
859
+ }
860
+ base_is_canonical = (
861
+ len(base_reasons) == 1
862
+ if self.recommended_epoch_reset
863
+ else base_reasons == {"irregular_intervals"}
864
+ )
865
+ if not base_is_canonical or not reason_set <= allowed:
866
+ _refuse("INFERENCE_STATE", "inference.reason_codes", "contradicts irregular state")
867
+ elif not {
868
+ "regular_intervals",
869
+ "publication_window_missed",
870
+ } <= reason_set or not reason_set <= {
871
+ "regular_intervals",
872
+ "publication_window_missed",
873
+ "probe_failures_observed",
874
+ }:
875
+ _refuse(
876
+ "INFERENCE_STATE",
877
+ "inference.reason_codes",
878
+ "contradicts changed regular state",
879
+ )
880
+ if self.recommended_epoch_reset != ("cadence_regime_changed" in reason_set):
881
+ _refuse(
882
+ "INFERENCE_STATE",
883
+ "inference.reason_codes",
884
+ "does not match the epoch-reset decision",
885
+ )
886
+ if self.cadence_kind in {"regular", "irregular"} and self.interval_jitter_seconds is None:
887
+ _refuse("INFERENCE_STATE", "inference.interval_jitter_seconds", "is required")
888
+
889
+ @property
890
+ def digest(self) -> str:
891
+ return canonical_sha256(self.to_dict())
892
+
893
+ def to_dict(self) -> dict[str, Any]:
894
+ return {
895
+ "schema_version": self.schema_version,
896
+ "algorithm_version": self.algorithm_version,
897
+ "algorithm_digest": self.algorithm_digest,
898
+ "source_id": self.source_id,
899
+ "inference_epoch": self.inference_epoch,
900
+ "history_digest": self.history_digest,
901
+ "state": self.state,
902
+ "cadence_kind": self.cadence_kind,
903
+ "estimated_interval_seconds": self.estimated_interval_seconds,
904
+ "interval_jitter_seconds": self.interval_jitter_seconds,
905
+ "publication_delay_min_seconds": self.publication_delay_min_seconds,
906
+ "publication_delay_median_seconds": self.publication_delay_median_seconds,
907
+ "publication_delay_max_seconds": self.publication_delay_max_seconds,
908
+ "expected_next_earliest": self.expected_next_earliest,
909
+ "expected_next_latest": self.expected_next_latest,
910
+ "confidence_basis_points": self.confidence_basis_points,
911
+ "evidence_count": self.evidence_count,
912
+ "evidence_started_at": self.evidence_started_at,
913
+ "evidence_completed_at": self.evidence_completed_at,
914
+ "recommended_probe_interval_seconds": self.recommended_probe_interval_seconds,
915
+ "revision_style": self.revision_style,
916
+ "recommended_epoch_reset": self.recommended_epoch_reset,
917
+ "reason_codes": list(self.reason_codes),
918
+ }
919
+
920
+
921
+ _OBSERVATION_KEYS = set(EmpiricalSourceObservation.__dataclass_fields__)
922
+ _INFERENCE_KEYS = set(CadenceInference.__dataclass_fields__)
923
+
924
+
925
+ def parse_empirical_source_observation(value: object) -> EmpiricalSourceObservation:
926
+ """Parse one exact observation dictionary and reject unknown or missing fields."""
927
+
928
+ if not isinstance(value, Mapping):
929
+ _refuse("TYPE", "observation", "must be an object")
930
+ payload = dict(value)
931
+ _exact(payload, _OBSERVATION_KEYS, "observation")
932
+ frontier_raw = payload["frontier"]
933
+ if not isinstance(frontier_raw, Mapping):
934
+ _refuse("TYPE", "observation.frontier", "must be an object")
935
+ frontier = dict(frontier_raw)
936
+ _exact(frontier, {"kind", "value", "derivation_rule"}, "observation.frontier")
937
+ partitions_raw = payload["partition_digests"]
938
+ if not isinstance(partitions_raw, list):
939
+ _refuse("TYPE", "observation.partition_digests", "must be an array")
940
+ partitions: list[PartitionDigest] = []
941
+ for index, raw in enumerate(partitions_raw):
942
+ if not isinstance(raw, Mapping):
943
+ _refuse("TYPE", f"observation.partition_digests[{index}]", "must be an object")
944
+ item = dict(raw)
945
+ _exact(
946
+ item,
947
+ {"partition_key_digest", "content_digest"},
948
+ f"observation.partition_digests[{index}]",
949
+ )
950
+ partitions.append(PartitionDigest(**item))
951
+ payload["frontier"] = SourceFrontier(**frontier)
952
+ payload["partition_digests"] = tuple(partitions)
953
+ return EmpiricalSourceObservation(**payload)
954
+
955
+
956
+ def parse_cadence_inference(value: object) -> CadenceInference:
957
+ """Parse one exact inference dictionary."""
958
+
959
+ if not isinstance(value, Mapping):
960
+ _refuse("TYPE", "inference", "must be an object")
961
+ payload = dict(value)
962
+ _exact(payload, _INFERENCE_KEYS, "inference")
963
+ reasons = payload["reason_codes"]
964
+ if not isinstance(reasons, list):
965
+ _refuse("TYPE", "inference.reason_codes", "must be an array")
966
+ payload["reason_codes"] = tuple(reasons)
967
+ return CadenceInference(**payload)
968
+
969
+
970
+ def validate_observation_history(
971
+ history: Sequence[EmpiricalSourceObservation],
972
+ ) -> tuple[EmpiricalSourceObservation, ...]:
973
+ """Validate order, predecessor links, coordinates, and change semantics."""
974
+
975
+ observations = tuple(history)
976
+ if len(observations) > 4_096:
977
+ _refuse("HISTORY_SIZE", "history", "exceeds 4096 observations")
978
+ if not observations:
979
+ return observations
980
+ if any(not isinstance(item, EmpiricalSourceObservation) for item in observations):
981
+ _refuse("TYPE", "history", "contains a non-observation item")
982
+ if sum(len(item.partition_digests) for item in observations) > 8_192:
983
+ _refuse("HISTORY_SIZE", "history", "exceeds 8192 partition records")
984
+ first = observations[0]
985
+ identity = (
986
+ first.source_id,
987
+ first.inference_epoch,
988
+ first.connector_configuration_digest,
989
+ first.locator_digest,
990
+ first.recipe_digest,
991
+ first.source_authority_digest,
992
+ first.reader_family_id,
993
+ first.reader_family_version,
994
+ first.reader_options_digest,
995
+ first.frontier_policy_digest,
996
+ first.time_interpretation_digest,
997
+ first.inference_algorithm_digest,
998
+ )
999
+ latest_acquired: EmpiricalSourceObservation | None = None
1000
+ latest_change_time: datetime | None = None
1001
+ latest_publication_time: datetime | None = None
1002
+ latest_edition_sequence: int | None = None
1003
+ publication_coordinates_present: bool | None = None
1004
+ previous: EmpiricalSourceObservation | None = None
1005
+ for index, item in enumerate(observations):
1006
+ if item.sequence != index:
1007
+ _refuse("HISTORY_SEQUENCE", f"history[{index}].sequence", "is not contiguous")
1008
+ current_identity = (
1009
+ item.source_id,
1010
+ item.inference_epoch,
1011
+ item.connector_configuration_digest,
1012
+ item.locator_digest,
1013
+ item.recipe_digest,
1014
+ item.source_authority_digest,
1015
+ item.reader_family_id,
1016
+ item.reader_family_version,
1017
+ item.reader_options_digest,
1018
+ item.frontier_policy_digest,
1019
+ item.time_interpretation_digest,
1020
+ item.inference_algorithm_digest,
1021
+ )
1022
+ if current_identity != identity:
1023
+ _refuse("HISTORY_IDENTITY", f"history[{index}]", "splices another source or epoch")
1024
+ if previous is None:
1025
+ if item.previous_observation_digest is not None:
1026
+ _refuse("HISTORY_PREDECESSOR", f"history[{index}]", "has an unexpected predecessor")
1027
+ else:
1028
+ if item.previous_observation_digest != previous.digest:
1029
+ _refuse("HISTORY_PREDECESSOR", f"history[{index}]", "does not bind its predecessor")
1030
+ if _timestamp(item.started_at, f"history[{index}].started_at") < _timestamp(
1031
+ previous.completed_at, f"history[{index - 1}].completed_at"
1032
+ ):
1033
+ _refuse("HISTORY_TIME", f"history[{index}].started_at", "overlaps its predecessor")
1034
+ if item.result_kind in {"changed", "unchanged"} and item.content_digest is not None:
1035
+ if item.result_kind == "unchanged" and latest_acquired is None:
1036
+ _refuse(
1037
+ "UNCHANGED_BASELINE",
1038
+ f"history[{index}]",
1039
+ "has no prior acquired fact in this epoch",
1040
+ )
1041
+ if latest_acquired is not None:
1042
+ if (
1043
+ item.frontier.kind != latest_acquired.frontier.kind
1044
+ or item.frontier.derivation_rule != latest_acquired.frontier.derivation_rule
1045
+ ):
1046
+ _refuse(
1047
+ "HISTORY_IDENTITY",
1048
+ f"history[{index}].frontier",
1049
+ "changes the frontier policy within one epoch",
1050
+ )
1051
+ same_content = item.content_digest == latest_acquired.content_digest
1052
+ same_rows = item.row_digest == latest_acquired.row_digest
1053
+ same_schema = item.schema_digest == latest_acquired.schema_digest
1054
+ same_frontier = item.frontier == latest_acquired.frontier
1055
+ same_acquired_facts = (
1056
+ item.acquired_size_bytes == latest_acquired.acquired_size_bytes
1057
+ and item.row_count == latest_acquired.row_count
1058
+ and item.max_event_time == latest_acquired.max_event_time
1059
+ and item.partition_digests == latest_acquired.partition_digests
1060
+ and item.next_watermark_candidate_digest
1061
+ == latest_acquired.next_watermark_candidate_digest
1062
+ and item.publication_reference_time
1063
+ == latest_acquired.publication_reference_time
1064
+ and item.edition_sequence == latest_acquired.edition_sequence
1065
+ )
1066
+ if item.result_kind == "changed" and same_content and same_rows and same_schema:
1067
+ _refuse(
1068
+ "CHANGE_SEMANTICS",
1069
+ f"history[{index}]",
1070
+ "claims changed for identical data",
1071
+ )
1072
+ if item.result_kind == "unchanged" and not (
1073
+ same_content
1074
+ and same_rows
1075
+ and same_schema
1076
+ and same_frontier
1077
+ and same_acquired_facts
1078
+ ):
1079
+ _refuse(
1080
+ "CHANGE_SEMANTICS",
1081
+ f"history[{index}]",
1082
+ "claims unchanged for changed data",
1083
+ )
1084
+ comparison = _compare_frontiers(latest_acquired.frontier, item.frontier)
1085
+ if comparison is not None and comparison < 0:
1086
+ _refuse("FRONTIER_REGRESSION", f"history[{index}].frontier", "regresses")
1087
+ if comparison is not None and comparison > 0 and same_content:
1088
+ _refuse(
1089
+ "FRONTIER_CONTENT",
1090
+ f"history[{index}]",
1091
+ "advances with identical content",
1092
+ )
1093
+ latest_acquired = item
1094
+ elif item.result_kind == "unchanged" and latest_acquired is None:
1095
+ _refuse(
1096
+ "UNCHANGED_BASELINE",
1097
+ f"history[{index}]",
1098
+ "has no prior acquired fact in this epoch",
1099
+ )
1100
+ if item.result_kind == "changed":
1101
+ has_publication_coordinates = item.publication_reference_time is not None
1102
+ if publication_coordinates_present is None:
1103
+ publication_coordinates_present = has_publication_coordinates
1104
+ elif publication_coordinates_present != has_publication_coordinates:
1105
+ _refuse(
1106
+ "HISTORY_IDENTITY",
1107
+ f"history[{index}]",
1108
+ "changes publication-coordinate policy within one epoch",
1109
+ )
1110
+ change_time = _timestamp(item.completed_at, f"history[{index}].completed_at")
1111
+ if latest_change_time is not None and change_time <= latest_change_time:
1112
+ _refuse(
1113
+ "HISTORY_CHANGE_TIME",
1114
+ f"history[{index}].completed_at",
1115
+ "does not follow the prior material change",
1116
+ )
1117
+ latest_change_time = change_time
1118
+ if item.publication_reference_time is not None:
1119
+ publication_time = _timestamp(
1120
+ item.publication_reference_time,
1121
+ f"history[{index}].publication_reference_time",
1122
+ )
1123
+ assert item.edition_sequence is not None
1124
+ if (
1125
+ latest_publication_time is not None
1126
+ and publication_time <= latest_publication_time
1127
+ ):
1128
+ _refuse(
1129
+ "PUBLICATION_ORDER",
1130
+ f"history[{index}].publication_reference_time",
1131
+ "does not follow the prior changed edition",
1132
+ )
1133
+ if (
1134
+ latest_edition_sequence is not None
1135
+ and item.edition_sequence <= latest_edition_sequence
1136
+ ):
1137
+ _refuse(
1138
+ "PUBLICATION_ORDER",
1139
+ f"history[{index}].edition_sequence",
1140
+ "does not follow the prior changed edition",
1141
+ )
1142
+ latest_publication_time = publication_time
1143
+ latest_edition_sequence = item.edition_sequence
1144
+ previous = item
1145
+ return observations
1146
+
1147
+
1148
+ def _frontier_value(frontier: SourceFrontier) -> object:
1149
+ if frontier.kind == "date":
1150
+ return datetime.strptime(frontier.value, "%Y-%m-%d").date()
1151
+ if frontier.kind == "timestamp":
1152
+ return _timestamp(frontier.value, "frontier.value")
1153
+ if frontier.kind == "integer":
1154
+ return int(frontier.value)
1155
+ return frontier.value
1156
+
1157
+
1158
+ def _compare_frontiers(left: SourceFrontier, right: SourceFrontier) -> int | None:
1159
+ if left.kind != right.kind or left.derivation_rule != right.derivation_rule:
1160
+ return None
1161
+ if left.kind in {"none", "opaque"}:
1162
+ return 0 if left.value == right.value else None
1163
+ left_value = _frontier_value(left)
1164
+ right_value = _frontier_value(right)
1165
+ return (right_value > left_value) - (right_value < left_value)
1166
+
1167
+
1168
+ def _median_int(values: Sequence[int]) -> int:
1169
+ ordered = sorted(values)
1170
+ middle = len(ordered) // 2
1171
+ if len(ordered) % 2:
1172
+ return ordered[middle]
1173
+ return (ordered[middle - 1] + ordered[middle]) // 2
1174
+
1175
+
1176
+ def _revision_style(changes: Sequence[EmpiricalSourceObservation]) -> str:
1177
+ if len(changes) < 2:
1178
+ return "unknown"
1179
+ saw_append = False
1180
+ saw_unclassified = False
1181
+ saw_replacement = False
1182
+ for previous, current in pairwise(changes):
1183
+ comparison = _compare_frontiers(previous.frontier, current.frontier)
1184
+ if comparison == 0 and current.content_digest != previous.content_digest:
1185
+ return "historical_revision"
1186
+ if (
1187
+ current.row_count is not None
1188
+ and previous.row_count is not None
1189
+ and current.row_count < previous.row_count
1190
+ ):
1191
+ saw_replacement = True
1192
+ previous_partitions = {
1193
+ item.partition_key_digest: item.content_digest for item in previous.partition_digests
1194
+ }
1195
+ current_partitions = {
1196
+ item.partition_key_digest: item.content_digest for item in current.partition_digests
1197
+ }
1198
+ if previous_partitions:
1199
+ prior_rewritten = any(
1200
+ current_partitions.get(key) != digest for key, digest in previous_partitions.items()
1201
+ )
1202
+ if prior_rewritten:
1203
+ saw_replacement = True
1204
+ elif set(current_partitions) > set(previous_partitions):
1205
+ saw_append = True
1206
+ else:
1207
+ saw_unclassified = True
1208
+ else:
1209
+ saw_unclassified = True
1210
+ if saw_replacement:
1211
+ return "replacement"
1212
+ if saw_append and not saw_unclassified:
1213
+ return "append"
1214
+ return "unknown"
1215
+
1216
+
1217
+ def infer_source_cadence(
1218
+ history: Sequence[EmpiricalSourceObservation],
1219
+ ) -> CadenceInference:
1220
+ """Return a deterministic empirical cadence result for one validated history."""
1221
+
1222
+ observations = validate_observation_history(history)
1223
+ if not observations:
1224
+ return CadenceInference(
1225
+ source_id="unknown",
1226
+ inference_epoch="unknown",
1227
+ history_digest=canonical_sha256([]),
1228
+ state="unknown",
1229
+ cadence_kind="unknown",
1230
+ estimated_interval_seconds=None,
1231
+ interval_jitter_seconds=None,
1232
+ publication_delay_min_seconds=None,
1233
+ publication_delay_median_seconds=None,
1234
+ publication_delay_max_seconds=None,
1235
+ expected_next_earliest=None,
1236
+ expected_next_latest=None,
1237
+ confidence_basis_points=0,
1238
+ evidence_count=0,
1239
+ evidence_started_at=None,
1240
+ evidence_completed_at=None,
1241
+ recommended_probe_interval_seconds=21_600,
1242
+ revision_style="unknown",
1243
+ recommended_epoch_reset=False,
1244
+ reason_codes=("no_observations",),
1245
+ )
1246
+ first = observations[0]
1247
+ if first.inference_algorithm_digest != SOURCE_CADENCE_ALGORITHM_DIGEST:
1248
+ _refuse("ALGORITHM", "history", "requires a different inference implementation")
1249
+ history_digest = canonical_sha256([item.to_dict() for item in observations])
1250
+ changes = tuple(item for item in observations if item.result_kind == "changed")
1251
+ publication_coordinates_complete = bool(changes) and all(
1252
+ item.publication_reference_time is not None and item.edition_sequence is not None
1253
+ for item in changes
1254
+ )
1255
+ publication_times = (
1256
+ tuple(
1257
+ _timestamp(item.publication_reference_time, "observation.publication_reference_time")
1258
+ for item in changes
1259
+ )
1260
+ if publication_coordinates_complete
1261
+ else ()
1262
+ )
1263
+ edition_sequences = (
1264
+ tuple(item.edition_sequence for item in changes) if publication_coordinates_complete else ()
1265
+ )
1266
+ interval_values: list[int] = []
1267
+ for (left_time, right_time), (left_sequence, right_sequence) in zip(
1268
+ pairwise(publication_times),
1269
+ pairwise(edition_sequences),
1270
+ strict=True,
1271
+ ):
1272
+ assert left_sequence is not None and right_sequence is not None
1273
+ editions = right_sequence - left_sequence
1274
+ seconds = _elapsed_whole_seconds(left_time, right_time)
1275
+ if editions <= 0 or seconds < editions:
1276
+ _refuse(
1277
+ "PUBLICATION_INTERVAL",
1278
+ "history",
1279
+ "cannot derive a positive whole-second interval per edition",
1280
+ )
1281
+ interval_values.append(seconds // editions)
1282
+ intervals = tuple(interval_values)
1283
+ delays_list: list[int] = []
1284
+ for item in changes:
1285
+ if item.publication_reference_time is None:
1286
+ continue
1287
+ completed = _timestamp(item.completed_at, "observation.completed_at")
1288
+ publication_time = _timestamp(
1289
+ item.publication_reference_time,
1290
+ "observation.publication_reference_time",
1291
+ )
1292
+ delays_list.append(_elapsed_whole_seconds(publication_time, completed))
1293
+ delays = tuple(delays_list)
1294
+ interval = _median_int(intervals) if intervals else None
1295
+ jitter = (
1296
+ max((abs(value - interval) for value in intervals), default=None)
1297
+ if interval is not None
1298
+ else None
1299
+ )
1300
+ state = "learning"
1301
+ cadence_kind = "unknown"
1302
+ reset = False
1303
+ reasons: list[str] = []
1304
+ if changes and not publication_coordinates_complete:
1305
+ reasons.append("missing_publication_coordinates")
1306
+ if len(changes) < 4 or not publication_coordinates_complete:
1307
+ reasons.append("insufficient_changes")
1308
+ else:
1309
+ assert interval is not None and jitter is not None
1310
+ allowed_spread = max(3_600, interval * 2_000 // 10_000)
1311
+ if jitter <= allowed_spread:
1312
+ state = "stable"
1313
+ cadence_kind = "regular"
1314
+ reasons.append("regular_intervals")
1315
+ else:
1316
+ state = "changed_pattern"
1317
+ cadence_kind = "irregular"
1318
+ interval = None
1319
+ reasons.append("irregular_intervals")
1320
+ if len(intervals) >= 5:
1321
+ baseline = _median_int(intervals[:-2])
1322
+ recent = intervals[-2:]
1323
+ materially_changed = all(
1324
+ abs(value - baseline) * 10_000 > baseline * 4_000 for value in recent
1325
+ )
1326
+ mutually_consistent = abs(recent[0] - recent[1]) * 10_000 <= max(recent) * 2_000
1327
+ if materially_changed and mutually_consistent:
1328
+ state = "changed_pattern"
1329
+ cadence_kind = "irregular"
1330
+ reset = True
1331
+ interval = None
1332
+ reasons.append("cadence_regime_changed")
1333
+ latest_time = _timestamp(observations[-1].completed_at, "observation.completed_at")
1334
+ regular_interval = _median_int(intervals) if intervals else None
1335
+ if state == "stable" and regular_interval is not None and changes:
1336
+ since_change = _elapsed_whole_seconds(publication_times[-1], latest_time)
1337
+ trailing = tuple(
1338
+ item.result_kind for item in observations[observations.index(changes[-1]) + 1 :]
1339
+ )
1340
+ if (
1341
+ since_change > regular_interval * 4
1342
+ and len(trailing) >= 3
1343
+ and all(kind in {"unchanged", "not_published"} for kind in trailing)
1344
+ ):
1345
+ state = "dormant"
1346
+ reasons.append("prolonged_no_change")
1347
+ expected_earliest = expected_latest = None
1348
+ if regular_interval is not None and changes and cadence_kind == "regular":
1349
+ margin = max(300, regular_interval // 20, jitter or 0)
1350
+ center = _safe_shift(
1351
+ publication_times[-1],
1352
+ regular_interval,
1353
+ "inference.expected_window",
1354
+ )
1355
+ earliest = _safe_shift(center, -margin, "inference.expected_next_earliest")
1356
+ latest = _safe_shift(center, margin, "inference.expected_next_latest")
1357
+ if state == "stable" and latest_time <= latest:
1358
+ expected_earliest = _format_timestamp(earliest)
1359
+ expected_latest = _format_timestamp(latest)
1360
+ elif state == "stable":
1361
+ state = "changed_pattern"
1362
+ reasons.append("publication_window_missed")
1363
+ failure_count = sum(item.result_kind == "failed" for item in observations)
1364
+ if failure_count:
1365
+ reasons.append("probe_failures_observed")
1366
+ confidence = 0
1367
+ if state == "learning":
1368
+ confidence = min(4_000, len(changes) * 1_000)
1369
+ elif state == "stable":
1370
+ confidence = min(9_500, 6_000 + max(0, len(intervals) - 3) * 500)
1371
+ elif state == "dormant":
1372
+ confidence = min(8_000, 5_000 + len(intervals) * 300)
1373
+ elif state == "changed_pattern":
1374
+ confidence = 2_500
1375
+ confidence = max(0, confidence - min(3_000, failure_count * 500))
1376
+ if state == "stable" and regular_interval is not None:
1377
+ probe = max(300, min(21_600, regular_interval // 8))
1378
+ elif state == "dormant" and regular_interval is not None:
1379
+ probe = max(86_400, min(7 * 86_400, regular_interval))
1380
+ elif intervals:
1381
+ probe = max(900, min(86_400, _median_int(intervals) // 4))
1382
+ else:
1383
+ probe = 21_600
1384
+ if reset:
1385
+ probe = min(probe, 3_600)
1386
+ delay_min = min(delays) if delays else None
1387
+ delay_median = _median_int(delays) if delays else None
1388
+ delay_max = max(delays) if delays else None
1389
+ return CadenceInference(
1390
+ source_id=first.source_id,
1391
+ inference_epoch=first.inference_epoch,
1392
+ history_digest=history_digest,
1393
+ state=state,
1394
+ cadence_kind=cadence_kind,
1395
+ estimated_interval_seconds=interval,
1396
+ interval_jitter_seconds=jitter,
1397
+ publication_delay_min_seconds=delay_min,
1398
+ publication_delay_median_seconds=delay_median,
1399
+ publication_delay_max_seconds=delay_max,
1400
+ expected_next_earliest=expected_earliest,
1401
+ expected_next_latest=expected_latest,
1402
+ confidence_basis_points=confidence,
1403
+ evidence_count=len(observations),
1404
+ evidence_started_at=observations[0].started_at,
1405
+ evidence_completed_at=observations[-1].completed_at,
1406
+ recommended_probe_interval_seconds=probe,
1407
+ revision_style=_revision_style(changes),
1408
+ recommended_epoch_reset=reset,
1409
+ reason_codes=tuple(reasons),
1410
+ )
1411
+
1412
+
1413
+ __all__ = [
1414
+ "SOURCE_CADENCE_ALGORITHM_DIGEST",
1415
+ "SOURCE_CADENCE_ALGORITHM_VERSION",
1416
+ "SOURCE_CADENCE_INFERENCE_SCHEMA",
1417
+ "SOURCE_CADENCE_OBSERVATION_SCHEMA",
1418
+ "CadenceContractError",
1419
+ "CadenceInference",
1420
+ "EmpiricalSourceObservation",
1421
+ "PartitionDigest",
1422
+ "SourceFrontier",
1423
+ "infer_source_cadence",
1424
+ "parse_cadence_inference",
1425
+ "parse_empirical_source_observation",
1426
+ "source_cadence_algorithm_spec",
1427
+ "validate_observation_history",
1428
+ ]