mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,701 @@
1
+ """The closed Data.gov v4 metadata authoring policy.
2
+
3
+ This module is the only place where a normalized Data.gov v4 provider record becomes a v2 catalog
4
+ entry, and it is deliberately a table rather than a program. Every known fact it authors is a
5
+ value the provider literally declared at an allowlisted JSON path, carried together with the exact
6
+ record digest that stated it. Everything else is ``unknown``. There is no inference here: no
7
+ heuristic, no model, no title-derived identity, no format guessing, and no resource fetch. The
8
+ module imports no transport and is given bytes-derived dictionaries, never a retriever.
9
+
10
+ Three separations are load-bearing.
11
+
12
+ *Semantic fields versus observations.* A normalized record has a semantic
13
+ ``fields`` partition and a volatile ``observations`` partition. ``last_harvested_date`` is a
14
+ catalog-ingestion timestamp that Data.gov rewrites on every re-harvest, and ``popularity`` is a
15
+ per-sweep counter; neither is a fact about the source. The policy reads ``fields`` only, re-derives
16
+ the record's own semantic digest before authoring anything from it, and refuses outright any record
17
+ whose semantic ``fields`` carry an observation path. Locator values -- landing pages, access and
18
+ download URLs, harvest-record links, organization logos -- do stay in ``fields`` as inert bounded
19
+ text; the policy simply never reads them, so no fact and no provenance can ever cite one.
20
+
21
+ *Facts versus layer text.* A fact keeps the complete declared value. The four retrieval layer
22
+ texts are a rendering of those facts and are capped at :data:`MAX_LAYER_TEXT_BYTES` UTF-8 bytes
23
+ with a Unicode-boundary-safe truncation plus an explicit per-entry disposition. The cap is what
24
+ keeps a 1,000-entry range/layer aggregate inside the 8 MiB ``encode_many`` bound downstream
25
+ (1000 * 8192 = 8,192,000 <= 8,388,608), so exactly one encode call per range/layer/backend stays
26
+ exact. The pinned MiniLM tokenizer truncates far below this cap, so retrieval is unaffected.
27
+
28
+ *Admission versus evidence.* Only an exact reviewed license mapping may state rights. Declared
29
+ rights prose, a federal publisher, or a public access level never grants one; ambiguity is flagged
30
+ and queued, never guessed.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ from collections.abc import Mapping
36
+ from dataclasses import dataclass
37
+ from types import MappingProxyType
38
+ from typing import Any
39
+
40
+ from mostlyright.data_harness.canonical import canonical_sha256, sha256_bytes
41
+ from mostlyright.data_harness.sources.catalog.contracts import (
42
+ EMBEDDING_LAYERS,
43
+ MAX_DECLARED_NAME,
44
+ MAX_DECLARED_VOCABULARY,
45
+ MAX_DESCRIPTION,
46
+ MAX_SPATIAL_SCOPE,
47
+ )
48
+ from mostlyright.data_harness.sources.catalog.entry_v2 import (
49
+ CATALOG_ENTRY_V2_CONTRACT_VERSION,
50
+ PROVIDER_CONTRACTS,
51
+ CatalogEntryV2,
52
+ CatalogFact,
53
+ CatalogObservation,
54
+ ProviderFactProvenance,
55
+ ProviderRecordIdentity,
56
+ )
57
+ from mostlyright.data_harness.sources.catalog.harvest.datagov_v4 import (
58
+ DATAGOV_V4_RECORD_VERSION,
59
+ )
60
+ from mostlyright.data_harness.sources.catalog.harvest.protocol import (
61
+ DATAGOV_V4_HARVESTER_COORDINATE,
62
+ )
63
+ from mostlyright.data_harness.sources.contracts import (
64
+ DATA_FORMATS,
65
+ RIGHTS_STATUSES,
66
+ SourceContractError,
67
+ _CanonicalContract,
68
+ )
69
+
70
+ AUTHORING_POLICY_SCHEMA = "mr-data-catalog-authoring-policy.v1"
71
+ DATAGOV_V4_POLICY_ID = "datagov-v4-policy.v1"
72
+
73
+ # One <=1000-entry range holds at most 1000 * 8192 = 8,192,000 bytes per layer, inside the
74
+ # 8,388,608-byte ``encode_many`` bound. Widening this constant breaks that equation.
75
+ MAX_LAYER_TEXT_BYTES = 8_192
76
+
77
+ # ``CatalogEntryV2`` bounds a known title at 200 characters. A longer or unsafe title cannot be
78
+ # carried, and truncating identity prose would author something the provider never declared, so
79
+ # such a record is skipped rather than reshaped.
80
+ MAX_TITLE = 200
81
+
82
+ LAYER_TEXT_UNKNOWN = "unknown"
83
+ LAYER_TEXT_NOT_APPLICABLE = "not_applicable"
84
+
85
+ AUTHORING_DISPOSITIONS = ("authored", "flagged", "skipped", "failed")
86
+
87
+ AUTHORING_REASON_CODES = frozenset(
88
+ {
89
+ "AUTHOR_OK",
90
+ "AUTHOR_RIGHTS_ABSENT",
91
+ "AUTHOR_RIGHTS_CONDITIONAL",
92
+ "AUTHOR_RIGHTS_PROHIBITED",
93
+ "AUTHOR_RIGHTS_PROSE",
94
+ "AUTHOR_RIGHTS_UNMAPPED",
95
+ "AUTHOR_TITLE_ABSENT",
96
+ "AUTHOR_TITLE_LIMIT",
97
+ "AUTHOR_TITLE_UNSAFE",
98
+ "AUTHOR_IDENTIFIER_UNSAFE",
99
+ "AUTHOR_RECORD_SCHEMA",
100
+ "AUTHOR_RECORD_DIGEST",
101
+ "AUTHOR_RECORD_FIELDS",
102
+ "AUTHOR_ENTRY_CONTRACT",
103
+ }
104
+ )
105
+
106
+ _RECORD_MEMBERS = frozenset(
107
+ {
108
+ "schema_version",
109
+ "protocol",
110
+ "record_id",
111
+ "fields",
112
+ "observations",
113
+ "raw_response_sha256",
114
+ "normalized_sha256",
115
+ }
116
+ )
117
+ _REQUIRED_RECORD_MEMBERS = _RECORD_MEMBERS - {"observations"}
118
+
119
+ # The exact provider paths this policy may read. Adding one is a reviewed source edit.
120
+ _SOURCE_JSON_PATHS = (
121
+ "$.dcat.distribution[*].format",
122
+ "$.dcat.distribution[*].mediaType",
123
+ "$.dcat.license",
124
+ "$.dcat.rights",
125
+ "$.dcat.spatial",
126
+ "$.dcat.spatial[*]",
127
+ "$.description",
128
+ "$.identifier",
129
+ "$.keyword[*]",
130
+ "$.publisher",
131
+ "$.theme[*]",
132
+ "$.title",
133
+ )
134
+
135
+ # Catalog-ingestion observations belong in a separate ``observations`` partition
136
+ # precisely because they are not facts about a source, so a semantic ``fields`` partition that
137
+ # carries one has been tampered with and the record is refused outright.
138
+ _OBSERVATION_JSON_PATHS = ("$.last_harvested_date", "$.popularity")
139
+
140
+ # Locator and catalog-ingestion paths this policy never reads. The raw evidence retains the
141
+ # locator values in raw and normalized evidence as inert bounded text; they are simply never facts,
142
+ # never provenance, and never fetched.
143
+ _REFUSED_JSON_PATHS = (
144
+ "$.dcat.describedBy",
145
+ "$.dcat.distribution[*].accessURL",
146
+ "$.dcat.distribution[*].describedBy",
147
+ "$.dcat.distribution[*].downloadURL",
148
+ "$.dcat.landingPage",
149
+ "$.dcat.references",
150
+ "$.harvest_record",
151
+ "$.harvest_record_raw",
152
+ "$.harvest_record_transformed",
153
+ "$.last_harvested_date",
154
+ "$.organization.logo",
155
+ "$.popularity",
156
+ )
157
+
158
+ # Exact declared license values only. Public-domain dedications carry no obligation and may be
159
+ # ``approved``; attribution licenses are ``conditional`` and therefore still flagged, because v2
160
+ # has no obligation fact that could carry the attribution requirement.
161
+ _RIGHTS_MAP = (
162
+ ("http://creativecommons.org/licenses/by/4.0/", "conditional"),
163
+ ("http://creativecommons.org/publicdomain/zero/1.0/", "approved"),
164
+ ("http://www.usa.gov/publicdomain/label/1.0/", "approved"),
165
+ ("https://creativecommons.org/licenses/by/4.0/", "conditional"),
166
+ ("https://creativecommons.org/publicdomain/zero/1.0/", "approved"),
167
+ ("https://www.usa.gov/publicdomain/label/1.0/", "approved"),
168
+ )
169
+
170
+ # Exact declared distribution ``format``/``mediaType`` spellings that name one wire encoding the
171
+ # harness can actually read. Anything else -- HTML, PDF, ZIP, API, XML, GeoJSON -- is deliberately
172
+ # unmapped: an unrecognized format is not a format we may claim.
173
+ _FORMAT_MAP = (
174
+ ("CSV", "csv"),
175
+ ("JSON", "json"),
176
+ ("NDJSON", "ndjson"),
177
+ ("PARQUET", "parquet"),
178
+ ("application/json", "json"),
179
+ ("application/vnd.apache.parquet", "parquet"),
180
+ ("application/x-ndjson", "ndjson"),
181
+ ("csv", "csv"),
182
+ ("json", "json"),
183
+ ("ndjson", "ndjson"),
184
+ ("parquet", "parquet"),
185
+ ("text/csv", "csv"),
186
+ )
187
+
188
+
189
+ class AuthoringPolicyRefused(SourceContractError):
190
+ """A stable refusal of an authoring policy coordinate or contract."""
191
+
192
+
193
+ @dataclass(frozen=True)
194
+ class AuthoringPolicy(_CanonicalContract):
195
+ """One reviewed, allowlisted, digest-bound provider authoring policy."""
196
+
197
+ policy_id: str
198
+ provider_id: str
199
+ harvester_coordinate: str
200
+ record_schema_version: str
201
+ entry_schema_version: str
202
+ entry_id_prefix: str
203
+ identifier_json_path: str
204
+ source_json_paths: tuple[str, ...]
205
+ refused_json_paths: tuple[str, ...]
206
+ observation_json_paths: tuple[str, ...]
207
+ rights_map: tuple[tuple[str, str], ...]
208
+ format_map: tuple[tuple[str, str], ...]
209
+ max_layer_text_bytes: int
210
+ schema_version: str = AUTHORING_POLICY_SCHEMA
211
+
212
+ def __post_init__(self) -> None:
213
+ contract = PROVIDER_CONTRACTS.get(self.provider_id)
214
+ if (
215
+ self.schema_version != AUTHORING_POLICY_SCHEMA
216
+ or contract is None
217
+ or self.harvester_coordinate != contract.harvester_coordinate
218
+ or self.identifier_json_path != contract.identifier_json_path
219
+ ):
220
+ raise AuthoringPolicyRefused(
221
+ "AUTHOR_POLICY_CONTRACT",
222
+ "policy.provider_id",
223
+ "policy must bind one registered provider contract",
224
+ )
225
+ source = set(self.source_json_paths)
226
+ refused = set(self.refused_json_paths)
227
+ if (
228
+ self.source_json_paths != tuple(sorted(source))
229
+ or self.refused_json_paths != tuple(sorted(refused))
230
+ or self.observation_json_paths != tuple(sorted(set(self.observation_json_paths)))
231
+ or not source
232
+ or not refused
233
+ or not self.observation_json_paths
234
+ or not source.isdisjoint(refused)
235
+ or not set(self.observation_json_paths) <= refused
236
+ or self.identifier_json_path not in source
237
+ ):
238
+ raise AuthoringPolicyRefused(
239
+ "AUTHOR_POLICY_CONTRACT",
240
+ "policy.source_json_paths",
241
+ "source and refused paths must be sorted, unique, and disjoint",
242
+ )
243
+ if self.rights_map != tuple(sorted(set(self.rights_map))) or any(
244
+ status not in RIGHTS_STATUSES for _value, status in self.rights_map
245
+ ):
246
+ raise AuthoringPolicyRefused(
247
+ "AUTHOR_POLICY_CONTRACT",
248
+ "policy.rights_map",
249
+ "rights mappings must be sorted, unique, and closed",
250
+ )
251
+ if self.format_map != tuple(sorted(set(self.format_map))) or any(
252
+ data_format not in DATA_FORMATS for _value, data_format in self.format_map
253
+ ):
254
+ raise AuthoringPolicyRefused(
255
+ "AUTHOR_POLICY_CONTRACT",
256
+ "policy.format_map",
257
+ "format mappings must be sorted, unique, and closed",
258
+ )
259
+ if type(self.max_layer_text_bytes) is not int or not 0 < self.max_layer_text_bytes <= (
260
+ MAX_LAYER_TEXT_BYTES
261
+ ):
262
+ raise AuthoringPolicyRefused(
263
+ "AUTHOR_POLICY_CONTRACT",
264
+ "policy.max_layer_text_bytes",
265
+ "layer text cap must stay inside the shared encode bound",
266
+ )
267
+
268
+ def to_dict(self) -> dict[str, Any]:
269
+ return {
270
+ "schema_version": self.schema_version,
271
+ "policy_id": self.policy_id,
272
+ "provider_id": self.provider_id,
273
+ "harvester_coordinate": self.harvester_coordinate,
274
+ "record_schema_version": self.record_schema_version,
275
+ "entry_schema_version": self.entry_schema_version,
276
+ "entry_id_prefix": self.entry_id_prefix,
277
+ "identifier_json_path": self.identifier_json_path,
278
+ "source_json_paths": list(self.source_json_paths),
279
+ "refused_json_paths": list(self.refused_json_paths),
280
+ "observation_json_paths": list(self.observation_json_paths),
281
+ "rights_map": [
282
+ {"declared": value, "status": status} for value, status in self.rights_map
283
+ ],
284
+ "format_map": [
285
+ {"declared": value, "data_format": data_format}
286
+ for value, data_format in self.format_map
287
+ ],
288
+ "max_layer_text_bytes": self.max_layer_text_bytes,
289
+ }
290
+
291
+ def rights_for(self, declared: str) -> str | None:
292
+ return dict(self.rights_map).get(declared)
293
+
294
+ def data_format_for(self, declared: str) -> str | None:
295
+ return dict(self.format_map).get(declared)
296
+
297
+
298
+ DATAGOV_V4_AUTHORING_POLICY = AuthoringPolicy(
299
+ policy_id=DATAGOV_V4_POLICY_ID,
300
+ provider_id="datagov_v4",
301
+ harvester_coordinate=DATAGOV_V4_HARVESTER_COORDINATE,
302
+ record_schema_version=DATAGOV_V4_RECORD_VERSION,
303
+ entry_schema_version=CATALOG_ENTRY_V2_CONTRACT_VERSION,
304
+ entry_id_prefix="datagov_v4.",
305
+ identifier_json_path="$.identifier",
306
+ source_json_paths=_SOURCE_JSON_PATHS,
307
+ refused_json_paths=_REFUSED_JSON_PATHS,
308
+ observation_json_paths=_OBSERVATION_JSON_PATHS,
309
+ rights_map=_RIGHTS_MAP,
310
+ format_map=_FORMAT_MAP,
311
+ max_layer_text_bytes=MAX_LAYER_TEXT_BYTES,
312
+ )
313
+
314
+ AUTHORING_POLICIES: Mapping[str, AuthoringPolicy] = MappingProxyType(
315
+ {DATAGOV_V4_POLICY_ID: DATAGOV_V4_AUTHORING_POLICY}
316
+ )
317
+
318
+
319
+ def resolve_authoring_policy(policy_id: Any) -> AuthoringPolicy:
320
+ """Return the exact reviewed policy for one coordinate; never infer a near match."""
321
+
322
+ policy = AUTHORING_POLICIES.get(policy_id) if isinstance(policy_id, str) else None
323
+ if policy is None:
324
+ raise AuthoringPolicyRefused(
325
+ "AUTHOR_POLICY_UNKNOWN",
326
+ "policy_id",
327
+ "policy coordinate is not one reviewed allowlisted policy",
328
+ )
329
+ return policy
330
+
331
+
332
+ @dataclass(frozen=True)
333
+ class AuthoredRecord:
334
+ """One record's exact authoring outcome: an entry or an explicit refusal, never a guess."""
335
+
336
+ provider_record_id: str
337
+ disposition: str
338
+ reason_code: str
339
+ entry: CatalogEntryV2 | None
340
+ layer_texts: tuple[tuple[str, str], ...]
341
+ truncated_layers: tuple[str, ...]
342
+
343
+ @property
344
+ def entry_id(self) -> str | None:
345
+ return None if self.entry is None else self.entry.entry_id
346
+
347
+ def to_dict(self) -> dict[str, Any]:
348
+ return {
349
+ "provider_record_id": self.provider_record_id,
350
+ "disposition": self.disposition,
351
+ "reason_code": self.reason_code,
352
+ "entry": None if self.entry is None else self.entry.to_dict(),
353
+ "layer_texts": [{"layer": layer, "text": text} for layer, text in self.layer_texts],
354
+ "layer_text_truncated": list(self.truncated_layers),
355
+ }
356
+
357
+
358
+ def author_catalog_record(
359
+ record: Mapping[str, Any],
360
+ *,
361
+ policy: AuthoringPolicy,
362
+ observed_at: str,
363
+ page_evidence_sha256: str,
364
+ ) -> AuthoredRecord:
365
+ """Map one normalized provider record onto one honest v2 entry and its four layer texts."""
366
+
367
+ envelope = _record_envelope(record, policy)
368
+ if envelope is not None:
369
+ return _refusal(record, "failed", envelope)
370
+ record_id = record["record_id"]
371
+ if canonical_sha256(_semantic_payload(record)) != record["normalized_sha256"]:
372
+ return _refusal(record, "failed", "AUTHOR_RECORD_DIGEST")
373
+ observations = {_base_path(path) for path in policy.observation_json_paths}
374
+ values: dict[str, Any] = {}
375
+ for field in record["fields"]:
376
+ if _base_path(field["path"]) in observations:
377
+ return _refusal(record, "failed", "AUTHOR_RECORD_FIELDS")
378
+ # A ``typed_json_ast.v1`` value carries a provider number too large for exact JSON. No fact
379
+ # this policy authors is numeric, so such a value simply declares nothing here.
380
+ if field.get("encoding", "canonical_json") == "canonical_json":
381
+ values[field["path"]] = field["value"]
382
+
383
+ if values.get(policy.identifier_json_path) != record_id or not _safe_text(
384
+ record_id, maximum=512
385
+ ):
386
+ return _refusal(record, "skipped", "AUTHOR_IDENTIFIER_UNSAFE")
387
+
388
+ title = values.get("$.title")
389
+ if not isinstance(title, str) or not title:
390
+ return _refusal(record, "skipped", "AUTHOR_TITLE_ABSENT")
391
+ if not _safe_text(title, maximum=len(title)):
392
+ return _refusal(record, "skipped", "AUTHOR_TITLE_UNSAFE")
393
+ if len(title) > MAX_TITLE:
394
+ return _refusal(record, "skipped", "AUTHOR_TITLE_LIMIT")
395
+
396
+ provenance = _provenance_factory(policy, record)
397
+ rights_fact, rights_reason = _rights(values, policy, provenance)
398
+ facts = {
399
+ "title": _known_text_fact(title, ("$.title",), provenance),
400
+ "publisher": _text_fact(values, "$.publisher", maximum=200, provenance=provenance),
401
+ "description": _text_fact(
402
+ values, "$.description", maximum=MAX_DESCRIPTION, provenance=provenance
403
+ ),
404
+ "spatial_scope": _text_tuple_fact(
405
+ values,
406
+ ("$.dcat.spatial", "$.dcat.spatial[*]"),
407
+ maximum=MAX_SPATIAL_SCOPE,
408
+ provenance=provenance,
409
+ ),
410
+ "data_formats": _data_formats(values, policy, provenance),
411
+ "declared_vocabulary": _text_tuple_fact(
412
+ values,
413
+ ("$.keyword[*]", "$.theme[*]"),
414
+ maximum=MAX_DECLARED_VOCABULARY,
415
+ provenance=provenance,
416
+ ),
417
+ "rights": rights_fact,
418
+ # No v4 field asserts any of these, so a closed policy states exactly that. ``accessLevel``
419
+ # is a disclosure classification, not one of the harness access kinds; the provider
420
+ # declares no column list, row count, profile, or dataset authentication requirement.
421
+ "access_kind": CatalogFact(state="unknown"),
422
+ "authentication_required": CatalogFact(state="unknown"),
423
+ "declared_columns": CatalogFact(state="unknown"),
424
+ "declared_row_count": CatalogFact(state="unknown"),
425
+ "profiles": CatalogFact(state="unknown"),
426
+ }
427
+ try:
428
+ entry = CatalogEntryV2(
429
+ entry_id=f"{policy.entry_id_prefix}{sha256_bytes(record_id.encode('utf-8'))}",
430
+ entry_version=1,
431
+ provider_record=ProviderRecordIdentity(
432
+ provider_id=policy.provider_id,
433
+ provider_record_id=record_id,
434
+ provider_record_sha256=record["normalized_sha256"],
435
+ identifier_json_path=policy.identifier_json_path,
436
+ ),
437
+ observations=(
438
+ CatalogObservation(
439
+ harvester_coordinate=policy.harvester_coordinate,
440
+ observed_at=observed_at,
441
+ response_evidence_sha256=record["raw_response_sha256"],
442
+ page_evidence_sha256=page_evidence_sha256,
443
+ ),
444
+ ),
445
+ **facts,
446
+ )
447
+ except SourceContractError:
448
+ return _refusal(record, "failed", "AUTHOR_ENTRY_CONTRACT")
449
+
450
+ texts, truncated = _layer_texts(entry, policy)
451
+ admitted = rights_fact.state == "known" and rights_fact.value == "approved"
452
+ return AuthoredRecord(
453
+ provider_record_id=record_id,
454
+ disposition="authored" if admitted else "flagged",
455
+ reason_code=rights_reason,
456
+ entry=entry,
457
+ layer_texts=texts,
458
+ truncated_layers=truncated,
459
+ )
460
+
461
+
462
+ def _record_envelope(record: Mapping[str, Any], policy: AuthoringPolicy) -> str | None:
463
+ if not isinstance(record, Mapping) or not _REQUIRED_RECORD_MEMBERS <= set(record) <= (
464
+ _RECORD_MEMBERS
465
+ ):
466
+ return "AUTHOR_RECORD_SCHEMA"
467
+ if (
468
+ record["schema_version"] != policy.record_schema_version
469
+ or record["protocol"] != policy.provider_id
470
+ or not isinstance(record["record_id"], str)
471
+ or not record["record_id"]
472
+ or not isinstance(record["fields"], list)
473
+ or not record["fields"]
474
+ or not _is_digest(record["raw_response_sha256"])
475
+ or not _is_digest(record["normalized_sha256"])
476
+ or not isinstance(record.get("observations", []), list)
477
+ ):
478
+ return "AUTHOR_RECORD_SCHEMA"
479
+ for field in record["fields"]:
480
+ if (
481
+ not isinstance(field, Mapping)
482
+ or not {"path", "value"} <= set(field) <= {"path", "value", "encoding"}
483
+ or not isinstance(field["path"], str)
484
+ ):
485
+ return "AUTHOR_RECORD_SCHEMA"
486
+ return None
487
+
488
+
489
+ def _semantic_payload(record: Mapping[str, Any]) -> dict[str, Any]:
490
+ return {
491
+ "schema_version": record["schema_version"],
492
+ "protocol": record["protocol"],
493
+ "record_id": record["record_id"],
494
+ "fields": record["fields"],
495
+ }
496
+
497
+
498
+ def _refusal(record: Mapping[str, Any], disposition: str, reason_code: str) -> AuthoredRecord:
499
+ identifier = record.get("record_id") if isinstance(record, Mapping) else None
500
+ return AuthoredRecord(
501
+ provider_record_id=identifier if isinstance(identifier, str) else "",
502
+ disposition=disposition,
503
+ reason_code=reason_code,
504
+ entry=None,
505
+ layer_texts=(),
506
+ truncated_layers=(),
507
+ )
508
+
509
+
510
+ def _provenance_factory(policy: AuthoringPolicy, record: Mapping[str, Any]):
511
+ def build(paths: tuple[str, ...]) -> ProviderFactProvenance:
512
+ return ProviderFactProvenance(
513
+ provider_id=policy.provider_id,
514
+ provider_record_id=record["record_id"],
515
+ provider_record_sha256=record["normalized_sha256"],
516
+ json_paths=tuple(sorted(set(paths))),
517
+ )
518
+
519
+ return build
520
+
521
+
522
+ def _known_text_fact(value: str, paths: tuple[str, ...], provenance) -> CatalogFact:
523
+ return CatalogFact(state="known", value=value, provenance=provenance(paths))
524
+
525
+
526
+ def _text_fact(values: Mapping[str, Any], path: str, *, maximum: int, provenance) -> CatalogFact:
527
+ value = values.get(path)
528
+ if not isinstance(value, str) or not _safe_text(value, maximum=maximum):
529
+ return CatalogFact(state="unknown")
530
+ return _known_text_fact(value, (path,), provenance)
531
+
532
+
533
+ def _text_tuple_fact(
534
+ values: Mapping[str, Any], paths: tuple[str, ...], *, maximum: int, provenance
535
+ ) -> CatalogFact:
536
+ """Admit exactly the declared values that satisfy the contract; an empty subset is unknown."""
537
+
538
+ admitted: set[str] = set()
539
+ cited: list[str] = []
540
+ for path in paths:
541
+ declared = _declared_strings(values.get(path))
542
+ safe = {item for item in declared if _safe_text(item, maximum=MAX_DECLARED_NAME)}
543
+ if safe:
544
+ admitted |= safe
545
+ cited.append(path)
546
+ if not admitted or len(admitted) > maximum:
547
+ return CatalogFact(state="unknown")
548
+ return CatalogFact(
549
+ state="known", value=tuple(sorted(admitted)), provenance=provenance(tuple(cited))
550
+ )
551
+
552
+
553
+ def _data_formats(values: Mapping[str, Any], policy: AuthoringPolicy, provenance) -> CatalogFact:
554
+ admitted: set[str] = set()
555
+ cited: list[str] = []
556
+ for path in ("$.dcat.distribution[*].format", "$.dcat.distribution[*].mediaType"):
557
+ mapped = {
558
+ policy.data_format_for(item)
559
+ for item in _declared_strings(values.get(path))
560
+ if policy.data_format_for(item) is not None
561
+ }
562
+ if mapped:
563
+ admitted |= {value for value in mapped if value is not None}
564
+ cited.append(path)
565
+ if not admitted:
566
+ return CatalogFact(state="unknown")
567
+ return CatalogFact(
568
+ state="known", value=tuple(sorted(admitted)), provenance=provenance(tuple(cited))
569
+ )
570
+
571
+
572
+ def _rights(
573
+ values: Mapping[str, Any], policy: AuthoringPolicy, provenance
574
+ ) -> tuple[CatalogFact, str]:
575
+ """Map an exact declared license, or flag. Declared prose never authorizes anything."""
576
+
577
+ prose = values.get("$.dcat.rights")
578
+ if isinstance(prose, str) and prose.strip():
579
+ return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_PROSE"
580
+ declared = values.get("$.dcat.license")
581
+ if not isinstance(declared, str) or not declared:
582
+ return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_ABSENT"
583
+ status = policy.rights_for(declared)
584
+ if status is None:
585
+ return CatalogFact(state="unknown"), "AUTHOR_RIGHTS_UNMAPPED"
586
+ fact = CatalogFact(state="known", value=status, provenance=provenance(("$.dcat.license",)))
587
+ reason = {
588
+ "approved": "AUTHOR_OK",
589
+ "conditional": "AUTHOR_RIGHTS_CONDITIONAL",
590
+ "prohibited": "AUTHOR_RIGHTS_PROHIBITED",
591
+ "unclear": "AUTHOR_RIGHTS_UNMAPPED",
592
+ }[status]
593
+ return fact, reason
594
+
595
+
596
+ def _layer_texts(
597
+ entry: CatalogEntryV2, policy: AuthoringPolicy
598
+ ) -> tuple[tuple[tuple[str, str], ...], tuple[str, ...]]:
599
+ """Render exactly four layer texts, using canonical state tokens instead of invented prose."""
600
+
601
+ rendered = {
602
+ "description": f"{entry.title.value}\n{_one(entry.description)}",
603
+ "metadata": " ".join(
604
+ (
605
+ _one(entry.publisher),
606
+ *_many(entry.spatial_scope),
607
+ *_many(entry.data_formats),
608
+ _one(entry.access_kind),
609
+ _one(entry.rights),
610
+ )
611
+ ),
612
+ "columns": " ".join((*_many(entry.declared_columns), *_many(entry.declared_vocabulary))),
613
+ "profiles": " ".join(
614
+ (
615
+ "rows",
616
+ _one(entry.declared_row_count),
617
+ "authentication",
618
+ _one(entry.authentication_required),
619
+ "profiles",
620
+ *_many(entry.profiles),
621
+ "formats",
622
+ *_many(entry.data_formats),
623
+ )
624
+ ),
625
+ }
626
+ texts: list[tuple[str, str]] = []
627
+ truncated: list[str] = []
628
+ for layer in EMBEDDING_LAYERS:
629
+ text, was_truncated = _truncate_utf8(rendered[layer], policy.max_layer_text_bytes)
630
+ texts.append((layer, text))
631
+ if was_truncated:
632
+ truncated.append(layer)
633
+ return tuple(texts), tuple(truncated)
634
+
635
+
636
+ def _one(fact: CatalogFact) -> str:
637
+ if fact.state == "not_applicable":
638
+ return LAYER_TEXT_NOT_APPLICABLE
639
+ if fact.state != "known":
640
+ return LAYER_TEXT_UNKNOWN
641
+ if type(fact.value) is bool:
642
+ return "true" if fact.value else "false"
643
+ return str(fact.value)
644
+
645
+
646
+ def _many(fact: CatalogFact) -> tuple[str, ...]:
647
+ if fact.state == "not_applicable":
648
+ return (LAYER_TEXT_NOT_APPLICABLE,)
649
+ if fact.state != "known" or not isinstance(fact.value, tuple):
650
+ return (LAYER_TEXT_UNKNOWN,)
651
+ return fact.value
652
+
653
+
654
+ def _truncate_utf8(value: str, maximum: int) -> tuple[str, bool]:
655
+ """Cut to at most ``maximum`` UTF-8 bytes without ever splitting a code point."""
656
+
657
+ encoded = value.encode("utf-8")
658
+ if len(encoded) <= maximum:
659
+ return value, False
660
+ cut = encoded[:maximum]
661
+ while cut:
662
+ try:
663
+ return cut.decode("utf-8"), True
664
+ except UnicodeDecodeError:
665
+ cut = cut[:-1]
666
+ return "", True
667
+
668
+
669
+ def _declared_strings(value: Any) -> tuple[str, ...]:
670
+ """Read a declared scalar or array of strings; every other shape declares nothing."""
671
+
672
+ if isinstance(value, str):
673
+ return (value,)
674
+ if isinstance(value, list):
675
+ return tuple(item for item in value if isinstance(item, str))
676
+ return ()
677
+
678
+
679
+ def _safe_text(value: str, *, maximum: int) -> bool:
680
+ if not value or len(value) > maximum:
681
+ return False
682
+ for character in value:
683
+ codepoint = ord(character)
684
+ if (
685
+ 0xD800 <= codepoint <= 0xDFFF
686
+ or codepoint < 0x20
687
+ or 0x7F <= codepoint <= 0x9F
688
+ or (character.isspace() and character != " ")
689
+ ):
690
+ return False
691
+ return True
692
+
693
+
694
+ def _base_path(path: str) -> str:
695
+ return path[:-3] if path.endswith("[*]") else path
696
+
697
+
698
+ def _is_digest(value: Any) -> bool:
699
+ return (
700
+ isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
701
+ )