mostlyright-data 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (314) hide show
  1. mostlyright/data_harness/__init__.py +158 -0
  2. mostlyright/data_harness/acquisition/__init__.py +55 -0
  3. mostlyright/data_harness/acquisition/http.py +2773 -0
  4. mostlyright/data_harness/acquisition/parsing.py +809 -0
  5. mostlyright/data_harness/acquisition/ranges.py +495 -0
  6. mostlyright/data_harness/acquisition/result_download.py +360 -0
  7. mostlyright/data_harness/acquisition/retention_admission.py +248 -0
  8. mostlyright/data_harness/acquisition/sandbox.py +4888 -0
  9. mostlyright/data_harness/acquisition/url_policy.py +530 -0
  10. mostlyright/data_harness/agent_runtime.py +2743 -0
  11. mostlyright/data_harness/assets/logo-ink.svg +31 -0
  12. mostlyright/data_harness/backends/__init__.py +28 -0
  13. mostlyright/data_harness/backends/pandas_backend.py +350 -0
  14. mostlyright/data_harness/backends/polars_backend.py +366 -0
  15. mostlyright/data_harness/backends/protocol.py +124 -0
  16. mostlyright/data_harness/backends/reference.py +83 -0
  17. mostlyright/data_harness/backends/registry.py +55 -0
  18. mostlyright/data_harness/backends/restrictions.py +126 -0
  19. mostlyright/data_harness/canonical.py +333 -0
  20. mostlyright/data_harness/catalog_job.py +625 -0
  21. mostlyright/data_harness/cli.py +5398 -0
  22. mostlyright/data_harness/contracts.py +53 -0
  23. mostlyright/data_harness/coordinator.py +1307 -0
  24. mostlyright/data_harness/deploy.py +924 -0
  25. mostlyright/data_harness/deploy_target.py +312 -0
  26. mostlyright/data_harness/deployment_evidence.py +1067 -0
  27. mostlyright/data_harness/event_presentation.py +576 -0
  28. mostlyright/data_harness/events.py +2152 -0
  29. mostlyright/data_harness/fast_delimited.py +239 -0
  30. mostlyright/data_harness/fleet.py +237 -0
  31. mostlyright/data_harness/formats.py +236 -0
  32. mostlyright/data_harness/governors.py +1163 -0
  33. mostlyright/data_harness/hosted_bootstrap.py +972 -0
  34. mostlyright/data_harness/hosted_crawler.py +1115 -0
  35. mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
  36. mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
  37. mostlyright/data_harness/hosted_crawler_job.py +1277 -0
  38. mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
  39. mostlyright/data_harness/hosted_dataset.py +1500 -0
  40. mostlyright/data_harness/hosted_deploy.py +3037 -0
  41. mostlyright/data_harness/hosted_handoff.py +62 -0
  42. mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
  43. mostlyright/data_harness/hosted_ingestion_job.py +356 -0
  44. mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
  45. mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
  46. mostlyright/data_harness/hosted_session_worker.py +3554 -0
  47. mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
  48. mostlyright/data_harness/hosted_worker.py +6784 -0
  49. mostlyright/data_harness/ingestion/__init__.py +56 -0
  50. mostlyright/data_harness/ingestion/contracts.py +461 -0
  51. mostlyright/data_harness/ingestion/faults.py +42 -0
  52. mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
  53. mostlyright/data_harness/ingestion/spool.py +130 -0
  54. mostlyright/data_harness/ingestion/store.py +885 -0
  55. mostlyright/data_harness/key_seam.py +434 -0
  56. mostlyright/data_harness/linux_process_boundary.py +262 -0
  57. mostlyright/data_harness/local_contracts.py +2880 -0
  58. mostlyright/data_harness/local_search/__init__.py +5 -0
  59. mostlyright/data_harness/local_search/build_index.py +1087 -0
  60. mostlyright/data_harness/local_search/contracts.py +920 -0
  61. mostlyright/data_harness/local_search/query_trace.py +266 -0
  62. mostlyright/data_harness/local_search/retrieval.py +700 -0
  63. mostlyright/data_harness/local_search/sealed.py +474 -0
  64. mostlyright/data_harness/local_search/service.py +784 -0
  65. mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
  66. mostlyright/data_harness/nbrender/__init__.py +12 -0
  67. mostlyright/data_harness/nbrender/chrome.py +359 -0
  68. mostlyright/data_harness/nbrender/code_body.py +266 -0
  69. mostlyright/data_harness/nbrender/document.py +407 -0
  70. mostlyright/data_harness/nbrender/frame.py +275 -0
  71. mostlyright/data_harness/nbrender/interactive.py +337 -0
  72. mostlyright/data_harness/nbrender/markdown_body.py +477 -0
  73. mostlyright/data_harness/nbrender/mr_components.py +134 -0
  74. mostlyright/data_harness/nbrender/outputs_data.py +595 -0
  75. mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
  76. mostlyright/data_harness/nbrender/outputs_source.py +260 -0
  77. mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
  78. mostlyright/data_harness/nbrender/outputs_text.py +400 -0
  79. mostlyright/data_harness/nbrender/parse.py +394 -0
  80. mostlyright/data_harness/nbrender/status.py +40 -0
  81. mostlyright/data_harness/nbrender/tokens.py +1295 -0
  82. mostlyright/data_harness/notebook.py +1710 -0
  83. mostlyright/data_harness/offline.py +2049 -0
  84. mostlyright/data_harness/operation_registry.py +1007 -0
  85. mostlyright/data_harness/operator_setup.py +239 -0
  86. mostlyright/data_harness/pipeline.py +6428 -0
  87. mostlyright/data_harness/plan_graph.py +2026 -0
  88. mostlyright/data_harness/preparation/__init__.py +104 -0
  89. mostlyright/data_harness/preparation/contracts.py +1017 -0
  90. mostlyright/data_harness/preparation/engine.py +221 -0
  91. mostlyright/data_harness/preparation/errors.py +14 -0
  92. mostlyright/data_harness/preparation/gates.py +751 -0
  93. mostlyright/data_harness/preparation/joins.py +574 -0
  94. mostlyright/data_harness/preparation/profile.py +384 -0
  95. mostlyright/data_harness/preparation/table.py +217 -0
  96. mostlyright/data_harness/preparation/transforms.py +568 -0
  97. mostlyright/data_harness/progress_events.py +534 -0
  98. mostlyright/data_harness/readers/__init__.py +46 -0
  99. mostlyright/data_harness/readers/containers.py +963 -0
  100. mostlyright/data_harness/readers/contracts.py +542 -0
  101. mostlyright/data_harness/readers/delimited.py +257 -0
  102. mostlyright/data_harness/readers/grib2/__init__.py +33 -0
  103. mostlyright/data_harness/readers/grib2/admission.py +722 -0
  104. mostlyright/data_harness/readers/grib2/decode.py +1009 -0
  105. mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
  106. mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
  107. mostlyright/data_harness/readers/json_tabular.py +485 -0
  108. mostlyright/data_harness/readers/registry.py +514 -0
  109. mostlyright/data_harness/readers/samples/README.md +110 -0
  110. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
  111. mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
  112. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
  113. mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
  114. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
  115. mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
  116. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
  117. mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
  118. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
  119. mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
  120. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  121. mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
  122. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
  123. mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
  124. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
  125. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
  126. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
  127. mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
  128. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
  129. mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
  130. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
  131. mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
  132. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
  133. mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
  134. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
  135. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
  136. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  137. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  138. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
  139. mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
  140. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
  141. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
  142. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
  143. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
  144. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  145. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
  146. mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
  147. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
  148. mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
  149. mostlyright/data_harness/readers/samples.py +582 -0
  150. mostlyright/data_harness/readers/spreadsheet.py +803 -0
  151. mostlyright/data_harness/readers/tabular.py +510 -0
  152. mostlyright/data_harness/recipe.py +5321 -0
  153. mostlyright/data_harness/repair/__init__.py +78 -0
  154. mostlyright/data_harness/repair/adapters.py +274 -0
  155. mostlyright/data_harness/repair/contracts.py +872 -0
  156. mostlyright/data_harness/repair/coordinator.py +1099 -0
  157. mostlyright/data_harness/repair/errors.py +16 -0
  158. mostlyright/data_harness/review.py +2533 -0
  159. mostlyright/data_harness/rowset.py +283 -0
  160. mostlyright/data_harness/serving.py +1975 -0
  161. mostlyright/data_harness/serving_edge.py +590 -0
  162. mostlyright/data_harness/serving_http.py +1031 -0
  163. mostlyright/data_harness/session_probes.py +759 -0
  164. mostlyright/data_harness/signing.py +101 -0
  165. mostlyright/data_harness/source_discovery.py +898 -0
  166. mostlyright/data_harness/sources/__init__.py +209 -0
  167. mostlyright/data_harness/sources/_adapter_steps.py +213 -0
  168. mostlyright/data_harness/sources/adapters.py +1214 -0
  169. mostlyright/data_harness/sources/cadence.py +1428 -0
  170. mostlyright/data_harness/sources/cadence_emission.py +453 -0
  171. mostlyright/data_harness/sources/cadence_history.py +546 -0
  172. mostlyright/data_harness/sources/catalog/__init__.py +17 -0
  173. mostlyright/data_harness/sources/catalog/admission.py +477 -0
  174. mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
  175. mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
  176. mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
  177. mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
  178. mostlyright/data_harness/sources/catalog/channel.py +523 -0
  179. mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
  180. mostlyright/data_harness/sources/catalog/contracts.py +825 -0
  181. mostlyright/data_harness/sources/catalog/coverage.py +137 -0
  182. mostlyright/data_harness/sources/catalog/delta.py +1340 -0
  183. mostlyright/data_harness/sources/catalog/embedding.py +532 -0
  184. mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
  185. mostlyright/data_harness/sources/catalog/fill.py +3889 -0
  186. mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
  187. mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
  188. mostlyright/data_harness/sources/catalog/gating.py +374 -0
  189. mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
  190. mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
  191. mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
  192. mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
  193. mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
  194. mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
  195. mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
  196. mostlyright/data_harness/sources/catalog/health.py +447 -0
  197. mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
  198. mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
  199. mostlyright/data_harness/sources/catalog/neural.py +1618 -0
  200. mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
  201. mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
  202. mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
  203. mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
  204. mostlyright/data_harness/sources/catalog/recommend.py +171 -0
  205. mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
  206. mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
  207. mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
  208. mostlyright/data_harness/sources/catalog/sealed.py +560 -0
  209. mostlyright/data_harness/sources/catalog/search.py +230 -0
  210. mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
  211. mostlyright/data_harness/sources/catalog/update.py +891 -0
  212. mostlyright/data_harness/sources/collections.py +815 -0
  213. mostlyright/data_harness/sources/contracts.py +2223 -0
  214. mostlyright/data_harness/sources/deletion.py +761 -0
  215. mostlyright/data_harness/sources/fitness.py +162 -0
  216. mostlyright/data_harness/sources/governance.py +163 -0
  217. mostlyright/data_harness/sources/hosted.py +173 -0
  218. mostlyright/data_harness/sources/integration.py +218 -0
  219. mostlyright/data_harness/sources/range_reader.py +418 -0
  220. mostlyright/data_harness/sources/registry.py +514 -0
  221. mostlyright/data_harness/sources/rights_rule.py +59 -0
  222. mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
  223. mostlyright/data_harness/sources/sports.py +521 -0
  224. mostlyright/data_harness/sources/stream.py +524 -0
  225. mostlyright/data_harness/sources/stream_connector.py +418 -0
  226. mostlyright/data_harness/sources/stream_recorder.py +1404 -0
  227. mostlyright/data_harness/studio_boundary.py +2019 -0
  228. mostlyright/data_harness/thin/__init__.py +37 -0
  229. mostlyright/data_harness/thin/acquire.py +1137 -0
  230. mostlyright/data_harness/thin/acquire_cancel.py +579 -0
  231. mostlyright/data_harness/thin/approvals.py +617 -0
  232. mostlyright/data_harness/thin/commands.py +406 -0
  233. mostlyright/data_harness/thin/download.py +194 -0
  234. mostlyright/data_harness/thin/narrative.py +589 -0
  235. mostlyright/data_harness/thin/parity.py +1070 -0
  236. mostlyright/data_harness/thin/propose.py +2759 -0
  237. mostlyright/data_harness/thin/research.py +1663 -0
  238. mostlyright/data_harness/thin/router.py +924 -0
  239. mostlyright/data_harness/thin/runs.py +519 -0
  240. mostlyright/data_harness/thin/session.py +281 -0
  241. mostlyright/data_harness/thin/stream.py +501 -0
  242. mostlyright/data_harness/thin/transport.py +187 -0
  243. mostlyright/data_harness/thin/vocabulary.py +368 -0
  244. mostlyright/data_harness/thin/workers.py +164 -0
  245. mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
  246. mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
  247. mostlyright/data_harness/unit_flow.py +927 -0
  248. mostlyright/data_harness/units.py +572 -0
  249. mostlyright/data_harness/ux/__init__.py +9 -0
  250. mostlyright/data_harness/ux/approve.py +485 -0
  251. mostlyright/data_harness/ux/author_yaml.py +597 -0
  252. mostlyright/data_harness/ux/cloud_auth.py +447 -0
  253. mostlyright/data_harness/ux/commands/__init__.py +260 -0
  254. mostlyright/data_harness/ux/commands/approve.py +136 -0
  255. mostlyright/data_harness/ux/commands/auth.py +744 -0
  256. mostlyright/data_harness/ux/commands/author.py +79 -0
  257. mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
  258. mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
  259. mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
  260. mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
  261. mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
  262. mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
  263. mostlyright/data_harness/ux/commands/deploy.py +134 -0
  264. mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
  265. mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
  266. mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
  267. mostlyright/data_harness/ux/commands/diff.py +74 -0
  268. mostlyright/data_harness/ux/commands/index.py +84 -0
  269. mostlyright/data_harness/ux/commands/inventory.py +47 -0
  270. mostlyright/data_harness/ux/commands/list_builds.py +143 -0
  271. mostlyright/data_harness/ux/commands/login.py +63 -0
  272. mostlyright/data_harness/ux/commands/peek.py +236 -0
  273. mostlyright/data_harness/ux/commands/plan_check.py +90 -0
  274. mostlyright/data_harness/ux/commands/preflight.py +97 -0
  275. mostlyright/data_harness/ux/commands/record.py +107 -0
  276. mostlyright/data_harness/ux/commands/review_setup.py +47 -0
  277. mostlyright/data_harness/ux/commands/search.py +440 -0
  278. mostlyright/data_harness/ux/commands/show.py +61 -0
  279. mostlyright/data_harness/ux/commands/whoami.py +37 -0
  280. mostlyright/data_harness/ux/credential_native.py +551 -0
  281. mostlyright/data_harness/ux/credential_store.py +1055 -0
  282. mostlyright/data_harness/ux/credentials.py +631 -0
  283. mostlyright/data_harness/ux/diffing.py +444 -0
  284. mostlyright/data_harness/ux/headline.py +671 -0
  285. mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
  286. mostlyright/data_harness/ux/hosted_run_status.py +619 -0
  287. mostlyright/data_harness/ux/inventory.py +427 -0
  288. mostlyright/data_harness/ux/local_review.py +375 -0
  289. mostlyright/data_harness/ux/login.py +691 -0
  290. mostlyright/data_harness/ux/path_kind.py +147 -0
  291. mostlyright/data_harness/ux/peek.py +1000 -0
  292. mostlyright/data_harness/ux/plain_file.py +178 -0
  293. mostlyright/data_harness/ux/plan_check.py +311 -0
  294. mostlyright/data_harness/ux/preflight.py +918 -0
  295. mostlyright/data_harness/ux/readers.py +1124 -0
  296. mostlyright/data_harness/ux/remediation.py +2195 -0
  297. mostlyright/data_harness/ux/render.py +657 -0
  298. mostlyright/data_harness/ux/workload.py +1077 -0
  299. mostlyright/data_harness/viewer.py +3713 -0
  300. mostlyright/data_harness/visual_run/__init__.py +83 -0
  301. mostlyright/data_harness/visual_run/authoring.py +235 -0
  302. mostlyright/data_harness/visual_run/contracts.py +673 -0
  303. mostlyright/data_harness/visual_run/materialize.py +486 -0
  304. mostlyright/data_harness/visual_run/observations.py +874 -0
  305. mostlyright/data_harness/visual_run/query.py +259 -0
  306. mostlyright/data_harness/visual_run/reducer.py +280 -0
  307. mostlyright/data_harness/visual_run/sdk.py +892 -0
  308. mostlyright/data_harness/visual_run/store.py +584 -0
  309. mostlyright/data_harness/visual_run/transport.py +239 -0
  310. mostlyright/data_harness/watch.py +2999 -0
  311. mostlyright_data-0.9.0.dist-info/METADATA +607 -0
  312. mostlyright_data-0.9.0.dist-info/RECORD +314 -0
  313. mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
  314. mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
@@ -0,0 +1,906 @@
1
+ """Group D · rich outputs (D19, D20, D21, D22): DataFrame, figure, rich HTML, JSON viewer.
2
+
3
+ Routed to from ``document._render_output``. The DataFrame renderer (D19) parses the pandas HTML
4
+ repr with the stdlib :class:`html.parser.HTMLParser` — never eval, never pass-through — and re-emits
5
+ its own table DOM: index column, signed-numeric coloring with the true minus U+2212, tabular-nums,
6
+ a first-10/last-10 ellipsis row over 20 rows, a frozen index over 15 columns, and a footer caption.
7
+ Rich HTML (D21) is sandboxed by parse-and-rebuild: the untrusted markup is parsed with the stdlib
8
+ ``HTMLParser`` and only an allowlist of elements/attributes is re-serialized — everything else is
9
+ dropped (dangerous subtrees discarded, unknown tags unwrapped to escaped text), which structurally
10
+ removes ``on*`` handlers, ``javascript:`` URLs, ``position: fixed``, and forgotten tags rather than
11
+ regex-matching them; Styler gradient backgrounds are remapped to the cobalt tint ramp. Figures
12
+ (D20) render as a passive image in both cases: a charset-filtered base64 PNG data URI, or the SVG
13
+ payload base64ed into an ``image/svg+xml`` data URI — never inline SVG markup (see the SVG rule
14
+ below). The JSON viewer (D22) expands two levels and collapses deeper nodes behind ``{n}`` /
15
+ ``[n]`` counts. Every value from notebook JSON is escaped at the point of interpolation through
16
+ :func:`parse.esc`; the D21 body is the one deliberately un-escaped surface, which is why it is
17
+ sandboxed instead.
18
+
19
+ One rule governs SVG everywhere in this module: **untrusted SVG is only ever a passive image**.
20
+ It is never inlined (D20), and untrusted markup may never name an SVG source of its own
21
+ (``_safe_img_src`` refuses ``data:image/svg+xml`` in D21, keeping that allowlist minimal). No
22
+ blocklist over raw SVG markup would be sound — ``/`` is a legal attribute separator, and ``href``
23
+ values are entity- and control-char-decoded before the scheme is read, so ``<animate/onbegin=…>``
24
+ and ``&#x6A;avascript:`` slip past any such filter. Inside ``<img>`` an SVG instead renders in the
25
+ SVG spec's secure static mode: no scripting, no external subresource loads, no interactivity.
26
+ ``.nb-fig img`` carries the same ``width: 100%; height: auto; display: block`` sizing an inline
27
+ ``<svg>`` would have.
28
+
29
+ A D19 numeric cell is colored only when it carries an explicit sign — a leading ``+`` (cobalt) or
30
+ ``-``/``U+2212`` (rust); any zero renders grey and a bare positive (``34``) stays ink, so a clean,
31
+ no-error DataFrame keeps cobalt to the prompt column alone.
32
+
33
+ The D19 wide-frame index freeze and the D21 Styler gradient ramp are emitted as hooks only
34
+ (``nb-df--wide``, ``var(--nb-styler-g0..3)``); their rules and hexes live in ``tokens.build_css``.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ import base64
40
+ import html
41
+ import json
42
+ import re
43
+ from html.parser import HTMLParser
44
+ from typing import Any
45
+
46
+ from mostlyright.data_harness.nbrender.parse import Output, RenderContext, esc
47
+ from mostlyright.data_harness.nbrender.status import is_status_column, render_status
48
+
49
+ # The sanctioned iconography (CONTRACT.md). Kept as named constants so no stray
50
+ # decorative unicode can slip into output and every glyph is auditable.
51
+ _ELLIPSIS = "⋯" # midline horizontal ellipsis · D19 truncation row
52
+ # true minus · D19 negative numerics (RUF001: U+2212 is required here, never a hyphen)
53
+ _TRUE_MINUS = "−" # noqa: RUF001
54
+ _CARET_OPEN = "▾" # down-pointing triangle · D22 expanded node
55
+ _CARET_CLOSED = "▸" # right-pointing triangle · D22 collapsed node
56
+ # multiplication sign · D19 caption dimensions (RUF001: U+00D7 is required here, never "x")
57
+ _TIMES = "×" # noqa: RUF001
58
+ _MIDDOT = "·" # middot · D19 caption separator
59
+
60
+ _TRUNCATE_ROWS = 20 # over this many rows -> first 10 / last 10 with an ellipsis row
61
+ _HEAD_TAIL = 10
62
+ _FREEZE_COLS = 15 # over this many data columns -> freeze the index column and scroll
63
+ _JSON_EXPAND_LEVELS = 2 # static mode expands two levels; deeper nodes collapse to a count
64
+
65
+ # ==============================================================================================
66
+ # D19 · DataFrame
67
+ # ==============================================================================================
68
+
69
+
70
+ class _TableParser(HTMLParser):
71
+ """Collect ``thead``/``tbody`` rows as ``(tag, text)`` cells; text is decoded, never trusted.
72
+
73
+ Only table structure is honored — any markup inside a cell (a stray ``<script>`` in a value)
74
+ contributes its text via :meth:`handle_data` and is dropped as a tag, so re-emitting the cell
75
+ text through :func:`esc` cannot carry an injection through.
76
+ """
77
+
78
+ def __init__(self) -> None:
79
+ super().__init__(convert_charrefs=True)
80
+ self.header_rows: list[list[tuple[str, str, str]]] = []
81
+ self.body_rows: list[list[tuple[str, str, str]]] = []
82
+ self.header_row_attrs: list[str] = []
83
+ self.body_row_attrs: list[str] = []
84
+ self.table_attrs = ""
85
+ self.thead_attrs = ""
86
+ self.tbody_attrs = ""
87
+ self.seen_thead = False
88
+ self.seen_tbody = False
89
+ self._section: str | None = None
90
+ self._row: list[tuple[str, str, str]] | None = None
91
+ self._row_attrs = ""
92
+ self._cell: list[Any] | None = None # [tag, [chunks], serialized attrs]
93
+ self._suppress = 0 # depth inside a <script>/<style> whose text must be dropped
94
+ self.table_count = 0
95
+ self.table_depth = 0
96
+ self.simple_table = True
97
+
98
+ def handle_starttag(self, tag: str, attrs: Any) -> None:
99
+ if any(name.lower() in ("colspan", "rowspan", "span") for name, _value in attrs):
100
+ self.simple_table = False
101
+ if tag == "table":
102
+ if self.table_depth or self._section is not None or self._row is not None:
103
+ self.simple_table = False
104
+ self.table_count += 1
105
+ self.table_depth += 1
106
+ self.table_attrs = _serialize_attrs(tag, attrs)
107
+ if self.table_depth != 1:
108
+ self.simple_table = False
109
+ elif tag in ("script", "style"):
110
+ self.simple_table = False
111
+ self._suppress += 1
112
+ elif tag == "thead":
113
+ if (
114
+ self.table_depth != 1
115
+ or self._section is not None
116
+ or self._row is not None
117
+ or self.seen_thead
118
+ or self.seen_tbody
119
+ ):
120
+ self.simple_table = False
121
+ self.seen_thead = True
122
+ self._section = "thead"
123
+ self.thead_attrs = _serialize_attrs(tag, attrs)
124
+ elif tag == "tbody":
125
+ if (
126
+ self.table_depth != 1
127
+ or self._section is not None
128
+ or self._row is not None
129
+ or not self.seen_thead
130
+ or self.seen_tbody
131
+ ):
132
+ self.simple_table = False
133
+ self.seen_tbody = True
134
+ self._section = "tbody"
135
+ self.tbody_attrs = _serialize_attrs(tag, attrs)
136
+ elif tag == "tr":
137
+ if (
138
+ self.table_depth != 1
139
+ or self._section not in ("thead", "tbody")
140
+ or self._row is not None
141
+ or self._cell is not None
142
+ ):
143
+ self.simple_table = False
144
+ self._row = []
145
+ self._row_attrs = _serialize_attrs(tag, attrs)
146
+ elif tag in ("th", "td") and self._row is not None:
147
+ if self._cell is not None:
148
+ self.simple_table = False
149
+ self._cell = [tag, [], _serialize_attrs(tag, attrs)]
150
+ elif tag in ("th", "td"):
151
+ self.simple_table = False
152
+ elif tag not in ("thead", "tbody", "tr", "th", "td"):
153
+ self.simple_table = False
154
+
155
+ def handle_startendtag(self, tag: str, attrs: Any) -> None:
156
+ self.simple_table = False
157
+ if tag in ("th", "td") and self._row is not None:
158
+ self._row.append((tag, "", _serialize_attrs(tag, attrs)))
159
+
160
+ def handle_data(self, data: str) -> None:
161
+ if self._suppress:
162
+ return
163
+ if self._cell is not None:
164
+ self._cell[1].append(data)
165
+ elif data.strip():
166
+ self.simple_table = False
167
+
168
+ def handle_endtag(self, tag: str) -> None:
169
+ if tag in ("script", "style"):
170
+ if self._suppress:
171
+ self._suppress -= 1
172
+ return
173
+ if tag == "table":
174
+ if self.table_depth != 1 or self._section is not None or self._row is not None:
175
+ self.simple_table = False
176
+ self.table_depth = max(0, self.table_depth - 1)
177
+ elif tag in ("th", "td") and self._cell is not None and self._row is not None:
178
+ if self._cell[0] != tag:
179
+ self.simple_table = False
180
+ self._row.append((self._cell[0], "".join(self._cell[1]), self._cell[2]))
181
+ self._cell = None
182
+ elif tag == "tr" and self._row is not None:
183
+ if self._section not in ("thead", "tbody") or self._cell is not None:
184
+ self.simple_table = False
185
+ if self._section == "thead":
186
+ self.header_rows.append(self._row)
187
+ self.header_row_attrs.append(self._row_attrs)
188
+ elif self._section == "tbody":
189
+ self.body_rows.append(self._row)
190
+ self.body_row_attrs.append(self._row_attrs)
191
+ elif not self.header_rows and not self.body_rows:
192
+ self.header_rows.append(self._row)
193
+ self.header_row_attrs.append(self._row_attrs)
194
+ else:
195
+ self.body_rows.append(self._row)
196
+ self.body_row_attrs.append(self._row_attrs)
197
+ self._row = None
198
+ self._row_attrs = ""
199
+ elif tag in ("thead", "tbody"):
200
+ if self._section != tag or self._row is not None or self._cell is not None:
201
+ self.simple_table = False
202
+ self._section = None
203
+ elif tag in ("th", "td", "tr"):
204
+ self.simple_table = False
205
+ elif tag not in ("th", "td", "tr"):
206
+ self.simple_table = False
207
+
208
+ def handle_comment(self, data: str) -> None:
209
+ self.simple_table = False
210
+
211
+
212
+ # The multiplication-sign alternative is deliberate: pandas writes the frame caption with U+00D7.
213
+ _FRAME_RE = re.compile(r"(\d+)\s*rows?\s*[×xX]\s*(\d+)\s*column", re.IGNORECASE) # noqa: RUF001
214
+ _NAN_WORDS = frozenset({"nan", "inf", "-inf", "+inf", "infinity", "-infinity", "+infinity"})
215
+
216
+
217
+ def _parse_number(text: str) -> tuple[str, float] | None:
218
+ """Return ``(original, value)`` when ``text`` is a plain number, else ``None``.
219
+
220
+ Commas, a trailing percent, and a true-minus sign are tolerated; ``nan``/``inf`` words are
221
+ treated as strings so they are not miscolored as signed numerics.
222
+ """
223
+ core = text.strip()
224
+ if not core:
225
+ return None
226
+ candidate = core.rstrip("%").replace(",", "").replace(_TRUE_MINUS, "-")
227
+ if candidate.lower() in _NAN_WORDS:
228
+ return None
229
+ try:
230
+ value = float(candidate)
231
+ except ValueError:
232
+ return None
233
+ return core, value
234
+
235
+
236
+ def _cell_render(text: str, column: str = "") -> tuple[str, str]:
237
+ """Classify one body data cell -> ``(css_class, display_html)``.
238
+
239
+ Numeric cells are right-aligned tabular-nums; a *signed* numeric is colored (positive cobalt,
240
+ negative rust, zero grey) and negatives render with the true minus. Everything else is a
241
+ nowrap string cell.
242
+ """
243
+ if is_status_column(column):
244
+ status = render_status(text)
245
+ if status is not None:
246
+ return "nb-df-str", status
247
+ parsed = _parse_number(text)
248
+ if parsed is None:
249
+ return "nb-df-str", esc(text.strip())
250
+ original, value = parsed
251
+ classes = ["nb-df-num"]
252
+ display = original
253
+ if display[:1] == "-":
254
+ display = _TRUE_MINUS + display[1:]
255
+ if value == 0:
256
+ classes.append("nb-df-zero")
257
+ elif value < 0:
258
+ classes.append("nb-df-neg")
259
+ elif original[:1] == "+":
260
+ classes.append("nb-df-pos")
261
+ return " ".join(classes), esc(display)
262
+
263
+
264
+ def _split_row(row: list[tuple[str, str, str]]) -> tuple[list[str], list[str]]:
265
+ """Leading ``th`` cells are the index; the rest are data cells."""
266
+ index: list[str] = []
267
+ data: list[str] = []
268
+ seen_data = False
269
+ for tag, text, _attrs in row:
270
+ if tag == "th" and not seen_data:
271
+ index.append(text)
272
+ else:
273
+ seen_data = True
274
+ data.append(text)
275
+ return index, data
276
+
277
+
278
+ def _index_col_count(body_split: list[tuple[list[str], list[str]]]) -> int:
279
+ """Number of leading index columns, from pandas' structural signal, not a value heuristic.
280
+
281
+ Pandas emits each index level as a leading ``<th>`` inside every ``tbody`` row, so the count of
282
+ leading ``th`` cells is authoritative regardless of whether the index *value* happens to be
283
+ empty (the old first-cell-empty heuristic misfired on a blank first value). The first body row
284
+ that carries any leading ``th`` decides the count. Rows containing only ``td`` elements are an
285
+ index-free frame and therefore have zero index columns.
286
+ """
287
+ for index_cells, _data in body_split:
288
+ if index_cells:
289
+ return len(index_cells)
290
+ return 0
291
+
292
+
293
+ def render_dataframe(out: Output, ctx: RenderContext) -> str:
294
+ """D19 · DataFrame — re-emitted table, signed-number coloring, truncation, footer caption."""
295
+ source = out.text_of("text/html")
296
+ parser = _TableParser()
297
+ try:
298
+ parser.feed(source)
299
+ parser.close()
300
+ except Exception:
301
+ pass
302
+
303
+ body_split = [_split_row(row) for row in parser.body_rows]
304
+ n_index = _index_col_count(body_split)
305
+
306
+ header = parser.header_rows[-1] if parser.header_rows else None
307
+ if header is not None:
308
+ header_texts = [text for _tag, text, _attrs in header]
309
+ columns = header_texts[n_index:]
310
+ else:
311
+ columns = []
312
+ if not columns and body_split:
313
+ columns = ["" for _ in body_split[0][1]]
314
+ data_col_count = len(columns)
315
+
316
+ frame_match = _FRAME_RE.search(source)
317
+ if frame_match:
318
+ frame_rows, frame_cols = int(frame_match.group(1)), int(frame_match.group(2))
319
+ else:
320
+ frame_rows, frame_cols = len(body_split), data_col_count
321
+
322
+ colspan = n_index + data_col_count
323
+
324
+ # Truncation · over 20 rows render first 10 / last 10 with a centered ellipsis row.
325
+ if len(body_split) > _TRUNCATE_ROWS:
326
+ shown = [*body_split[:_HEAD_TAIL], None, *body_split[-_HEAD_TAIL:]]
327
+ disp_rows = _HEAD_TAIL * 2
328
+ else:
329
+ shown = list(body_split)
330
+ disp_rows = len(body_split)
331
+
332
+ head_cells = "".join("<th></th>" for _ in range(n_index)) + "".join(
333
+ f"<th>{esc(col)}</th>" for col in columns
334
+ )
335
+
336
+ body_parts: list[str] = []
337
+ for item in shown:
338
+ if item is None:
339
+ body_parts.append(
340
+ f'<tr><td class="nb-df-ellipsis" colspan="{colspan}">{_ELLIPSIS}</td></tr>'
341
+ )
342
+ continue
343
+ index_cells, data_cells = item
344
+ idx_html = "".join(f'<td class="nb-df-idx">{esc(v)}</td>' for v in index_cells)
345
+ rendered_cells = (
346
+ _cell_render(value, columns[index] if index < len(columns) else "")
347
+ for index, value in enumerate(data_cells)
348
+ )
349
+ cell_html = "".join(f'<td class="{cls}">{disp}</td>' for cls, disp in rendered_cells)
350
+ body_parts.append(f"<tr>{idx_html}{cell_html}</tr>")
351
+
352
+ caption = (
353
+ f"{disp_rows} rows {_TIMES} {data_col_count} columns {_MIDDOT} "
354
+ f"{frame_rows} {_TIMES} {frame_cols} in frame"
355
+ )
356
+ wide = " nb-df--wide" if data_col_count > _FREEZE_COLS else ""
357
+
358
+ return (
359
+ f'<div class="nb-df{wide}">'
360
+ f'<div class="nb-df-scroll"><table>'
361
+ f"<thead><tr>{head_cells}</tr></thead>"
362
+ f"<tbody>{''.join(body_parts)}</tbody>"
363
+ f"</table></div>"
364
+ f'<div class="nb-df-caption">{esc(caption)}</div>'
365
+ f"</div>"
366
+ )
367
+
368
+
369
+ # ==============================================================================================
370
+ # D20 · Figure output
371
+ # ==============================================================================================
372
+
373
+ _B64_STRIP_RE = re.compile(r"[^A-Za-z0-9+/=]")
374
+
375
+ # D20 series line-swatch styles, in series order. Colors reference CSS variables so no literal hex
376
+ # escapes this module; series 2 is dashed and series 4 dotted so the legend survives grayscale.
377
+ _LEGEND_SWATCHES = (
378
+ "stroke:var(--nb-cobalt);stroke-width:2",
379
+ "stroke:var(--nb-ink);stroke-width:2;stroke-dasharray:5 4",
380
+ "stroke:var(--nb-orange);stroke-width:2",
381
+ "stroke:var(--nb-faint);stroke-width:2;stroke-dasharray:1 3",
382
+ )
383
+
384
+
385
+ def _fig_meta(out: Output) -> dict[str, Any]:
386
+ """Figure title/subtitle/legend live under ``metadata.mostlyright`` (mr.mpl_style), else
387
+ flat."""
388
+ meta = out.metadata if isinstance(out.metadata, dict) else {}
389
+ nested = meta.get("mostlyright")
390
+ return nested if isinstance(nested, dict) else meta
391
+
392
+
393
+ def _svg_image(svg: str) -> str:
394
+ """Base64 an SVG payload into a passive ``<img>`` (never inline markup — see the module doc).
395
+
396
+ Inside ``<img>`` an SVG renders in the SVG spec's secure static mode: no scripting, no external
397
+ subresource loads, no interactivity. Every execution vector — ``<script>``, ``<foreignObject>``
398
+ handlers, ``on*`` attributes under any separator, and ``javascript:`` in any obfuscation — is
399
+ therefore inert without the renderer having to enumerate it. Base64 also makes the payload
400
+ attribute-safe by construction: the alphabet cannot close the ``src`` quote.
401
+ """
402
+ if not svg.strip():
403
+ return ""
404
+ blob = base64.b64encode(svg.encode("utf-8", "replace")).decode("ascii")
405
+ return f'<img src="data:image/svg+xml;base64,{blob}" alt="figure output">'
406
+
407
+
408
+ def _clean_base64(value: str) -> str:
409
+ """Keep only base64 characters so an untrusted PNG blob is attribute-safe in the data URI."""
410
+ return _B64_STRIP_RE.sub("", value)
411
+
412
+
413
+ def _fig_legend(series: Any) -> str:
414
+ """Render the optional legend from metadata series names, each with its line swatch."""
415
+ if not isinstance(series, list) or not series:
416
+ return ""
417
+ entries: list[str] = []
418
+ for idx, entry in enumerate(series):
419
+ label = entry.get("name") if isinstance(entry, dict) else entry
420
+ style = _LEGEND_SWATCHES[idx] if idx < len(_LEGEND_SWATCHES) else _LEGEND_SWATCHES[-1]
421
+ swatch = (
422
+ '<svg width="16" height="8" aria-hidden="true">'
423
+ f'<line x1="0" y1="4" x2="16" y2="4" style="{style}"/></svg>'
424
+ )
425
+ entries.append(
426
+ '<span style="display:inline-flex;align-items:center;gap:8px">'
427
+ f"{swatch}{esc(label)}</span>"
428
+ )
429
+ return f'<div class="nb-fig-legend">{"".join(entries)}</div>'
430
+
431
+
432
+ def render_figure(out: Output, ctx: RenderContext) -> str:
433
+ """D20 · figure — hoisted title/subtitle, sanitized SVG or base64 PNG, optional legend."""
434
+ meta = _fig_meta(out)
435
+ head = ""
436
+ title = meta.get("title")
437
+ subtitle = meta.get("subtitle")
438
+ if title:
439
+ head += f'<div class="nb-fig-title">{esc(title)}</div>'
440
+ if subtitle:
441
+ head += f'<div class="nb-fig-sub">{esc(subtitle)}</div>'
442
+
443
+ mimetype = out.primary_mimetype()
444
+ if mimetype == "image/svg+xml":
445
+ image = _svg_image(out.text_of("image/svg+xml"))
446
+ elif mimetype == "image/png":
447
+ blob = _clean_base64(out.text_of("image/png"))
448
+ image = f'<img src="data:image/png;base64,{blob}" alt="figure output">'
449
+ else:
450
+ image = ""
451
+
452
+ legend = _fig_legend(meta.get("legend") or meta.get("series"))
453
+ return f'<div class="nb-fig">{head}{image}{legend}</div>'
454
+
455
+
456
+ # ==============================================================================================
457
+ # D21 · Rich HTML (_repr_html_, Styler)
458
+ # ==============================================================================================
459
+
460
+ _STYLER_HINTS = ("col_heading", "row_heading", 'id="t_', "id='t_")
461
+
462
+ # Parse-and-rebuild sandbox (D21). A blocklist regex is bypassable (slash/backtick attribute
463
+ # separators, entity/whitespace-obfuscated ``javascript:``, and any tag the list forgot); instead
464
+ # we parse the untrusted HTML with the stdlib parser and re-serialize ONLY an allowlist of elements
465
+ # and attributes. Anything not on the allowlist is dropped — script-bearing containers discard
466
+ # their whole subtree, every other unknown tag is unwrapped to its (escaped) text — so a construct
467
+ # the allowlist does not name cannot reach the page in an active form.
468
+ _ALLOWED_TAGS = frozenset(
469
+ {
470
+ "div",
471
+ "span",
472
+ "p",
473
+ "br",
474
+ "hr",
475
+ "b",
476
+ "i",
477
+ "em",
478
+ "strong",
479
+ "u",
480
+ "s",
481
+ "sub",
482
+ "sup",
483
+ "code",
484
+ "pre",
485
+ "blockquote",
486
+ "h1",
487
+ "h2",
488
+ "h3",
489
+ "h4",
490
+ "h5",
491
+ "h6",
492
+ "ul",
493
+ "ol",
494
+ "li",
495
+ "dl",
496
+ "dt",
497
+ "dd",
498
+ "table",
499
+ "thead",
500
+ "tbody",
501
+ "tfoot",
502
+ "tr",
503
+ "th",
504
+ "td",
505
+ "caption",
506
+ "col",
507
+ "colgroup",
508
+ "img",
509
+ "a",
510
+ "details",
511
+ "summary",
512
+ "figure",
513
+ "figcaption",
514
+ }
515
+ )
516
+ # HTML void elements — emitted (if allowed) or dropped at the start tag; never pushed as a frame.
517
+ _VOID_TAGS = frozenset(
518
+ {
519
+ "area",
520
+ "base",
521
+ "br",
522
+ "col",
523
+ "embed",
524
+ "hr",
525
+ "img",
526
+ "input",
527
+ "link",
528
+ "meta",
529
+ "param",
530
+ "source",
531
+ "track",
532
+ "wbr",
533
+ }
534
+ )
535
+ # Non-void elements whose entire subtree is discarded (they can carry executable or style payloads
536
+ # no attribute-stripping would neutralize — e.g. a <script> body or an <object> data island).
537
+ _DROP_TAGS = frozenset(
538
+ {
539
+ "script",
540
+ "style",
541
+ "iframe",
542
+ "object",
543
+ "form",
544
+ "noscript",
545
+ "template",
546
+ "svg",
547
+ "math",
548
+ "head",
549
+ "title",
550
+ "applet",
551
+ "frame",
552
+ "frameset",
553
+ "canvas",
554
+ "audio",
555
+ "video",
556
+ "button",
557
+ "select",
558
+ "textarea",
559
+ "option",
560
+ }
561
+ )
562
+
563
+ _CTRL_RE = re.compile(r"[\x00-\x20\x7f]+")
564
+ _SCHEME_RE = re.compile(r"^[a-z][a-z0-9+.\-]*:")
565
+ _STYLE_HEX_RE = re.compile(r"#([0-9A-Fa-f]{6})")
566
+
567
+
568
+ def _ramp_var(hexes: str) -> str:
569
+ """Map a Styler ``background-color`` hex onto one of four cobalt-ramp CSS variables.
570
+
571
+ Darker source colors (higher gradient value) map to more-saturated ramp steps. The ramp hexes
572
+ live in ``tokens.build_css`` as ``--nb-styler-g0..3``; no literal color is emitted here — only
573
+ the variable reference — and the original hex is dropped.
574
+ """
575
+ red, green, blue = (int(hexes[i : i + 2], 16) for i in (0, 2, 4))
576
+ luminance = 0.299 * red + 0.587 * green + 0.114 * blue
577
+ bucket = min(3, max(0, int((255 - luminance) // 64)))
578
+ return f"var(--nb-styler-g{bucket})"
579
+
580
+
581
+ def _safe_href(value: str) -> str | None:
582
+ """Return an ``a[href]`` value only when it is http/https or a ``#fragment``, else ``None``.
583
+
584
+ Entities are decoded and control/whitespace characters stripped BEFORE the scheme test, so
585
+ ``&#x6A;avascript:``, ``java\\nscript:``, and leading-space tricks all collapse to a bare
586
+ scheme that fails the allowlist and is dropped.
587
+ """
588
+ decoded = html.unescape(value or "")
589
+ probe = _CTRL_RE.sub("", decoded).lower()
590
+ if probe.startswith("#"):
591
+ return decoded.strip()
592
+ if probe.startswith("http://") or probe.startswith("https://"):
593
+ return decoded.strip()
594
+ return None
595
+
596
+
597
+ def _safe_img_src(value: str) -> str | None:
598
+ """Return an ``img[src]`` value only for a safe raster data URI or a scheme-less relative path.
599
+
600
+ ``data:image/svg+xml`` is refused (nested-script risk) — only png/jpeg/gif data URIs pass — as
601
+ is any explicit scheme (``javascript:``/``data:text/html``/…) or a protocol-relative ``//host``.
602
+ """
603
+ decoded = html.unescape(value or "")
604
+ probe = _CTRL_RE.sub("", decoded).lower()
605
+ if probe.startswith(("data:image/png", "data:image/jpeg", "data:image/gif")):
606
+ return decoded.strip()
607
+ # Browsers normalize backslashes to slashes in URLs, so \\host and /\host are
608
+ # protocol-relative too — refuse any leading slash/backslash pair.
609
+ if _SCHEME_RE.match(probe) or probe[:2] in ("//", "\\\\", "/\\", "\\/"):
610
+ return None
611
+ return decoded.strip()
612
+
613
+
614
+ def _sanitize_style(value: str) -> str:
615
+ """Keep only ``background``/``background-color`` hex declarations, remapped to the cobalt ramp.
616
+
617
+ Every other declaration is discarded — which structurally removes ``position: fixed`` and any
618
+ ``url(...)`` exfiltration target rather than trying to pattern-match them out.
619
+ """
620
+ decls: list[str] = []
621
+ for chunk in value.split(";"):
622
+ prop, sep, val = chunk.partition(":")
623
+ if not sep:
624
+ continue
625
+ if prop.strip().lower() in ("background", "background-color"):
626
+ match = _STYLE_HEX_RE.search(val)
627
+ if match:
628
+ decls.append("background-color:" + _ramp_var(match.group(1)))
629
+ return ";".join(decls)
630
+
631
+
632
+ def _serialize_attrs(tag: str, attrs: list[tuple[str, Any]]) -> str:
633
+ """Serialize only the allowlisted, validated attributes for ``tag`` (leading-space per attr)."""
634
+ out: list[str] = []
635
+ for name, raw in attrs:
636
+ value = "" if raw is None else str(raw)
637
+ key = name.lower()
638
+ if key == "class":
639
+ out.append(f' class="{esc(value)}"')
640
+ elif key == "style":
641
+ styled = _sanitize_style(value)
642
+ if styled:
643
+ out.append(f' style="{esc(styled)}"')
644
+ elif key in ("colspan", "rowspan", "scope") and tag in ("th", "td"):
645
+ out.append(f' {key}="{esc(value)}"')
646
+ elif key == "span" and tag in ("col", "colgroup"):
647
+ out.append(f' span="{esc(value)}"')
648
+ elif key in ("width", "height") and tag == "img":
649
+ out.append(f' {key}="{esc(value)}"')
650
+ elif key == "alt" and tag == "img":
651
+ out.append(f' alt="{esc(value)}"')
652
+ elif key == "title":
653
+ out.append(f' title="{esc(value)}"')
654
+ elif key == "href" and tag == "a":
655
+ safe = _safe_href(value)
656
+ if safe is not None:
657
+ out.append(f' href="{esc(safe)}"')
658
+ return "".join(out)
659
+
660
+
661
+ def _render_void(tag: str, attrs: list[tuple[str, Any]]) -> str:
662
+ """Serialize an allowlisted void element (dropping a media element that has no safe source)."""
663
+ if tag not in _ALLOWED_TAGS:
664
+ return ""
665
+ if tag == "img":
666
+ src = next((v for n, v in attrs if n.lower() == "src"), None)
667
+ safe = _safe_img_src(str(src)) if src is not None else None
668
+ if safe is None:
669
+ return "" # media with no safe source is removed, not left as a broken tag
670
+ rest = _serialize_attrs("img", [(n, v) for n, v in attrs if n.lower() != "src"])
671
+ return f'<img src="{esc(safe)}"{rest}>'
672
+ return f"<{tag}{_serialize_attrs(tag, attrs)}>"
673
+
674
+
675
+ class _HtmlSanitizer(HTMLParser):
676
+ """Parse untrusted HTML into a tree and re-serialize only the allowlist (D21 sandbox)."""
677
+
678
+ def __init__(self) -> None:
679
+ super().__init__(convert_charrefs=True)
680
+ self._root: list[str] = []
681
+ self._stack: list[dict[str, Any]] = []
682
+
683
+ def _sink(self) -> list[str]:
684
+ return self._stack[-1]["children"] if self._stack else self._root
685
+
686
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, Any]]) -> None:
687
+ tag = tag.lower()
688
+ if tag in _VOID_TAGS:
689
+ self._sink().append(_render_void(tag, attrs))
690
+ return
691
+ if tag in _ALLOWED_TAGS:
692
+ mode = "keep"
693
+ elif tag in _DROP_TAGS:
694
+ mode = "drop"
695
+ else:
696
+ mode = "unwrap"
697
+ self._stack.append(
698
+ {"tag": tag, "mode": mode, "attrs": _serialize_attrs(tag, attrs), "children": []}
699
+ )
700
+
701
+ def handle_startendtag(self, tag: str, attrs: list[tuple[str, Any]]) -> None:
702
+ tag = tag.lower()
703
+ if tag in _VOID_TAGS:
704
+ self._sink().append(_render_void(tag, attrs))
705
+ elif tag in _ALLOWED_TAGS:
706
+ self._sink().append(f"<{tag}{_serialize_attrs(tag, attrs)}></{tag}>")
707
+ # a self-closed drop/unwrap element has no content and contributes nothing
708
+
709
+ def handle_data(self, data: str) -> None:
710
+ self._sink().append(esc(data))
711
+
712
+ def handle_endtag(self, tag: str) -> None:
713
+ tag = tag.lower()
714
+ for depth in range(len(self._stack) - 1, -1, -1):
715
+ if self._stack[depth]["tag"] == tag:
716
+ while len(self._stack) > depth:
717
+ self._close_top()
718
+ return
719
+ # a stray end tag with no open match is ignored
720
+
721
+ def _close_top(self) -> None:
722
+ frame = self._stack.pop()
723
+ inner = "".join(frame["children"])
724
+ if frame["mode"] == "keep":
725
+ rendered = f"<{frame['tag']}{frame['attrs']}>{inner}</{frame['tag']}>"
726
+ elif frame["mode"] == "unwrap":
727
+ rendered = inner
728
+ else: # drop — the subtree (including any script/style text) is discarded
729
+ rendered = ""
730
+ self._sink().append(rendered)
731
+
732
+ def result(self) -> str:
733
+ while self._stack:
734
+ self._close_top()
735
+ return "".join(self._root)
736
+
737
+
738
+ def _sandbox_html(source: str) -> str:
739
+ """Neutralize untrusted HTML per D21 by parse-and-rebuild against the element/attr allowlist.
740
+
741
+ Unknown or malformed markup degrades to escaped text, never raw pass-through; Styler gradient
742
+ backgrounds are remapped to the cobalt ramp inside :func:`_sanitize_style`.
743
+ """
744
+ parser = _HtmlSanitizer()
745
+ try:
746
+ parser.feed(source)
747
+ parser.close()
748
+ except Exception:
749
+ pass
750
+ return parser.result()
751
+
752
+
753
+ def _simple_status_table(source: str) -> str | None:
754
+ """Re-emit one plain status table with canonical status labels.
755
+
756
+ This path is intentionally narrow. Rich, nested, malformed, or multi-table HTML continues
757
+ through the general D21 sanitizer unchanged.
758
+ """
759
+ parser = _TableParser()
760
+ try:
761
+ parser.feed(source)
762
+ parser.close()
763
+ except Exception:
764
+ return None
765
+ if (
766
+ not parser.simple_table
767
+ or parser.table_count != 1
768
+ or parser.table_depth != 0
769
+ or parser._section is not None
770
+ or parser._row is not None
771
+ or parser._cell is not None
772
+ or not parser.seen_thead
773
+ or not parser.seen_tbody
774
+ or len(parser.header_rows) != 1
775
+ or not parser.body_rows
776
+ ):
777
+ return None
778
+ header = parser.header_rows[0]
779
+ headers = [text for _tag, text, _attrs in header]
780
+ status_columns = [i for i, text in enumerate(headers) if is_status_column(text)]
781
+ if len(status_columns) != 1 or any(len(row) != len(header) for row in parser.body_rows):
782
+ return None
783
+ status_column = status_columns[0]
784
+ head_html = "".join(f"<th{attrs}>{esc(text.strip())}</th>" for _tag, text, attrs in header)
785
+ rows: list[str] = []
786
+ for row_index, row in enumerate(parser.body_rows):
787
+ cells: list[str] = []
788
+ for index, (tag, text, attrs) in enumerate(row):
789
+ display = render_status(text) if index == status_column else None
790
+ cells.append(
791
+ f"<{tag}{attrs}>{display if display is not None else esc(text.strip())}</{tag}>"
792
+ )
793
+ rows.append(f"<tr{parser.body_row_attrs[row_index]}>{''.join(cells)}</tr>")
794
+ return (
795
+ f"<table{parser.table_attrs}><thead{parser.thead_attrs}>"
796
+ f"<tr{parser.header_row_attrs[0]}>{head_html}</tr></thead>"
797
+ f"<tbody{parser.tbody_attrs}>{''.join(rows)}</tbody></table>"
798
+ )
799
+
800
+
801
+ def _detect_producer(source: str) -> str | None:
802
+ """Name the producing class when detectable — only pandas Styler is reliably fingerprinted."""
803
+ lowered = source.lower()
804
+ if any(hint in lowered for hint in _STYLER_HINTS):
805
+ return "pandas Styler"
806
+ return None
807
+
808
+
809
+ def render_html(out: Output, ctx: RenderContext) -> str:
810
+ """D21 · rich HTML / Styler — sandboxed body, mimetype header strip, Styler gradient remap."""
811
+ mimetype = out.primary_mimetype() or "text/html"
812
+ source = out.text_of(mimetype)
813
+ producer = _detect_producer(source)
814
+ body = _simple_status_table(source) or _sandbox_html(source)
815
+
816
+ head = f'<span class="nb-rich-mime">{esc(mimetype)}</span>'
817
+ if producer:
818
+ head += f'<span class="nb-rich-cls">{esc(producer)}</span>'
819
+ return (
820
+ '<div class="nb-rich">'
821
+ f'<div class="nb-rich-head">{head}</div>'
822
+ f'<div class="nb-rich-body">{body}</div>'
823
+ "</div>"
824
+ )
825
+
826
+
827
+ # ==============================================================================================
828
+ # D22 · JSON viewer
829
+ # ==============================================================================================
830
+
831
+
832
+ def _coerce_json(raw: Any) -> Any:
833
+ """A bundle value may already be a JSON object; a string is parsed, or kept as a string leaf."""
834
+ if isinstance(raw, str):
835
+ try:
836
+ return json.loads(raw)
837
+ except (ValueError, TypeError):
838
+ return raw
839
+ return raw
840
+
841
+
842
+ def _json_leaf(value: Any) -> str:
843
+ if isinstance(value, str):
844
+ return f'<span class="nb-json-str">"{esc(value)}"</span>'
845
+ if isinstance(value, bool):
846
+ return f'<span class="nb-json-num">{"true" if value else "false"}</span>'
847
+ if value is None:
848
+ return '<span class="nb-json-num">null</span>'
849
+ if isinstance(value, (int, float)):
850
+ return f'<span class="nb-json-num">{esc(value)}</span>'
851
+ # Any other type (should not occur from json.loads) degrades to an escaped string leaf.
852
+ return f'<span class="nb-json-str">"{esc(value)}"</span>'
853
+
854
+
855
+ def _json_container(
856
+ open_b: str, close_b: str, items: list[tuple[Any, Any]], depth: int, keyed: bool
857
+ ) -> str:
858
+ rows: list[str] = []
859
+ last = len(items) - 1
860
+ for position, (key, value) in enumerate(items):
861
+ comma = "" if position == last else '<span class="nb-json-punct">,</span>'
862
+ key_html = (
863
+ f'<span class="nb-json-key">"{esc(key)}"</span><span class="nb-json-punct">: </span>'
864
+ if keyed
865
+ else ""
866
+ )
867
+ rows.append(
868
+ f'<div style="padding-left:22px">{key_html}{_json_node(value, depth + 1)}{comma}</div>'
869
+ )
870
+ return (
871
+ f'<span class="nb-json-caret">{_CARET_OPEN}</span> '
872
+ f'<span class="nb-json-punct">{open_b}</span>'
873
+ f"{''.join(rows)}"
874
+ f'<span class="nb-json-punct">{close_b}</span>'
875
+ )
876
+
877
+
878
+ def _json_node(value: Any, depth: int) -> str:
879
+ """Render one JSON node; past the expansion depth, containers collapse to a count."""
880
+ if isinstance(value, dict):
881
+ if depth >= _JSON_EXPAND_LEVELS:
882
+ return (
883
+ f'<span class="nb-json-caret">{_CARET_CLOSED}</span> '
884
+ f'<span class="nb-json-punct">{{{len(value)}}}</span>'
885
+ )
886
+ return _json_container("{", "}", list(value.items()), depth, keyed=True)
887
+ if isinstance(value, list):
888
+ if depth >= _JSON_EXPAND_LEVELS:
889
+ return (
890
+ f'<span class="nb-json-caret">{_CARET_CLOSED}</span> '
891
+ f'<span class="nb-json-punct">[{len(value)}]</span>'
892
+ )
893
+ return _json_container("[", "]", list(enumerate(value)), depth, keyed=False)
894
+ return _json_leaf(value)
895
+
896
+
897
+ def render_json(out: Output, ctx: RenderContext) -> str:
898
+ """D22 · JSON viewer — two-level expansion, colored keys/values, deeper levels collapsed."""
899
+ value = _coerce_json(out.data.get("application/json"))
900
+ body = _json_node(value, 0)
901
+ return (
902
+ '<div class="nb-json">'
903
+ '<div class="nb-json-head"><span class="nb-rich-mime">application/json</span></div>'
904
+ f'<div class="nb-json-body">{body}</div>'
905
+ "</div>"
906
+ )