mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,874 @@
|
|
|
1
|
+
"""Bounded, data-only observations derived from deterministic graph tables."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
from collections.abc import Mapping
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from datetime import UTC, date, datetime, timedelta
|
|
11
|
+
from decimal import Decimal
|
|
12
|
+
from typing import Any
|
|
13
|
+
from urllib.parse import parse_qsl, unquote, urlsplit
|
|
14
|
+
|
|
15
|
+
from mostlyright.data_harness.canonical import canonical_json_bytes
|
|
16
|
+
from mostlyright.data_harness.plan_graph import GraphTable, NodeAudit
|
|
17
|
+
|
|
18
|
+
OBSERVATION_SCHEMA_VERSION = "stage-observation.v2"
|
|
19
|
+
LEGACY_OBSERVATION_SCHEMA_VERSION = "stage-observation.v1"
|
|
20
|
+
MAX_SAMPLE_ROWS = 20
|
|
21
|
+
MAX_PROFILE_ROWS = 1_000
|
|
22
|
+
MAX_COLUMNS = 512
|
|
23
|
+
MAX_INPUTS = 32
|
|
24
|
+
MAX_LINEAGE_SOURCES = 64
|
|
25
|
+
MAX_FINDINGS = 128
|
|
26
|
+
MAX_PARAMETERS_BYTES = 64 * 1024
|
|
27
|
+
MAX_OBSERVATION_BYTES = 512 * 1024
|
|
28
|
+
MAX_TEXT_CHARS = 4_096
|
|
29
|
+
MAX_NESTING = 8
|
|
30
|
+
DISPLAY_DISPOSITIONS = frozenset({"full", "redacted", "aggregate_only", "schema_only"})
|
|
31
|
+
_SECRET_ASSIGNMENT = re.compile(
|
|
32
|
+
r"(?i)(?:api[_-]?key|account[_-]?key|private[_-]?key|access[_-]?token|authorization|"
|
|
33
|
+
r"bearer|client[_-]?secret|password|passwd|secret|signature|credential|"
|
|
34
|
+
r"session[_-]?token|sas[_-]?token|sharedaccesssignature)\s*[:=]"
|
|
35
|
+
)
|
|
36
|
+
_BEARER = re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._~+/=-]{8,}")
|
|
37
|
+
_SECRET_QUERY_NAME = re.compile(
|
|
38
|
+
r"(?i)^(?:"
|
|
39
|
+
r"x-amz-(?:credential|signature|security-token)|"
|
|
40
|
+
r"x-goog-(?:credential|signature)|googleaccessid|awsaccesskeyid|"
|
|
41
|
+
r"sig|signature|token|access[_-]?token|api[_-]?key|authorization|auth|"
|
|
42
|
+
r"password|passwd|secret|credential|client[_-]?secret|session[_-]?token|sas[_-]?token|"
|
|
43
|
+
r"account[_-]?key|private[_-]?key|sharedaccesssignature"
|
|
44
|
+
r")$"
|
|
45
|
+
)
|
|
46
|
+
_SECRET_NAME_PART = re.compile(
|
|
47
|
+
r"(?i)(?:^|[_-])(?:api[_-]?key|account[_-]?key|private[_-]?key|access[_-]?token|"
|
|
48
|
+
r"authorization|password|passwd|secret|signature|credential|session[_-]?token|"
|
|
49
|
+
r"sas[_-]?token|sharedaccesssignature|sig|token)(?:$|[_-])"
|
|
50
|
+
)
|
|
51
|
+
_DIGEST = re.compile(r"^[0-9a-f]{64}$")
|
|
52
|
+
_LOGICAL_TYPES = frozenset(
|
|
53
|
+
{
|
|
54
|
+
"null",
|
|
55
|
+
"boolean",
|
|
56
|
+
"int64",
|
|
57
|
+
"float64",
|
|
58
|
+
"decimal",
|
|
59
|
+
"date",
|
|
60
|
+
"timestamp",
|
|
61
|
+
"timestamp_utc",
|
|
62
|
+
"string",
|
|
63
|
+
"mixed",
|
|
64
|
+
}
|
|
65
|
+
)
|
|
66
|
+
V2_FIELDS = frozenset(
|
|
67
|
+
{
|
|
68
|
+
"schema_version",
|
|
69
|
+
"node_id",
|
|
70
|
+
"operation",
|
|
71
|
+
"operation_version",
|
|
72
|
+
"inputs",
|
|
73
|
+
"source_coordinates",
|
|
74
|
+
"operation_parameters",
|
|
75
|
+
"audit",
|
|
76
|
+
"input_rows",
|
|
77
|
+
"output_rows",
|
|
78
|
+
"input_schemas",
|
|
79
|
+
"output_schema",
|
|
80
|
+
"columns",
|
|
81
|
+
"lineage",
|
|
82
|
+
"temporal",
|
|
83
|
+
"coverage",
|
|
84
|
+
"quality",
|
|
85
|
+
"profiles",
|
|
86
|
+
"findings",
|
|
87
|
+
"sample",
|
|
88
|
+
"rejected_rows",
|
|
89
|
+
"duplicate_rows",
|
|
90
|
+
"output_digest",
|
|
91
|
+
"display",
|
|
92
|
+
"truncated",
|
|
93
|
+
}
|
|
94
|
+
)
|
|
95
|
+
LEGACY_FIELDS = frozenset(
|
|
96
|
+
{
|
|
97
|
+
"schema_version",
|
|
98
|
+
"node_id",
|
|
99
|
+
"operation",
|
|
100
|
+
"operation_version",
|
|
101
|
+
"input_rows",
|
|
102
|
+
"output_rows",
|
|
103
|
+
"columns",
|
|
104
|
+
"sample",
|
|
105
|
+
"rejected_rows",
|
|
106
|
+
"duplicate_rows",
|
|
107
|
+
"output_digest",
|
|
108
|
+
"display",
|
|
109
|
+
"truncated",
|
|
110
|
+
}
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _decoded(value: str) -> str:
|
|
115
|
+
current = value
|
|
116
|
+
for _ in range(3):
|
|
117
|
+
decoded = unquote(current)
|
|
118
|
+
if decoded == current:
|
|
119
|
+
break
|
|
120
|
+
current = decoded
|
|
121
|
+
return current
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _sensitive_name(value: str) -> bool:
|
|
125
|
+
decoded = _decoded(value).strip()
|
|
126
|
+
return (
|
|
127
|
+
_SECRET_QUERY_NAME.fullmatch(decoded) is not None
|
|
128
|
+
or _SECRET_NAME_PART.search(decoded) is not None
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def safe_column_for_display(value: str) -> bool:
|
|
133
|
+
"""Return whether a column coordinate is safe to reveal at a row-query boundary."""
|
|
134
|
+
|
|
135
|
+
return isinstance(value, str) and not _sensitive_name(value)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def safe_for_display(value: Any) -> bool:
|
|
139
|
+
"""Fail closed for plain or percent-encoded credential-bearing strings."""
|
|
140
|
+
|
|
141
|
+
if not isinstance(value, str):
|
|
142
|
+
return True
|
|
143
|
+
decoded = _decoded(value)
|
|
144
|
+
if _SECRET_ASSIGNMENT.search(decoded) is not None or _BEARER.search(decoded) is not None:
|
|
145
|
+
return False
|
|
146
|
+
try:
|
|
147
|
+
parsed = urlsplit(decoded)
|
|
148
|
+
except ValueError:
|
|
149
|
+
return False
|
|
150
|
+
if parsed.scheme and (parsed.username is not None or parsed.password is not None):
|
|
151
|
+
return False
|
|
152
|
+
query = _decoded(parsed.query)
|
|
153
|
+
for name, _ in parse_qsl(query, keep_blank_values=True):
|
|
154
|
+
if _sensitive_name(name):
|
|
155
|
+
return False
|
|
156
|
+
# Encoded separators can leave a credential assignment outside parse_qsl's view.
|
|
157
|
+
for item in re.split(r"[&;]", query):
|
|
158
|
+
name = _decoded(item.partition("=")[0]).strip()
|
|
159
|
+
if _sensitive_name(name):
|
|
160
|
+
return False
|
|
161
|
+
return True
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _bounded_text(value: Any, name: str) -> str:
|
|
165
|
+
if not isinstance(value, str) or not value or len(value) > MAX_TEXT_CHARS:
|
|
166
|
+
raise ValueError(f"observation {name} is not bounded text")
|
|
167
|
+
return value
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _integer(value: Any, name: str, *, maximum: int = 2**63 - 1) -> int:
|
|
171
|
+
if isinstance(value, bool) or not isinstance(value, int) or not 0 <= value <= maximum:
|
|
172
|
+
raise ValueError(f"observation {name} is not a bounded integer")
|
|
173
|
+
return value
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _json_value(value: Any, *, depth: int = 0) -> Any:
|
|
177
|
+
if depth > MAX_NESTING:
|
|
178
|
+
raise ValueError("observation parameters exceed the nesting bound")
|
|
179
|
+
if value is None or type(value) in {bool, int, str}:
|
|
180
|
+
if isinstance(value, str) and len(value) > MAX_TEXT_CHARS:
|
|
181
|
+
raise ValueError("observation parameter text exceeds its bound")
|
|
182
|
+
return value
|
|
183
|
+
if isinstance(value, float):
|
|
184
|
+
raise ValueError("observation parameters must not contain binary floats")
|
|
185
|
+
if isinstance(value, Decimal):
|
|
186
|
+
return format(value, "f")
|
|
187
|
+
if isinstance(value, (date, datetime)):
|
|
188
|
+
return value.isoformat().replace("+00:00", "Z")
|
|
189
|
+
if isinstance(value, (list, tuple)):
|
|
190
|
+
return [_json_value(item, depth=depth + 1) for item in value]
|
|
191
|
+
if isinstance(value, dict):
|
|
192
|
+
if any(not isinstance(key, str) or not key or len(key) > MAX_TEXT_CHARS for key in value):
|
|
193
|
+
raise ValueError("observation parameter keys must be bounded text")
|
|
194
|
+
return {key: _json_value(value[key], depth=depth + 1) for key in sorted(value)}
|
|
195
|
+
raise ValueError("observation parameters contain an unsupported value")
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _artifact_value(value: Any, *, depth: int = 0) -> Any:
|
|
199
|
+
"""Normalize typed table values into canonical observation JSON."""
|
|
200
|
+
|
|
201
|
+
if depth > MAX_NESTING:
|
|
202
|
+
raise ValueError("observation value exceeds the nesting bound")
|
|
203
|
+
if value is None or type(value) in {bool, int, str}:
|
|
204
|
+
return value
|
|
205
|
+
if isinstance(value, float):
|
|
206
|
+
if not math.isfinite(value):
|
|
207
|
+
raise ValueError("observation values must be finite")
|
|
208
|
+
return {"$float": value.hex()}
|
|
209
|
+
if isinstance(value, Decimal):
|
|
210
|
+
if not value.is_finite():
|
|
211
|
+
raise ValueError("observation decimal values must be finite")
|
|
212
|
+
return {"$decimal": format(value, "f")}
|
|
213
|
+
if isinstance(value, datetime):
|
|
214
|
+
return {"$timestamp": value.isoformat().replace("+00:00", "Z")}
|
|
215
|
+
if isinstance(value, date):
|
|
216
|
+
return {"$date": value.isoformat()}
|
|
217
|
+
if isinstance(value, (list, tuple)):
|
|
218
|
+
return [_artifact_value(item, depth=depth + 1) for item in value]
|
|
219
|
+
if isinstance(value, dict):
|
|
220
|
+
return {
|
|
221
|
+
str(key): _artifact_value(item, depth=depth + 1)
|
|
222
|
+
for key, item in sorted(value.items(), key=lambda item: str(item[0]))
|
|
223
|
+
}
|
|
224
|
+
raise ValueError("observation contains an unsupported table value")
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _scrub(value: Any) -> tuple[Any, bool]:
|
|
228
|
+
"""Remove secret values before they can enter an observation artifact."""
|
|
229
|
+
|
|
230
|
+
if isinstance(value, str):
|
|
231
|
+
return (value, False) if safe_for_display(value) else ("[redacted]", True)
|
|
232
|
+
if isinstance(value, list):
|
|
233
|
+
items = [_scrub(item) for item in value]
|
|
234
|
+
return [item for item, _ in items], any(changed for _, changed in items)
|
|
235
|
+
if isinstance(value, dict):
|
|
236
|
+
result: dict[str, Any] = {}
|
|
237
|
+
changed = False
|
|
238
|
+
for key, item in value.items():
|
|
239
|
+
if _sensitive_name(key):
|
|
240
|
+
result[key] = "[redacted]"
|
|
241
|
+
changed = True
|
|
242
|
+
else:
|
|
243
|
+
result[key], item_changed = _scrub(item)
|
|
244
|
+
changed = changed or item_changed
|
|
245
|
+
return result, changed
|
|
246
|
+
return value, False
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def scrub_for_display(value: Any) -> tuple[Any, bool]:
|
|
250
|
+
"""Return a deterministic recursively redacted display value."""
|
|
251
|
+
|
|
252
|
+
return _scrub(value)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def validate_stage_observation(value: Any) -> dict[str, Any]:
|
|
256
|
+
"""Validate the closed artifact envelope and its strict global bounds."""
|
|
257
|
+
|
|
258
|
+
if not isinstance(value, dict):
|
|
259
|
+
raise ValueError("stage observation must be an object")
|
|
260
|
+
schema_version = value.get("schema_version")
|
|
261
|
+
expected = V2_FIELDS if schema_version == OBSERVATION_SCHEMA_VERSION else LEGACY_FIELDS
|
|
262
|
+
if schema_version not in {OBSERVATION_SCHEMA_VERSION, LEGACY_OBSERVATION_SCHEMA_VERSION}:
|
|
263
|
+
raise ValueError("stage observation schema version is unsupported")
|
|
264
|
+
if set(value) != expected:
|
|
265
|
+
raise ValueError("stage observation has unknown or missing fields")
|
|
266
|
+
encoded = canonical_json_bytes(value)
|
|
267
|
+
if len(encoded) > MAX_OBSERVATION_BYTES:
|
|
268
|
+
raise ValueError("stage observation exceeds its byte bound")
|
|
269
|
+
_bounded_text(value["node_id"], "node_id")
|
|
270
|
+
_bounded_text(value["operation"], "operation")
|
|
271
|
+
_bounded_text(value["operation_version"], "operation_version")
|
|
272
|
+
_integer(value["input_rows"], "input_rows")
|
|
273
|
+
_integer(value["output_rows"], "output_rows")
|
|
274
|
+
_integer(value["rejected_rows"], "rejected_rows")
|
|
275
|
+
_integer(value["duplicate_rows"], "duplicate_rows")
|
|
276
|
+
if _DIGEST.fullmatch(str(value["output_digest"])) is None:
|
|
277
|
+
raise ValueError("stage observation output digest is invalid")
|
|
278
|
+
if value["display"] not in DISPLAY_DISPOSITIONS or not isinstance(value["truncated"], bool):
|
|
279
|
+
raise ValueError("stage observation display fields are invalid")
|
|
280
|
+
columns = value["columns"]
|
|
281
|
+
sample = value["sample"]
|
|
282
|
+
if not isinstance(columns, list) or len(columns) > MAX_COLUMNS:
|
|
283
|
+
raise ValueError("stage observation columns exceed their bound")
|
|
284
|
+
if not isinstance(sample, list) or len(sample) > MAX_SAMPLE_ROWS:
|
|
285
|
+
raise ValueError("stage observation sample exceeds its bound")
|
|
286
|
+
for column in columns:
|
|
287
|
+
_bounded_text(column, "column")
|
|
288
|
+
if value["display"] in {"aggregate_only", "schema_only"} and sample:
|
|
289
|
+
raise ValueError("stage observation display policy forbids sampled values")
|
|
290
|
+
for row in sample:
|
|
291
|
+
if not isinstance(row, dict) or list(row) != columns:
|
|
292
|
+
raise ValueError("stage observation sample differs from its ordered columns")
|
|
293
|
+
_artifact_value(row)
|
|
294
|
+
if schema_version == LEGACY_OBSERVATION_SCHEMA_VERSION:
|
|
295
|
+
return value
|
|
296
|
+
if not isinstance(value["operation_parameters"], dict):
|
|
297
|
+
raise ValueError("stage observation parameters must be an object")
|
|
298
|
+
_json_value(value["operation_parameters"])
|
|
299
|
+
if len(canonical_json_bytes(value["operation_parameters"])) > MAX_PARAMETERS_BYTES:
|
|
300
|
+
raise ValueError("stage observation parameters exceed their byte bound")
|
|
301
|
+
if not isinstance(value["inputs"], list) or len(value["inputs"]) > MAX_INPUTS:
|
|
302
|
+
raise ValueError("stage observation inputs exceed their bound")
|
|
303
|
+
for item in value["inputs"]:
|
|
304
|
+
if not isinstance(item, dict) or set(item) != {"coordinate", "digest", "row_count"}:
|
|
305
|
+
raise ValueError("stage observation input is invalid")
|
|
306
|
+
_bounded_text(item["coordinate"], "input coordinate")
|
|
307
|
+
if _DIGEST.fullmatch(str(item["digest"])) is None:
|
|
308
|
+
raise ValueError("stage observation input digest is invalid")
|
|
309
|
+
if item["row_count"] is not None:
|
|
310
|
+
_integer(item["row_count"], "input row_count")
|
|
311
|
+
if (
|
|
312
|
+
not isinstance(value["source_coordinates"], list)
|
|
313
|
+
or len(value["source_coordinates"]) > MAX_INPUTS
|
|
314
|
+
):
|
|
315
|
+
raise ValueError("stage observation source coordinates exceed their bound")
|
|
316
|
+
for coordinate in value["source_coordinates"]:
|
|
317
|
+
_bounded_text(coordinate, "source coordinate")
|
|
318
|
+
audit = value["audit"]
|
|
319
|
+
audit_fields = {
|
|
320
|
+
"parameter_digest",
|
|
321
|
+
"input_digests",
|
|
322
|
+
"output_digest",
|
|
323
|
+
"order_contract",
|
|
324
|
+
"engine",
|
|
325
|
+
"input_bytes",
|
|
326
|
+
"output_bytes",
|
|
327
|
+
"retained_bytes",
|
|
328
|
+
}
|
|
329
|
+
if not isinstance(audit, dict) or set(audit) != audit_fields:
|
|
330
|
+
raise ValueError("stage observation audit is invalid")
|
|
331
|
+
if not isinstance(audit["input_digests"], list) or len(audit["input_digests"]) > MAX_INPUTS:
|
|
332
|
+
raise ValueError("stage observation audit input digests are invalid")
|
|
333
|
+
for digest in [audit["parameter_digest"], audit["output_digest"], *audit["input_digests"]]:
|
|
334
|
+
if _DIGEST.fullmatch(str(digest)) is None:
|
|
335
|
+
raise ValueError("stage observation audit digest is invalid")
|
|
336
|
+
_bounded_text(audit["order_contract"], "order contract")
|
|
337
|
+
_bounded_text(audit["engine"], "engine")
|
|
338
|
+
for name in ("input_bytes", "output_bytes", "retained_bytes"):
|
|
339
|
+
_integer(audit[name], name)
|
|
340
|
+
for schema_group in (value["input_schemas"],):
|
|
341
|
+
if not isinstance(schema_group, list) or len(schema_group) > MAX_INPUTS:
|
|
342
|
+
raise ValueError("stage observation input schemas exceed their bound")
|
|
343
|
+
for item in schema_group:
|
|
344
|
+
if not isinstance(item, dict) or set(item) != {"coordinate", "columns"}:
|
|
345
|
+
raise ValueError("stage observation input schema is invalid")
|
|
346
|
+
_bounded_text(item["coordinate"], "schema coordinate")
|
|
347
|
+
_validate_columns(item["columns"])
|
|
348
|
+
_validate_columns(value["output_schema"])
|
|
349
|
+
lineage = value["lineage"]
|
|
350
|
+
if not isinstance(lineage, list) or len(lineage) > MAX_COLUMNS:
|
|
351
|
+
raise ValueError("stage observation lineage exceeds its bound")
|
|
352
|
+
for item in lineage:
|
|
353
|
+
if not isinstance(item, dict) or set(item) != {
|
|
354
|
+
"column",
|
|
355
|
+
"sources",
|
|
356
|
+
"operations",
|
|
357
|
+
"truncated",
|
|
358
|
+
}:
|
|
359
|
+
raise ValueError("stage observation lineage item is invalid")
|
|
360
|
+
_bounded_text(item["column"], "lineage column")
|
|
361
|
+
if (
|
|
362
|
+
not isinstance(item["truncated"], bool)
|
|
363
|
+
or not isinstance(item["sources"], list)
|
|
364
|
+
or len(item["sources"]) > MAX_LINEAGE_SOURCES
|
|
365
|
+
):
|
|
366
|
+
raise ValueError("stage observation lineage sources are invalid")
|
|
367
|
+
for source in item["sources"]:
|
|
368
|
+
if not isinstance(source, dict) or set(source) != {"coordinate", "column"}:
|
|
369
|
+
raise ValueError("stage observation lineage source is invalid")
|
|
370
|
+
_bounded_text(source["coordinate"], "lineage coordinate")
|
|
371
|
+
_bounded_text(source["column"], "lineage source column")
|
|
372
|
+
if (
|
|
373
|
+
not isinstance(item["operations"], list)
|
|
374
|
+
or len(item["operations"]) > MAX_LINEAGE_SOURCES
|
|
375
|
+
):
|
|
376
|
+
raise ValueError("stage observation lineage operations are invalid")
|
|
377
|
+
for operation in item["operations"]:
|
|
378
|
+
_bounded_text(operation, "lineage operation")
|
|
379
|
+
_validate_facts(value)
|
|
380
|
+
return value
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _validate_columns(value: Any) -> None:
|
|
384
|
+
if not isinstance(value, list) or len(value) > MAX_COLUMNS:
|
|
385
|
+
raise ValueError("stage observation schema columns exceed their bound")
|
|
386
|
+
for column in value:
|
|
387
|
+
if not isinstance(column, dict) or set(column) != {"name", "logical_type", "nullable"}:
|
|
388
|
+
raise ValueError("stage observation schema column is invalid")
|
|
389
|
+
_bounded_text(column["name"], "schema column")
|
|
390
|
+
if column["logical_type"] not in _LOGICAL_TYPES or (
|
|
391
|
+
column["nullable"] is not None and not isinstance(column["nullable"], bool)
|
|
392
|
+
):
|
|
393
|
+
raise ValueError("stage observation schema type is invalid")
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _validate_facts(value: dict[str, Any]) -> None:
|
|
397
|
+
temporal = value["temporal"]
|
|
398
|
+
temporal_fields = {"time_column", "timezone", "window_start", "window_end", "granularity"}
|
|
399
|
+
if temporal is not None:
|
|
400
|
+
if not isinstance(temporal, dict) or set(temporal) != temporal_fields:
|
|
401
|
+
raise ValueError("stage observation temporal facts are invalid")
|
|
402
|
+
for item in temporal.values():
|
|
403
|
+
if item is not None:
|
|
404
|
+
_bounded_text(item, "temporal fact")
|
|
405
|
+
coverage = value["coverage"]
|
|
406
|
+
coverage_fields = {"expected_rows", "observed_rows", "complete"}
|
|
407
|
+
if not isinstance(coverage, dict) or frozenset(coverage) not in {
|
|
408
|
+
frozenset(coverage_fields),
|
|
409
|
+
frozenset((*coverage_fields, "representative_window")),
|
|
410
|
+
}:
|
|
411
|
+
raise ValueError("stage observation coverage facts are invalid")
|
|
412
|
+
for name in ("expected_rows", "observed_rows"):
|
|
413
|
+
if coverage[name] is not None:
|
|
414
|
+
_integer(coverage[name], f"coverage {name}")
|
|
415
|
+
if coverage["complete"] is not None and not isinstance(coverage["complete"], bool):
|
|
416
|
+
raise ValueError("stage observation coverage completeness is invalid")
|
|
417
|
+
representative = coverage.get("representative_window")
|
|
418
|
+
if representative is not None:
|
|
419
|
+
if not isinstance(representative, dict) or set(representative) != {
|
|
420
|
+
"start",
|
|
421
|
+
"end",
|
|
422
|
+
"output",
|
|
423
|
+
"inputs",
|
|
424
|
+
"truncated",
|
|
425
|
+
}:
|
|
426
|
+
raise ValueError("stage observation representative window is invalid")
|
|
427
|
+
_bounded_text(representative["start"], "representative window start")
|
|
428
|
+
_bounded_text(representative["end"], "representative window end")
|
|
429
|
+
if not isinstance(representative["output"], dict):
|
|
430
|
+
raise ValueError("stage observation representative output is invalid")
|
|
431
|
+
if not isinstance(representative["truncated"], bool):
|
|
432
|
+
raise ValueError("stage observation representative truncation is invalid")
|
|
433
|
+
inputs = representative["inputs"]
|
|
434
|
+
if not isinstance(inputs, list) or len(inputs) > MAX_INPUTS:
|
|
435
|
+
raise ValueError("stage observation representative inputs are invalid")
|
|
436
|
+
for item in inputs:
|
|
437
|
+
if not isinstance(item, dict) or set(item) != {"coordinate", "rows"}:
|
|
438
|
+
raise ValueError("stage observation representative input is invalid")
|
|
439
|
+
_bounded_text(item["coordinate"], "representative input coordinate")
|
|
440
|
+
if not isinstance(item["rows"], list) or len(item["rows"]) > MAX_SAMPLE_ROWS:
|
|
441
|
+
raise ValueError("stage observation representative input rows are invalid")
|
|
442
|
+
if any(not isinstance(row, dict) for row in item["rows"]):
|
|
443
|
+
raise ValueError("stage observation representative input row is invalid")
|
|
444
|
+
if value["display"] != "full" and (
|
|
445
|
+
representative["output"] or any(item["rows"] for item in representative["inputs"])
|
|
446
|
+
):
|
|
447
|
+
raise ValueError("stage observation display policy forbids representative row values")
|
|
448
|
+
quality = value["quality"]
|
|
449
|
+
quality_fields = {"checks_passed", "checks_total", "rejected_rows", "duplicate_rows"}
|
|
450
|
+
if not isinstance(quality, dict) or set(quality) != quality_fields:
|
|
451
|
+
raise ValueError("stage observation quality facts are invalid")
|
|
452
|
+
for name, item in quality.items():
|
|
453
|
+
if item is not None:
|
|
454
|
+
_integer(item, f"quality {name}")
|
|
455
|
+
profiles = value["profiles"]
|
|
456
|
+
if not isinstance(profiles, list) or len(profiles) > MAX_COLUMNS:
|
|
457
|
+
raise ValueError("stage observation profiles exceed their bound")
|
|
458
|
+
for profile in profiles:
|
|
459
|
+
if not isinstance(profile, dict) or set(profile) != {
|
|
460
|
+
"column",
|
|
461
|
+
"rows_examined",
|
|
462
|
+
"null_count",
|
|
463
|
+
"distinct_count",
|
|
464
|
+
}:
|
|
465
|
+
raise ValueError("stage observation profile is invalid")
|
|
466
|
+
_bounded_text(profile["column"], "profile column")
|
|
467
|
+
for name in ("rows_examined", "null_count", "distinct_count"):
|
|
468
|
+
_integer(profile[name], f"profile {name}", maximum=MAX_PROFILE_ROWS)
|
|
469
|
+
findings = value["findings"]
|
|
470
|
+
if not isinstance(findings, list) or len(findings) > MAX_FINDINGS:
|
|
471
|
+
raise ValueError("stage observation findings exceed their bound")
|
|
472
|
+
for finding in findings:
|
|
473
|
+
if not isinstance(finding, dict) or set(finding) != {
|
|
474
|
+
"severity",
|
|
475
|
+
"code",
|
|
476
|
+
"message",
|
|
477
|
+
"column",
|
|
478
|
+
}:
|
|
479
|
+
raise ValueError("stage observation finding is invalid")
|
|
480
|
+
if finding["severity"] not in {"info", "warning", "error"}:
|
|
481
|
+
raise ValueError("stage observation finding severity is invalid")
|
|
482
|
+
_bounded_text(finding["code"], "finding code")
|
|
483
|
+
_bounded_text(finding["message"], "finding message")
|
|
484
|
+
if finding["column"] is not None:
|
|
485
|
+
_bounded_text(finding["column"], "finding column")
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _schema(table: GraphTable) -> tuple[dict[str, Any], ...]:
|
|
489
|
+
if len(table.columns) > MAX_COLUMNS:
|
|
490
|
+
raise ValueError("observation schema exceeds its column bound")
|
|
491
|
+
nullable = [False] * len(table.columns)
|
|
492
|
+
for row in table.rows:
|
|
493
|
+
for index, value in enumerate(row):
|
|
494
|
+
nullable[index] = nullable[index] or value is None
|
|
495
|
+
result = []
|
|
496
|
+
for index, column in enumerate(table.columns):
|
|
497
|
+
_bounded_text(column, "column")
|
|
498
|
+
result.append(
|
|
499
|
+
{
|
|
500
|
+
"name": column,
|
|
501
|
+
"logical_type": table.schema[column],
|
|
502
|
+
"nullable": nullable[index],
|
|
503
|
+
}
|
|
504
|
+
)
|
|
505
|
+
return tuple(result)
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _profiles(table: GraphTable) -> tuple[dict[str, Any], ...]:
|
|
509
|
+
retained = table.rows[:MAX_PROFILE_ROWS]
|
|
510
|
+
result = []
|
|
511
|
+
for index, column in enumerate(table.columns):
|
|
512
|
+
values = tuple(row[index] for row in retained)
|
|
513
|
+
distinct = {
|
|
514
|
+
json.dumps(
|
|
515
|
+
_artifact_value(value), sort_keys=True, ensure_ascii=False, separators=(",", ":")
|
|
516
|
+
)
|
|
517
|
+
for value in values
|
|
518
|
+
if value is not None
|
|
519
|
+
}
|
|
520
|
+
result.append(
|
|
521
|
+
{
|
|
522
|
+
"column": column,
|
|
523
|
+
"rows_examined": len(values),
|
|
524
|
+
"null_count": sum(value is None for value in values),
|
|
525
|
+
"distinct_count": len(distinct),
|
|
526
|
+
}
|
|
527
|
+
)
|
|
528
|
+
return tuple(result)
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
@dataclass(frozen=True)
|
|
532
|
+
class StageObservation:
|
|
533
|
+
node_id: str
|
|
534
|
+
operation: str
|
|
535
|
+
operation_version: str
|
|
536
|
+
inputs: tuple[dict[str, Any], ...]
|
|
537
|
+
source_coordinates: tuple[str, ...]
|
|
538
|
+
operation_parameters: dict[str, Any]
|
|
539
|
+
audit: dict[str, Any]
|
|
540
|
+
input_rows: int
|
|
541
|
+
output_rows: int
|
|
542
|
+
input_schemas: tuple[dict[str, Any], ...]
|
|
543
|
+
output_schema: tuple[dict[str, Any], ...]
|
|
544
|
+
columns: tuple[str, ...]
|
|
545
|
+
lineage: tuple[dict[str, Any], ...]
|
|
546
|
+
temporal: dict[str, Any] | None
|
|
547
|
+
coverage: dict[str, Any]
|
|
548
|
+
quality: dict[str, Any]
|
|
549
|
+
profiles: tuple[dict[str, Any], ...]
|
|
550
|
+
findings: tuple[dict[str, Any], ...]
|
|
551
|
+
sample: tuple[dict[str, Any], ...]
|
|
552
|
+
rejected_rows: int
|
|
553
|
+
duplicate_rows: int
|
|
554
|
+
output_digest: str
|
|
555
|
+
display: str
|
|
556
|
+
truncated: bool
|
|
557
|
+
|
|
558
|
+
def __post_init__(self) -> None:
|
|
559
|
+
_bounded_text(self.node_id, "node_id")
|
|
560
|
+
_bounded_text(self.operation, "operation")
|
|
561
|
+
_bounded_text(self.operation_version, "operation_version")
|
|
562
|
+
if self.display not in DISPLAY_DISPOSITIONS:
|
|
563
|
+
raise ValueError("observation display disposition is invalid")
|
|
564
|
+
if len(self.inputs) > MAX_INPUTS or len(self.input_schemas) > MAX_INPUTS:
|
|
565
|
+
raise ValueError("observation exceeds its input bound")
|
|
566
|
+
if len(self.sample) > MAX_SAMPLE_ROWS:
|
|
567
|
+
raise ValueError("observation sample exceeds its row bound")
|
|
568
|
+
if len(self.profiles) > MAX_COLUMNS or len(self.findings) > MAX_FINDINGS:
|
|
569
|
+
raise ValueError("observation evidence exceeds its bound")
|
|
570
|
+
if self.display in {"aggregate_only", "schema_only"} and self.sample:
|
|
571
|
+
raise ValueError("display policy forbids sampled values")
|
|
572
|
+
if _DIGEST.fullmatch(self.output_digest) is None:
|
|
573
|
+
raise ValueError("observation output digest is invalid")
|
|
574
|
+
if len(canonical_json_bytes(self.operation_parameters)) > MAX_PARAMETERS_BYTES:
|
|
575
|
+
raise ValueError("observation parameters exceed their byte bound")
|
|
576
|
+
|
|
577
|
+
def to_dict(self) -> dict[str, Any]:
|
|
578
|
+
value = {
|
|
579
|
+
"schema_version": OBSERVATION_SCHEMA_VERSION,
|
|
580
|
+
"node_id": self.node_id,
|
|
581
|
+
"operation": self.operation,
|
|
582
|
+
"operation_version": self.operation_version,
|
|
583
|
+
"inputs": list(self.inputs),
|
|
584
|
+
"source_coordinates": list(self.source_coordinates),
|
|
585
|
+
"operation_parameters": self.operation_parameters,
|
|
586
|
+
"audit": self.audit,
|
|
587
|
+
"input_rows": self.input_rows,
|
|
588
|
+
"output_rows": self.output_rows,
|
|
589
|
+
"input_schemas": list(self.input_schemas),
|
|
590
|
+
"output_schema": list(self.output_schema),
|
|
591
|
+
"columns": list(self.columns),
|
|
592
|
+
"lineage": list(self.lineage),
|
|
593
|
+
"temporal": self.temporal,
|
|
594
|
+
"coverage": self.coverage,
|
|
595
|
+
"quality": self.quality,
|
|
596
|
+
"profiles": list(self.profiles),
|
|
597
|
+
"findings": list(self.findings),
|
|
598
|
+
"sample": list(self.sample),
|
|
599
|
+
"rejected_rows": self.rejected_rows,
|
|
600
|
+
"duplicate_rows": self.duplicate_rows,
|
|
601
|
+
"output_digest": self.output_digest,
|
|
602
|
+
"display": self.display,
|
|
603
|
+
"truncated": self.truncated,
|
|
604
|
+
}
|
|
605
|
+
if len(canonical_json_bytes(value)) > MAX_OBSERVATION_BYTES:
|
|
606
|
+
raise ValueError("stage observation exceeds its byte bound")
|
|
607
|
+
return value
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def observe_table(
|
|
611
|
+
audit: NodeAudit,
|
|
612
|
+
table: GraphTable,
|
|
613
|
+
*,
|
|
614
|
+
display: str,
|
|
615
|
+
sample_rows: int = 5,
|
|
616
|
+
input_tables: dict[str, GraphTable] | None = None,
|
|
617
|
+
source_coordinates: tuple[str, ...] = (),
|
|
618
|
+
operation_parameters: dict[str, Any] | None = None,
|
|
619
|
+
temporal: dict[str, Any] | None = None,
|
|
620
|
+
coverage: dict[str, Any] | None = None,
|
|
621
|
+
quality: dict[str, Any] | None = None,
|
|
622
|
+
findings: tuple[dict[str, Any], ...] = (),
|
|
623
|
+
) -> StageObservation:
|
|
624
|
+
"""Derive the closed v2 observation; optional context is supplied by the graph owner."""
|
|
625
|
+
|
|
626
|
+
if (
|
|
627
|
+
isinstance(sample_rows, bool)
|
|
628
|
+
or not isinstance(sample_rows, int)
|
|
629
|
+
or not 0 <= sample_rows <= MAX_SAMPLE_ROWS
|
|
630
|
+
):
|
|
631
|
+
raise ValueError("sample row bound is invalid")
|
|
632
|
+
if display not in DISPLAY_DISPOSITIONS:
|
|
633
|
+
raise ValueError("observation display disposition is invalid")
|
|
634
|
+
if len(audit.inputs) != len(audit.input_digests):
|
|
635
|
+
raise ValueError("observation audit input coordinates and digests differ")
|
|
636
|
+
input_tables = input_tables or {}
|
|
637
|
+
if set(input_tables) - set(audit.inputs):
|
|
638
|
+
raise ValueError("observation input tables name an unknown coordinate")
|
|
639
|
+
parameters = _json_value(operation_parameters or {})
|
|
640
|
+
assert isinstance(parameters, dict)
|
|
641
|
+
parameters, parameters_redacted = _scrub(parameters)
|
|
642
|
+
assert isinstance(parameters, dict)
|
|
643
|
+
source_values = tuple(_bounded_text(item, "source coordinate") for item in source_coordinates)
|
|
644
|
+
scrubbed_sources, sources_redacted = _scrub(list(source_values))
|
|
645
|
+
source_values = tuple(scrubbed_sources)
|
|
646
|
+
retained = table.rows[:sample_rows] if display in {"full", "redacted"} else ()
|
|
647
|
+
unsafe_sample = any(_sensitive_name(column) for column in table.columns) or any(
|
|
648
|
+
not safe_for_display(value) for row in retained for value in row
|
|
649
|
+
)
|
|
650
|
+
if display == "full" and (unsafe_sample or parameters_redacted or sources_redacted):
|
|
651
|
+
display = "schema_only"
|
|
652
|
+
retained = ()
|
|
653
|
+
sample = []
|
|
654
|
+
for row in retained:
|
|
655
|
+
values = {
|
|
656
|
+
column: _artifact_value(value) for column, value in zip(table.columns, row, strict=True)
|
|
657
|
+
}
|
|
658
|
+
if display == "redacted":
|
|
659
|
+
values = {
|
|
660
|
+
column: None if value is None else "[redacted]" for column, value in values.items()
|
|
661
|
+
}
|
|
662
|
+
sample.append(values)
|
|
663
|
+
inputs = tuple(
|
|
664
|
+
{
|
|
665
|
+
"coordinate": coordinate,
|
|
666
|
+
"digest": digest,
|
|
667
|
+
"row_count": len(input_tables[coordinate].rows) if coordinate in input_tables else None,
|
|
668
|
+
}
|
|
669
|
+
for coordinate, digest in zip(audit.inputs, audit.input_digests, strict=True)
|
|
670
|
+
)
|
|
671
|
+
input_schemas = []
|
|
672
|
+
for index, coordinate in enumerate(audit.inputs):
|
|
673
|
+
if coordinate in input_tables:
|
|
674
|
+
columns = list(_schema(input_tables[coordinate]))
|
|
675
|
+
elif index < len(audit.input_schemas):
|
|
676
|
+
columns = [
|
|
677
|
+
{"name": name, "logical_type": logical_type, "nullable": None}
|
|
678
|
+
for name, logical_type in audit.input_schemas[index]
|
|
679
|
+
]
|
|
680
|
+
else:
|
|
681
|
+
columns = []
|
|
682
|
+
input_schemas.append({"coordinate": coordinate, "columns": columns})
|
|
683
|
+
lineage = []
|
|
684
|
+
for column in table.columns:
|
|
685
|
+
column_lineage = table.lineage.get(column)
|
|
686
|
+
sources = () if column_lineage is None else tuple(column_lineage)
|
|
687
|
+
operations = () if column_lineage is None else column_lineage.operations
|
|
688
|
+
lineage.append(
|
|
689
|
+
{
|
|
690
|
+
"column": column,
|
|
691
|
+
"sources": [
|
|
692
|
+
{"coordinate": coordinate, "column": source_column}
|
|
693
|
+
for coordinate, source_column in sources[:MAX_LINEAGE_SOURCES]
|
|
694
|
+
],
|
|
695
|
+
"operations": list(operations[:MAX_LINEAGE_SOURCES]),
|
|
696
|
+
"truncated": max(len(sources), len(operations)) > MAX_LINEAGE_SOURCES,
|
|
697
|
+
}
|
|
698
|
+
)
|
|
699
|
+
normalized_temporal = None if temporal is None else _json_value(temporal)
|
|
700
|
+
normalized_temporal, temporal_redacted = _scrub(normalized_temporal)
|
|
701
|
+
normalized_coverage = _json_value(
|
|
702
|
+
coverage
|
|
703
|
+
or {
|
|
704
|
+
"expected_rows": None,
|
|
705
|
+
"observed_rows": audit.output_rows,
|
|
706
|
+
"complete": None,
|
|
707
|
+
"representative_window": None,
|
|
708
|
+
}
|
|
709
|
+
)
|
|
710
|
+
normalized_coverage, coverage_redacted = _scrub(normalized_coverage)
|
|
711
|
+
normalized_quality = _json_value(
|
|
712
|
+
quality
|
|
713
|
+
or {
|
|
714
|
+
"checks_passed": None,
|
|
715
|
+
"checks_total": None,
|
|
716
|
+
"rejected_rows": audit.rejected_rows,
|
|
717
|
+
"duplicate_rows": audit.duplicate_rows,
|
|
718
|
+
}
|
|
719
|
+
)
|
|
720
|
+
normalized_findings = _json_value(list(findings))
|
|
721
|
+
if not isinstance(normalized_findings, list) or len(normalized_findings) > MAX_FINDINGS:
|
|
722
|
+
raise ValueError("observation findings exceed their bound")
|
|
723
|
+
normalized_findings, findings_redacted = _scrub(normalized_findings)
|
|
724
|
+
if display == "full" and (findings_redacted or temporal_redacted or coverage_redacted):
|
|
725
|
+
display = "schema_only"
|
|
726
|
+
sample = []
|
|
727
|
+
if isinstance(normalized_coverage, dict):
|
|
728
|
+
normalized_coverage["representative_window"] = None
|
|
729
|
+
observation = StageObservation(
|
|
730
|
+
node_id=audit.node_id,
|
|
731
|
+
operation=audit.operation,
|
|
732
|
+
operation_version=audit.operation_version,
|
|
733
|
+
inputs=inputs,
|
|
734
|
+
source_coordinates=source_values,
|
|
735
|
+
operation_parameters=parameters,
|
|
736
|
+
audit={
|
|
737
|
+
"parameter_digest": audit.parameter_digest,
|
|
738
|
+
"input_digests": list(audit.input_digests),
|
|
739
|
+
"output_digest": audit.output_digest,
|
|
740
|
+
"order_contract": audit.order_contract,
|
|
741
|
+
"engine": audit.engine,
|
|
742
|
+
"input_bytes": audit.input_bytes,
|
|
743
|
+
"output_bytes": audit.output_bytes,
|
|
744
|
+
"retained_bytes": audit.retained_bytes,
|
|
745
|
+
},
|
|
746
|
+
input_rows=audit.input_rows,
|
|
747
|
+
output_rows=audit.output_rows,
|
|
748
|
+
input_schemas=tuple(input_schemas),
|
|
749
|
+
output_schema=_schema(table),
|
|
750
|
+
columns=table.columns,
|
|
751
|
+
lineage=tuple(lineage),
|
|
752
|
+
temporal=normalized_temporal,
|
|
753
|
+
coverage=normalized_coverage,
|
|
754
|
+
quality=normalized_quality,
|
|
755
|
+
profiles=_profiles(table),
|
|
756
|
+
findings=tuple(normalized_findings),
|
|
757
|
+
sample=tuple(sample),
|
|
758
|
+
rejected_rows=audit.rejected_rows,
|
|
759
|
+
duplicate_rows=audit.duplicate_rows,
|
|
760
|
+
output_digest=audit.output_digest,
|
|
761
|
+
display=display,
|
|
762
|
+
truncated=len(table.rows) > len(sample),
|
|
763
|
+
)
|
|
764
|
+
validate_stage_observation(observation.to_dict())
|
|
765
|
+
return observation
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
def resample_window_evidence(
|
|
769
|
+
table: GraphTable,
|
|
770
|
+
input_tables: Mapping[str, GraphTable],
|
|
771
|
+
parameters: Mapping[str, Any],
|
|
772
|
+
*,
|
|
773
|
+
include_rows: bool,
|
|
774
|
+
) -> dict[str, Any] | None:
|
|
775
|
+
"""Bind one deterministic resample output window to bounded contributing rows."""
|
|
776
|
+
|
|
777
|
+
if not table.rows:
|
|
778
|
+
return None
|
|
779
|
+
event_time = parameters.get("event_time")
|
|
780
|
+
duration = parameters.get("duration_seconds")
|
|
781
|
+
group_by = parameters.get("group_by")
|
|
782
|
+
if (
|
|
783
|
+
not isinstance(event_time, str)
|
|
784
|
+
or isinstance(duration, bool)
|
|
785
|
+
or not isinstance(duration, int)
|
|
786
|
+
or duration <= 0
|
|
787
|
+
or not isinstance(group_by, list)
|
|
788
|
+
or any(not isinstance(item, str) for item in group_by)
|
|
789
|
+
):
|
|
790
|
+
raise ValueError("resample parameters cannot produce representative evidence")
|
|
791
|
+
bucket_column = f"{event_time}_bucket"
|
|
792
|
+
if bucket_column not in table.columns:
|
|
793
|
+
raise ValueError("resample output omits its bucket column")
|
|
794
|
+
output_index = {column: index for index, column in enumerate(table.columns)}
|
|
795
|
+
output_row = table.rows[0]
|
|
796
|
+
start_value = output_row[output_index[bucket_column]]
|
|
797
|
+
if isinstance(start_value, str) and start_value.endswith("Z"):
|
|
798
|
+
start = datetime.fromisoformat(start_value.replace("Z", "+00:00"))
|
|
799
|
+
elif isinstance(start_value, datetime):
|
|
800
|
+
start = start_value
|
|
801
|
+
else:
|
|
802
|
+
raise ValueError("resample output bucket is not a UTC timestamp")
|
|
803
|
+
if start.tzinfo is None or start.utcoffset() is None:
|
|
804
|
+
raise ValueError("resample output bucket is not timezone-aware")
|
|
805
|
+
start = start.astimezone(UTC)
|
|
806
|
+
end = start + timedelta(seconds=duration)
|
|
807
|
+
group_values = {column: output_row[output_index[column]] for column in group_by}
|
|
808
|
+
representative_inputs: list[dict[str, Any]] = []
|
|
809
|
+
total_matches = 0
|
|
810
|
+
retained_matches = 0
|
|
811
|
+
if include_rows:
|
|
812
|
+
for coordinate, input_table in sorted(input_tables.items()):
|
|
813
|
+
input_index = {column: index for index, column in enumerate(input_table.columns)}
|
|
814
|
+
if event_time not in input_index or any(
|
|
815
|
+
column not in input_index for column in group_by
|
|
816
|
+
):
|
|
817
|
+
continue
|
|
818
|
+
retained: list[dict[str, Any]] = []
|
|
819
|
+
for row in input_table.rows:
|
|
820
|
+
if any(
|
|
821
|
+
row[input_index[column]] != expected
|
|
822
|
+
for column, expected in group_values.items()
|
|
823
|
+
):
|
|
824
|
+
continue
|
|
825
|
+
raw_time = row[input_index[event_time]]
|
|
826
|
+
if isinstance(raw_time, str) and raw_time.endswith("Z"):
|
|
827
|
+
observed = datetime.fromisoformat(raw_time.replace("Z", "+00:00"))
|
|
828
|
+
elif isinstance(raw_time, datetime):
|
|
829
|
+
observed = raw_time
|
|
830
|
+
else:
|
|
831
|
+
continue
|
|
832
|
+
if observed.tzinfo is None or observed.utcoffset() is None:
|
|
833
|
+
continue
|
|
834
|
+
if not start <= observed.astimezone(UTC) < end:
|
|
835
|
+
continue
|
|
836
|
+
total_matches += 1
|
|
837
|
+
if retained_matches < MAX_SAMPLE_ROWS:
|
|
838
|
+
retained.append(
|
|
839
|
+
{
|
|
840
|
+
column: _artifact_value(value)
|
|
841
|
+
for column, value in zip(input_table.columns, row, strict=True)
|
|
842
|
+
}
|
|
843
|
+
)
|
|
844
|
+
retained_matches += 1
|
|
845
|
+
if retained:
|
|
846
|
+
representative_inputs.append({"coordinate": coordinate, "rows": retained})
|
|
847
|
+
return {
|
|
848
|
+
"start": start.isoformat().replace("+00:00", "Z"),
|
|
849
|
+
"end": end.isoformat().replace("+00:00", "Z"),
|
|
850
|
+
"output": (
|
|
851
|
+
{
|
|
852
|
+
column: _artifact_value(value)
|
|
853
|
+
for column, value in zip(table.columns, output_row, strict=True)
|
|
854
|
+
}
|
|
855
|
+
if include_rows
|
|
856
|
+
else {}
|
|
857
|
+
),
|
|
858
|
+
"inputs": representative_inputs,
|
|
859
|
+
"truncated": total_matches > retained_matches,
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
__all__ = [
|
|
864
|
+
"DISPLAY_DISPOSITIONS",
|
|
865
|
+
"LEGACY_OBSERVATION_SCHEMA_VERSION",
|
|
866
|
+
"MAX_SAMPLE_ROWS",
|
|
867
|
+
"OBSERVATION_SCHEMA_VERSION",
|
|
868
|
+
"StageObservation",
|
|
869
|
+
"observe_table",
|
|
870
|
+
"resample_window_evidence",
|
|
871
|
+
"safe_for_display",
|
|
872
|
+
"scrub_for_display",
|
|
873
|
+
"validate_stage_observation",
|
|
874
|
+
]
|