mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,1217 @@
|
|
|
1
|
+
"""Bounded streaming authoring shards behind one atomic manifest.
|
|
2
|
+
|
|
3
|
+
A converged Data.gov v4 generation is immutable staging: raw response blobs, normalized page
|
|
4
|
+
manifests over bounded record segments, segmented evidence indexes, and one terminal externally
|
|
5
|
+
merged pass map. This module turns that generation into immutable authoring
|
|
6
|
+
shards without ever holding it in memory, and replaces the 64 MiB monolithic
|
|
7
|
+
``mr-data-catalog-fill-authoring.v1`` bundle for the provider-scale workflow. The legacy bounded
|
|
8
|
+
authoring path in :mod:`authoring` is untouched and still readable.
|
|
9
|
+
|
|
10
|
+
The shape of the work is deliberate.
|
|
11
|
+
|
|
12
|
+
*Two bounded passes, never one big map.* The first pass streams the terminal pass's normalized
|
|
13
|
+
pages one segment at a time into a disk-backed scratch index keyed by the exact provider identifier
|
|
14
|
+
bytes. The second pass walks that index in ascending identifier order and authors one record at a
|
|
15
|
+
time into one open output shard. Peak retained records is therefore one input segment plus one
|
|
16
|
+
output shard, not one generation, and the authoring order is a property of the provider identifiers
|
|
17
|
+
rather than of page arrival.
|
|
18
|
+
|
|
19
|
+
*Immutable members, manifest last.* Every shard is content-addressed, installed through a
|
|
20
|
+
same-directory link with fsync and exact readback, and never rewritten. A small journal records
|
|
21
|
+
the committed shards and the exact last committed provider record so an interrupted run resumes
|
|
22
|
+
without duplicating or reordering anything. ``manifest.json`` is written only when the whole
|
|
23
|
+
generation is authored, so a crash leaves either a resumable partial generation or a complete one
|
|
24
|
+
-- never a half-installed head.
|
|
25
|
+
|
|
26
|
+
*Exact coordinates only.* Authoring refuses to start unless the caller names the exact live fill
|
|
27
|
+
checkpoint digest, the fill reports a converged terminal pass whose pass map authenticates, and the
|
|
28
|
+
indexed record population equals that pass map's unique record count. A resume additionally
|
|
29
|
+
requires the exact retained journal digest.
|
|
30
|
+
|
|
31
|
+
*One stated coverage, bounded or not.* A converged sweep can be larger than a release can carry,
|
|
32
|
+
so a caller may ask for a bounded selection of it instead of all of it: an ``AuthoringSelection``
|
|
33
|
+
names one registered rule and how many records to take under it. The rule is a total order over
|
|
34
|
+
the *whole* corpus, evaluated after the index is built and before one record is authored, so the
|
|
35
|
+
records it selects are a property of the corpus rather than of how far this invocation got. That
|
|
36
|
+
is what makes a bounded generation ``complete``: it authored the whole of what it said it would.
|
|
37
|
+
|
|
38
|
+
A bound is emphatically not the record budget. ``limits.max_records`` is a resource guard, it is
|
|
39
|
+
live on every invocation bounded or not, and a run that reaches it still stops at
|
|
40
|
+
``AUTHOR_RECORD_LIMIT`` with ``status="incomplete"`` and publishes no manifest. An operator who
|
|
41
|
+
sets a budget below their own bound gets an incomplete generation, exactly as they would have
|
|
42
|
+
before this existed. Nothing here relaxes that; ``coverage`` is a claim about which records were
|
|
43
|
+
asked for, ``limits`` is a claim about what this process was allowed to spend, and neither stands
|
|
44
|
+
in for the other.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
from __future__ import annotations
|
|
48
|
+
|
|
49
|
+
import contextlib
|
|
50
|
+
import os
|
|
51
|
+
import re
|
|
52
|
+
import secrets
|
|
53
|
+
import sqlite3
|
|
54
|
+
import stat
|
|
55
|
+
import tempfile
|
|
56
|
+
import time
|
|
57
|
+
from collections.abc import Callable, Iterator
|
|
58
|
+
from dataclasses import dataclass
|
|
59
|
+
from datetime import UTC, datetime
|
|
60
|
+
from pathlib import Path
|
|
61
|
+
from typing import Any
|
|
62
|
+
|
|
63
|
+
from mostlyright.data_harness.canonical import (
|
|
64
|
+
CanonicalJSONError,
|
|
65
|
+
canonical_json_bytes,
|
|
66
|
+
canonical_sha256,
|
|
67
|
+
parse_canonical_json,
|
|
68
|
+
sha256_bytes,
|
|
69
|
+
)
|
|
70
|
+
from mostlyright.data_harness.sources.catalog.authoring_policy import (
|
|
71
|
+
AuthoringPolicy,
|
|
72
|
+
author_catalog_record,
|
|
73
|
+
resolve_authoring_policy,
|
|
74
|
+
)
|
|
75
|
+
from mostlyright.data_harness.sources.catalog.bounded_io import (
|
|
76
|
+
BoundedReadFailure,
|
|
77
|
+
read_bounded_at,
|
|
78
|
+
)
|
|
79
|
+
from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
|
|
80
|
+
from mostlyright.data_harness.sources.catalog.coverage import (
|
|
81
|
+
COVERAGE_EXHAUSTIVE_RULE,
|
|
82
|
+
SELECTION_RULES,
|
|
83
|
+
coverage_is_valid,
|
|
84
|
+
)
|
|
85
|
+
from mostlyright.data_harness.sources.catalog.fill_partitions import validate_pass_map
|
|
86
|
+
from mostlyright.data_harness.sources.catalog.fill_staging import FillStaging
|
|
87
|
+
from mostlyright.data_harness.sources.contracts import SourceContractError
|
|
88
|
+
|
|
89
|
+
AUTHORING_SHARD_SCHEMA = "harness-catalog-authoring-shard.v1"
|
|
90
|
+
AUTHORING_MANIFEST_SCHEMA = "harness-catalog-authoring-manifest.v1"
|
|
91
|
+
AUTHORING_JOURNAL_SCHEMA = "harness-catalog-authoring-journal.v1"
|
|
92
|
+
AUTHORING_RECEIPT_SCHEMA = "mr-data-catalog-author.v1"
|
|
93
|
+
|
|
94
|
+
MANIFEST_FILENAME = "manifest.json"
|
|
95
|
+
JOURNAL_FILENAME = "journal.json"
|
|
96
|
+
SHARDS_DIRNAME = "shards"
|
|
97
|
+
|
|
98
|
+
MAX_AUTHORING_SHARD_ENTRIES = 1_000
|
|
99
|
+
MAX_AUTHORING_SHARD_BYTES = 8 * 1024 * 1024
|
|
100
|
+
MAX_AUTHORING_MANIFEST_BYTES = 8 * 1024 * 1024
|
|
101
|
+
MAX_AUTHORING_JOURNAL_BYTES = 8 * 1024 * 1024
|
|
102
|
+
|
|
103
|
+
#: Exactly what one authored shard descriptor states, and nothing else. The journal and the
|
|
104
|
+
#: manifest both carry these descriptors, and the composite generation receipt carries them
|
|
105
|
+
#: onward as authoritative evidence, so every reader closes the key set rather than reading the
|
|
106
|
+
#: members it happens to know and copying whatever else the document brought with it.
|
|
107
|
+
AUTHORING_SHARD_DESCRIPTOR_MEMBERS = frozenset(
|
|
108
|
+
{
|
|
109
|
+
"sha256",
|
|
110
|
+
"bytes",
|
|
111
|
+
"item_count",
|
|
112
|
+
"entry_count",
|
|
113
|
+
"first_provider_record_id",
|
|
114
|
+
"last_provider_record_id",
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def shard_descriptor_is_valid(descriptor: Any) -> bool:
|
|
120
|
+
"""Whether one authored shard descriptor is exactly its contract, with admissible values.
|
|
121
|
+
|
|
122
|
+
Every reader that admits a descriptor answers this one question, so that adding a member to
|
|
123
|
+
the contract is one edit rather than three that have to agree. Each caller still raises its
|
|
124
|
+
own typed refusal, because the boundary the descriptor failed at is what an operator needs.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
return (
|
|
128
|
+
isinstance(descriptor, dict)
|
|
129
|
+
and set(descriptor) == AUTHORING_SHARD_DESCRIPTOR_MEMBERS
|
|
130
|
+
and _is_digest(descriptor["sha256"])
|
|
131
|
+
and type(descriptor["bytes"]) is int
|
|
132
|
+
and 1 <= descriptor["bytes"] <= MAX_AUTHORING_SHARD_BYTES
|
|
133
|
+
and type(descriptor["item_count"]) is int
|
|
134
|
+
and 1 <= descriptor["item_count"] <= MAX_AUTHORING_SHARD_ENTRIES
|
|
135
|
+
and type(descriptor["entry_count"]) is int
|
|
136
|
+
and 0 <= descriptor["entry_count"] <= descriptor["item_count"]
|
|
137
|
+
and isinstance(descriptor["first_provider_record_id"], str)
|
|
138
|
+
and isinstance(descriptor["last_provider_record_id"], str)
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
# Conservative allowance for the shard envelope around the items array: the schema label, the
|
|
143
|
+
# policy digest, the two bounded provider record identifiers, and the JSON punctuation. The real
|
|
144
|
+
# ceiling is still enforced exactly when the shard bytes are composed.
|
|
145
|
+
_SHARD_ENVELOPE_BYTES = 4_096
|
|
146
|
+
|
|
147
|
+
#: The provider observation ``last_harvested_date_desc`` orders by. It is a *volatile* field --
|
|
148
|
+
#: deliberately outside the semantic digest the pass map converges on, because a digest that moved
|
|
149
|
+
#: with a nightly re-harvest would make two adjacent equal pass maps unsatisfiable. That is
|
|
150
|
+
#: exactly right for a selection key: the corpus it orders is immutable staging, so the order is
|
|
151
|
+
#: reproducible from the generation even though the value is not part of any record's identity.
|
|
152
|
+
_RANK_OBSERVATION_PATH = "$.last_harvested_date"
|
|
153
|
+
|
|
154
|
+
#: The longest provider harvest date this rule will order by. A well-formed instant is 20
|
|
155
|
+
#: characters; 64 is room for every spelling of one. Anything longer is not a date this rule can
|
|
156
|
+
#: order by, so it ranks with the undated rather than sorting on its leading bytes.
|
|
157
|
+
MAX_SELECTION_RANK_CHARS = 64
|
|
158
|
+
|
|
159
|
+
#: The shape a provider value must have before this rule will treat it as a harvest date.
|
|
160
|
+
#:
|
|
161
|
+
#: Byte order is only chronological order over values that are actually dates. Comparison here is
|
|
162
|
+
#: bytewise, so *any* string sorts somewhere -- and because ``'N' < '2' < 'u'`` is false in the
|
|
163
|
+
#: middle, a record whose harvest date the provider filled in as ``unknown`` or ``pending`` would
|
|
164
|
+
#: sort above every real ISO-8601 instant and head the catalogue. "The N most recently harvested
|
|
165
|
+
#: datasets" would then be led by the N records with no harvest date at all, which is the exact
|
|
166
|
+
#: opposite of what the rule's name promises.
|
|
167
|
+
#:
|
|
168
|
+
#: So a value is a rank only if it opens with a calendar date, ``YYYY-MM-DD``. Every spelling of
|
|
169
|
+
#: an ISO-8601 instant does, and no English word does. The rule deliberately checks the opening
|
|
170
|
+
#: rather than the whole grammar: what the bytes after the date look like varies by provider and
|
|
171
|
+
#: does not change which day is later, and refusing a real date over its seconds field would drop
|
|
172
|
+
#: a record from the catalogue to satisfy a parser.
|
|
173
|
+
_RANK_DATE_PREFIX = re.compile(r"\d{4}-\d{2}-\d{2}")
|
|
174
|
+
|
|
175
|
+
#: What a record with no orderable harvest date ranks as. The empty string sorts below every
|
|
176
|
+
#: non-empty one, so undated records fall to the bottom of a descending order without a separate
|
|
177
|
+
#: null-handling clause that a query planner could get wrong.
|
|
178
|
+
_UNRANKED = ""
|
|
179
|
+
|
|
180
|
+
_NORMALIZED_PAGE_SCHEMA = "harness-datagov-v4-normalized-page.v2"
|
|
181
|
+
_NORMALIZED_SEGMENT_SCHEMA = "harness-datagov-v4-normalized-record-segment.v1"
|
|
182
|
+
_OPEN_DIRECTORY = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
|
|
183
|
+
_OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class CatalogAuthoringRefused(SourceContractError):
|
|
187
|
+
"""A stable refusal of an authoring coordinate, input, or output boundary."""
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
@dataclass(frozen=True)
|
|
191
|
+
class AuthoringLimits:
|
|
192
|
+
"""Explicit caller-supplied authoring bounds; none of them has a hidden default widening."""
|
|
193
|
+
|
|
194
|
+
max_records: int
|
|
195
|
+
max_input_bytes: int
|
|
196
|
+
max_shard_entries: int = MAX_AUTHORING_SHARD_ENTRIES
|
|
197
|
+
max_shard_bytes: int = MAX_AUTHORING_SHARD_BYTES
|
|
198
|
+
max_disk_bytes: int = 30 * 1024 * 1024 * 1024
|
|
199
|
+
max_wall_seconds: int = 12 * 60 * 60
|
|
200
|
+
|
|
201
|
+
def __post_init__(self) -> None:
|
|
202
|
+
for name in (
|
|
203
|
+
"max_records",
|
|
204
|
+
"max_input_bytes",
|
|
205
|
+
"max_shard_entries",
|
|
206
|
+
"max_shard_bytes",
|
|
207
|
+
"max_disk_bytes",
|
|
208
|
+
"max_wall_seconds",
|
|
209
|
+
):
|
|
210
|
+
value = getattr(self, name)
|
|
211
|
+
if type(value) is not int or value < 1:
|
|
212
|
+
raise CatalogAuthoringRefused(
|
|
213
|
+
"AUTHOR_LIMIT", f"limits.{name}", "must be a positive integer"
|
|
214
|
+
)
|
|
215
|
+
if (
|
|
216
|
+
self.max_shard_entries > MAX_AUTHORING_SHARD_ENTRIES
|
|
217
|
+
or self.max_shard_bytes > MAX_AUTHORING_SHARD_BYTES
|
|
218
|
+
):
|
|
219
|
+
raise CatalogAuthoringRefused(
|
|
220
|
+
"AUTHOR_LIMIT",
|
|
221
|
+
"limits.max_shard_entries",
|
|
222
|
+
"shard bounds may not exceed the fixed range contract",
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
def to_dict(self) -> dict[str, int]:
|
|
226
|
+
return {
|
|
227
|
+
"max_records": self.max_records,
|
|
228
|
+
"max_input_bytes": self.max_input_bytes,
|
|
229
|
+
"max_shard_entries": self.max_shard_entries,
|
|
230
|
+
"max_shard_bytes": self.max_shard_bytes,
|
|
231
|
+
"max_disk_bytes": self.max_disk_bytes,
|
|
232
|
+
"max_wall_seconds": self.max_wall_seconds,
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@dataclass(frozen=True)
|
|
237
|
+
class AuthoringSelection:
|
|
238
|
+
"""Which of a converged generation's records this authoring was asked to author.
|
|
239
|
+
|
|
240
|
+
The default is every one of them, which is what authoring did before a selection could be
|
|
241
|
+
stated and is what it still does when none is. Defaulting to exhaustive is the safe
|
|
242
|
+
direction: a caller who forgets the flag publishes a whole catalogue, never a quiet part of
|
|
243
|
+
one.
|
|
244
|
+
|
|
245
|
+
A bounded selection names a *registered* rule rather than describing one, so that "which
|
|
246
|
+
records are in this catalogue" has an answer a reader can look up instead of a sentence
|
|
247
|
+
somebody wrote into a receipt.
|
|
248
|
+
"""
|
|
249
|
+
|
|
250
|
+
rule: str = COVERAGE_EXHAUSTIVE_RULE
|
|
251
|
+
bound: int | None = None
|
|
252
|
+
|
|
253
|
+
def __post_init__(self) -> None:
|
|
254
|
+
if self.rule == COVERAGE_EXHAUSTIVE_RULE:
|
|
255
|
+
if self.bound is not None:
|
|
256
|
+
raise CatalogAuthoringRefused(
|
|
257
|
+
"AUTHOR_COVERAGE",
|
|
258
|
+
"selection.bound",
|
|
259
|
+
"an exhaustive authoring states no bound",
|
|
260
|
+
)
|
|
261
|
+
return
|
|
262
|
+
if self.rule not in SELECTION_RULES:
|
|
263
|
+
raise CatalogAuthoringRefused(
|
|
264
|
+
"AUTHOR_COVERAGE",
|
|
265
|
+
"selection.rule",
|
|
266
|
+
f"a bounded authoring names one registered rule {sorted(SELECTION_RULES)}",
|
|
267
|
+
)
|
|
268
|
+
if type(self.bound) is not int or self.bound < 1:
|
|
269
|
+
raise CatalogAuthoringRefused(
|
|
270
|
+
"AUTHOR_COVERAGE",
|
|
271
|
+
"selection.bound",
|
|
272
|
+
"a bounded authoring states a positive record bound",
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
def coverage(self, *, corpus_records: int, selected_records: int) -> dict[str, Any]:
|
|
276
|
+
"""What this selection, applied to a corpus this size, says it covers."""
|
|
277
|
+
|
|
278
|
+
coverage = {
|
|
279
|
+
"rule": self.rule,
|
|
280
|
+
"bound": self.bound,
|
|
281
|
+
"corpus_records": corpus_records,
|
|
282
|
+
"selected_records": selected_records,
|
|
283
|
+
}
|
|
284
|
+
if not coverage_is_valid(coverage):
|
|
285
|
+
raise CatalogAuthoringRefused(
|
|
286
|
+
"AUTHOR_COVERAGE",
|
|
287
|
+
"authoring.coverage",
|
|
288
|
+
"the selected population does not reproduce from the stated rule and bound",
|
|
289
|
+
)
|
|
290
|
+
return coverage
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
@dataclass(frozen=True)
|
|
294
|
+
class AuthoringResult:
|
|
295
|
+
"""One authoring invocation's exact outcome."""
|
|
296
|
+
|
|
297
|
+
status: str
|
|
298
|
+
reason_code: str | None
|
|
299
|
+
manifest_sha256: str | None
|
|
300
|
+
journal_sha256: str | None
|
|
301
|
+
coverage: dict[str, Any]
|
|
302
|
+
receipt: dict[str, Any]
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def run_catalog_authoring(
|
|
306
|
+
*,
|
|
307
|
+
staging_root: Path,
|
|
308
|
+
expected_fill_checkpoint_sha256: str,
|
|
309
|
+
policy_id: str,
|
|
310
|
+
output_root: Path,
|
|
311
|
+
limits: AuthoringLimits,
|
|
312
|
+
selection: AuthoringSelection | None = None,
|
|
313
|
+
expected_journal_sha256: str | None = None,
|
|
314
|
+
output_descriptor: int | None = None,
|
|
315
|
+
monotonic: Callable[[], float] = time.monotonic,
|
|
316
|
+
) -> AuthoringResult:
|
|
317
|
+
"""Author one converged v4 generation into bounded shards behind one atomic manifest."""
|
|
318
|
+
|
|
319
|
+
selection = AuthoringSelection() if selection is None else selection
|
|
320
|
+
policy = resolve_authoring_policy(policy_id)
|
|
321
|
+
if not _is_digest(expected_fill_checkpoint_sha256):
|
|
322
|
+
raise CatalogAuthoringRefused(
|
|
323
|
+
"AUTHOR_FILL_CHECKPOINT_MISMATCH",
|
|
324
|
+
"expected_fill_checkpoint_sha256",
|
|
325
|
+
"must be a lowercase SHA-256",
|
|
326
|
+
)
|
|
327
|
+
if expected_journal_sha256 is not None and not _is_digest(expected_journal_sha256):
|
|
328
|
+
raise CatalogAuthoringRefused(
|
|
329
|
+
"AUTHOR_JOURNAL_MISMATCH", "expected_journal_sha256", "must be a lowercase SHA-256"
|
|
330
|
+
)
|
|
331
|
+
deadline = monotonic() + limits.max_wall_seconds
|
|
332
|
+
output = _AuthoringOutput(output_root, root_descriptor=output_descriptor)
|
|
333
|
+
with output.opened():
|
|
334
|
+
journal = _resume_point(
|
|
335
|
+
output, expected_journal_sha256, policy, expected_fill_checkpoint_sha256, selection
|
|
336
|
+
)
|
|
337
|
+
_require_bounds_admit_retained(journal, limits)
|
|
338
|
+
staging = FillStaging(staging_root)
|
|
339
|
+
with staging.locked():
|
|
340
|
+
state = _converged_state(staging, expected_fill_checkpoint_sha256)
|
|
341
|
+
pass_map = validate_pass_map(
|
|
342
|
+
staging, state["last_pass_manifest_sha256"], state["final_map_sha256"]
|
|
343
|
+
)
|
|
344
|
+
scratch_root = _scratch_root(output_root, staging_root)
|
|
345
|
+
with tempfile.TemporaryDirectory(
|
|
346
|
+
prefix="mostlyright-authoring-", dir=scratch_root
|
|
347
|
+
) as directory:
|
|
348
|
+
index = _RecordIndex(Path(directory) / "records.sqlite3")
|
|
349
|
+
try:
|
|
350
|
+
index.build(staging, state, limits=limits)
|
|
351
|
+
if index.count != pass_map.unique_records:
|
|
352
|
+
raise CatalogAuthoringRefused(
|
|
353
|
+
"AUTHOR_RECORD_COUNT",
|
|
354
|
+
"authoring.records",
|
|
355
|
+
"indexed record population differs from the terminal pass map",
|
|
356
|
+
)
|
|
357
|
+
# The rule is applied to the whole corpus before one record is authored, so
|
|
358
|
+
# what it selects is a property of the corpus rather than of how far this
|
|
359
|
+
# invocation gets. `corpus_records` is the pass map's own count, never the
|
|
360
|
+
# caller's, so a coverage statement cannot overstate what it drew from.
|
|
361
|
+
coverage = selection.coverage(
|
|
362
|
+
corpus_records=pass_map.unique_records,
|
|
363
|
+
selected_records=index.select(selection),
|
|
364
|
+
)
|
|
365
|
+
_require_coverage_matches_retained(journal, coverage)
|
|
366
|
+
_refuse_disk(output, index, limits)
|
|
367
|
+
return _author_generation(
|
|
368
|
+
index,
|
|
369
|
+
output=output,
|
|
370
|
+
policy=policy,
|
|
371
|
+
staging_coordinate=_staging_coordinate(
|
|
372
|
+
state, pass_map, expected_fill_checkpoint_sha256
|
|
373
|
+
),
|
|
374
|
+
coverage=coverage,
|
|
375
|
+
limits=limits,
|
|
376
|
+
journal=journal,
|
|
377
|
+
deadline=deadline,
|
|
378
|
+
monotonic=monotonic,
|
|
379
|
+
)
|
|
380
|
+
finally:
|
|
381
|
+
index.close()
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# ------------------------------------------------------------------------------------------
|
|
385
|
+
# Converged staging
|
|
386
|
+
# ------------------------------------------------------------------------------------------
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _converged_state(staging: FillStaging, expected_checkpoint_sha256: str) -> dict[str, Any]:
|
|
390
|
+
state = staging.load()
|
|
391
|
+
if state is None or staging.state_digest() != expected_checkpoint_sha256:
|
|
392
|
+
raise CatalogAuthoringRefused(
|
|
393
|
+
"AUTHOR_FILL_CHECKPOINT_MISMATCH",
|
|
394
|
+
"fill.staging.state",
|
|
395
|
+
"no live checkpoint matches the caller's exact digest",
|
|
396
|
+
)
|
|
397
|
+
if "v4_checkpoint_schema_version" not in state or not isinstance(
|
|
398
|
+
state.get("evidence_indexes"), dict
|
|
399
|
+
):
|
|
400
|
+
raise CatalogAuthoringRefused(
|
|
401
|
+
"AUTHOR_FILL_PROTOCOL",
|
|
402
|
+
"fill.staging.state",
|
|
403
|
+
"streaming authoring reads only a segmented v4 generation",
|
|
404
|
+
)
|
|
405
|
+
if (
|
|
406
|
+
state.get("completed") is not True
|
|
407
|
+
or not _is_digest(state.get("final_map_sha256"))
|
|
408
|
+
or not _is_digest(state.get("last_pass_manifest_sha256"))
|
|
409
|
+
):
|
|
410
|
+
raise CatalogAuthoringRefused(
|
|
411
|
+
"AUTHOR_FILL_INCOMPLETE",
|
|
412
|
+
"fill.staging.state",
|
|
413
|
+
"authoring requires a converged terminal fill generation",
|
|
414
|
+
)
|
|
415
|
+
return state
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _selection_rank(record: Any) -> str:
|
|
419
|
+
"""The exact provider text ``last_harvested_date_desc`` orders this record by.
|
|
420
|
+
|
|
421
|
+
A record ranks by whatever the provider stated and nothing derived from it: no parsing, no
|
|
422
|
+
normalising, no reinterpretation. Ordering the provider's own bytes is what makes the rule
|
|
423
|
+
reproducible by anyone holding the staging generation, and for the ISO-8601 instants this
|
|
424
|
+
provider publishes, byte order *is* chronological order.
|
|
425
|
+
|
|
426
|
+
A value this rule cannot order by returns ``_UNRANKED``, which sorts below every value it
|
|
427
|
+
can. Five things are unorderable and all of them mean the same thing -- that the provider
|
|
428
|
+
published no usable harvest date for this record: no such observation at all, an observation
|
|
429
|
+
the harvester had to retain in a typed encoding rather than as canonical JSON, a value that
|
|
430
|
+
is not a string, a string too long to be any spelling of an instant, and a string that does
|
|
431
|
+
not open with a calendar date. None of them is treated as recent, because none of them is
|
|
432
|
+
evidence of recency -- and the last of them would otherwise be treated as the *most* recent
|
|
433
|
+
of all, since bytewise ``"unknown" > "2026-08-09T00:00:00Z"``.
|
|
434
|
+
"""
|
|
435
|
+
|
|
436
|
+
observations = record.get("observations") if isinstance(record, dict) else None
|
|
437
|
+
if not isinstance(observations, list):
|
|
438
|
+
return _UNRANKED
|
|
439
|
+
for field in observations:
|
|
440
|
+
if not isinstance(field, dict) or field.get("path") != _RANK_OBSERVATION_PATH:
|
|
441
|
+
continue
|
|
442
|
+
if field.get("encoding", "canonical_json") != "canonical_json":
|
|
443
|
+
return _UNRANKED
|
|
444
|
+
value = field.get("value")
|
|
445
|
+
if (
|
|
446
|
+
isinstance(value, str)
|
|
447
|
+
and 0 < len(value) <= MAX_SELECTION_RANK_CHARS
|
|
448
|
+
and _RANK_DATE_PREFIX.match(value) is not None
|
|
449
|
+
):
|
|
450
|
+
return value
|
|
451
|
+
return _UNRANKED
|
|
452
|
+
return _UNRANKED
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
class _RecordIndex:
|
|
456
|
+
"""A disk-backed identifier-ordered index over one terminal pass's normalized records."""
|
|
457
|
+
|
|
458
|
+
def __init__(self, path: Path) -> None:
|
|
459
|
+
self.path = path
|
|
460
|
+
self.count = 0
|
|
461
|
+
self.selected = 0
|
|
462
|
+
self.input_bytes = 0
|
|
463
|
+
self._bounded = False
|
|
464
|
+
self._connection = sqlite3.connect(path)
|
|
465
|
+
self._connection.execute("PRAGMA journal_mode=OFF")
|
|
466
|
+
self._connection.execute("PRAGMA synchronous=OFF")
|
|
467
|
+
self._connection.execute("PRAGMA temp_store=FILE")
|
|
468
|
+
self._connection.execute(
|
|
469
|
+
"CREATE TABLE record ("
|
|
470
|
+
"identifier BLOB NOT NULL PRIMARY KEY, semantic TEXT NOT NULL, "
|
|
471
|
+
"page TEXT NOT NULL, observed TEXT NOT NULL, rank TEXT NOT NULL, "
|
|
472
|
+
"payload BLOB NOT NULL"
|
|
473
|
+
") WITHOUT ROWID"
|
|
474
|
+
)
|
|
475
|
+
self._connection.execute(
|
|
476
|
+
"CREATE TABLE response (raw TEXT NOT NULL PRIMARY KEY, observed TEXT NOT NULL) "
|
|
477
|
+
"WITHOUT ROWID"
|
|
478
|
+
)
|
|
479
|
+
|
|
480
|
+
def close(self) -> None:
|
|
481
|
+
self._connection.close()
|
|
482
|
+
|
|
483
|
+
@property
|
|
484
|
+
def disk_bytes(self) -> int:
|
|
485
|
+
try:
|
|
486
|
+
return self.path.stat().st_size
|
|
487
|
+
except OSError:
|
|
488
|
+
return 0
|
|
489
|
+
|
|
490
|
+
def build(
|
|
491
|
+
self, staging: FillStaging, state: dict[str, Any], *, limits: AuthoringLimits
|
|
492
|
+
) -> None:
|
|
493
|
+
indexes = state["evidence_indexes"]
|
|
494
|
+
for item in staging.iter_index(indexes["response"], stream="response"):
|
|
495
|
+
raw = item.get("raw_response_sha256")
|
|
496
|
+
observed = item.get("observed_epoch_seconds")
|
|
497
|
+
if _is_digest(raw) and type(observed) is int:
|
|
498
|
+
self._connection.execute(
|
|
499
|
+
"INSERT INTO response(raw, observed) VALUES (?, ?) ON CONFLICT(raw) DO NOTHING",
|
|
500
|
+
(raw, _utc_second(observed)),
|
|
501
|
+
)
|
|
502
|
+
self._connection.commit()
|
|
503
|
+
for item in staging.iter_index(indexes["page"], stream="page"):
|
|
504
|
+
if item.get("pass") != state["pass_number"]:
|
|
505
|
+
continue
|
|
506
|
+
self._read_page(staging, item["normalized_sha256"], limits=limits)
|
|
507
|
+
self._connection.commit()
|
|
508
|
+
self.count = int(self._connection.execute("SELECT count(*) FROM record").fetchone()[0])
|
|
509
|
+
|
|
510
|
+
def _read_page(
|
|
511
|
+
self, staging: FillStaging, page_sha256: str, *, limits: AuthoringLimits
|
|
512
|
+
) -> None:
|
|
513
|
+
manifest = staging.read_shard("normalized", page_sha256)
|
|
514
|
+
if manifest.get("schema_version") != _NORMALIZED_PAGE_SCHEMA or not isinstance(
|
|
515
|
+
manifest.get("record_segments"), list
|
|
516
|
+
):
|
|
517
|
+
raise CatalogAuthoringRefused(
|
|
518
|
+
"AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized page manifest is invalid"
|
|
519
|
+
)
|
|
520
|
+
for segment in manifest["record_segments"]:
|
|
521
|
+
payload = staging.read_shard("normalized", segment["sha256"])
|
|
522
|
+
if payload.get("schema_version") != _NORMALIZED_SEGMENT_SCHEMA or not isinstance(
|
|
523
|
+
payload.get("records"), list
|
|
524
|
+
):
|
|
525
|
+
raise CatalogAuthoringRefused(
|
|
526
|
+
"AUTHOR_STAGING_CORRUPT",
|
|
527
|
+
"fill.normalized",
|
|
528
|
+
"normalized record segment is invalid",
|
|
529
|
+
)
|
|
530
|
+
for record in payload["records"]:
|
|
531
|
+
self._insert(record, page_sha256=page_sha256, limits=limits)
|
|
532
|
+
|
|
533
|
+
def _insert(self, record: Any, *, page_sha256: str, limits: AuthoringLimits) -> None:
|
|
534
|
+
if (
|
|
535
|
+
not isinstance(record, dict)
|
|
536
|
+
or not isinstance(record.get("record_id"), str)
|
|
537
|
+
or not _is_digest(record.get("normalized_sha256"))
|
|
538
|
+
or not _is_digest(record.get("raw_response_sha256"))
|
|
539
|
+
):
|
|
540
|
+
raise CatalogAuthoringRefused(
|
|
541
|
+
"AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized record is invalid"
|
|
542
|
+
)
|
|
543
|
+
try:
|
|
544
|
+
raw = canonical_json_bytes(record)
|
|
545
|
+
except CanonicalJSONError as error:
|
|
546
|
+
raise CatalogAuthoringRefused(
|
|
547
|
+
"AUTHOR_STAGING_CORRUPT", "fill.normalized", "normalized record is not canonical"
|
|
548
|
+
) from error
|
|
549
|
+
self.input_bytes += len(raw)
|
|
550
|
+
if self.input_bytes > limits.max_input_bytes:
|
|
551
|
+
raise CatalogAuthoringRefused(
|
|
552
|
+
"AUTHOR_INPUT_LIMIT",
|
|
553
|
+
"authoring.input",
|
|
554
|
+
"normalized input exceeds the caller's byte bound",
|
|
555
|
+
)
|
|
556
|
+
row = self._connection.execute(
|
|
557
|
+
"SELECT observed FROM response WHERE raw = ?", (record["raw_response_sha256"],)
|
|
558
|
+
).fetchone()
|
|
559
|
+
if row is None:
|
|
560
|
+
raise CatalogAuthoringRefused(
|
|
561
|
+
"AUTHOR_STAGING_CORRUPT",
|
|
562
|
+
"fill.normalized",
|
|
563
|
+
"normalized record cites no retained response observation",
|
|
564
|
+
)
|
|
565
|
+
identifier = record["record_id"].encode("utf-8")
|
|
566
|
+
existing = self._connection.execute(
|
|
567
|
+
"SELECT semantic FROM record WHERE identifier = ?", (identifier,)
|
|
568
|
+
).fetchone()
|
|
569
|
+
if existing is not None:
|
|
570
|
+
if existing[0] != record["normalized_sha256"]:
|
|
571
|
+
raise CatalogAuthoringRefused(
|
|
572
|
+
"AUTHOR_RECORD_CONFLICT",
|
|
573
|
+
"authoring.records",
|
|
574
|
+
"one provider identifier carries two different semantic digests",
|
|
575
|
+
)
|
|
576
|
+
return
|
|
577
|
+
self._connection.execute(
|
|
578
|
+
"INSERT INTO record(identifier, semantic, page, observed, rank, payload) "
|
|
579
|
+
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
580
|
+
(
|
|
581
|
+
identifier,
|
|
582
|
+
record["normalized_sha256"],
|
|
583
|
+
page_sha256,
|
|
584
|
+
row[0],
|
|
585
|
+
_selection_rank(record),
|
|
586
|
+
raw,
|
|
587
|
+
),
|
|
588
|
+
)
|
|
589
|
+
|
|
590
|
+
def select(self, selection: AuthoringSelection) -> int:
|
|
591
|
+
"""Apply one selection rule to the whole indexed corpus and answer what it selected.
|
|
592
|
+
|
|
593
|
+
The selection is a set, computed once, before any record is authored. Authoring still
|
|
594
|
+
walks provider identifiers in ascending order afterwards, because that order is what the
|
|
595
|
+
journal resumes on, what a shard's two range identifiers describe, and what every reader
|
|
596
|
+
of an authored generation merges on. Ranking decides *which* records; the identifier
|
|
597
|
+
decides *when* each one is authored. Confusing the two would make the selection depend
|
|
598
|
+
on where an interrupted run stopped, which is the truncation this whole path exists to
|
|
599
|
+
avoid.
|
|
600
|
+
"""
|
|
601
|
+
|
|
602
|
+
if selection.bound is None:
|
|
603
|
+
self.selected = self.count
|
|
604
|
+
return self.selected
|
|
605
|
+
# The ordering index is what keeps the selection from re-sorting the whole corpus on
|
|
606
|
+
# every read; its pages live inside the index file, which `_refuse_disk` measures. The
|
|
607
|
+
# sort that *builds* it spills to SQLite's temp directory under `temp_store=FILE`, which
|
|
608
|
+
# is transient and outside that measure -- so the bound's disk cost is accounted where it
|
|
609
|
+
# is durable and merely bounded by the corpus where it is not. A bound is the only thing
|
|
610
|
+
# that needs an order other than the primary key, so an exhaustive authoring pays neither.
|
|
611
|
+
self._connection.execute("CREATE INDEX record_rank ON record(rank DESC, identifier ASC)")
|
|
612
|
+
self._connection.execute(
|
|
613
|
+
"CREATE TABLE selected (identifier BLOB NOT NULL PRIMARY KEY) WITHOUT ROWID"
|
|
614
|
+
)
|
|
615
|
+
self._connection.execute(
|
|
616
|
+
"INSERT INTO selected(identifier) "
|
|
617
|
+
"SELECT identifier FROM record ORDER BY rank DESC, identifier ASC LIMIT ?",
|
|
618
|
+
(selection.bound,),
|
|
619
|
+
)
|
|
620
|
+
self._connection.commit()
|
|
621
|
+
self._bounded = True
|
|
622
|
+
self.selected = int(self._connection.execute("SELECT count(*) FROM selected").fetchone()[0])
|
|
623
|
+
return self.selected
|
|
624
|
+
|
|
625
|
+
def stream(self, after: str | None) -> Iterator[tuple[str, str, str, bytes]]:
|
|
626
|
+
source = (
|
|
627
|
+
"record JOIN selected ON selected.identifier = record.identifier"
|
|
628
|
+
if self._bounded
|
|
629
|
+
else "record"
|
|
630
|
+
)
|
|
631
|
+
columns = "record.identifier, record.page, record.observed, record.payload"
|
|
632
|
+
if after is None:
|
|
633
|
+
cursor = self._connection.execute(
|
|
634
|
+
f"SELECT {columns} FROM {source} ORDER BY record.identifier"
|
|
635
|
+
)
|
|
636
|
+
else:
|
|
637
|
+
cursor = self._connection.execute(
|
|
638
|
+
f"SELECT {columns} FROM {source} "
|
|
639
|
+
"WHERE record.identifier > ? ORDER BY record.identifier",
|
|
640
|
+
(after.encode("utf-8"),),
|
|
641
|
+
)
|
|
642
|
+
while rows := cursor.fetchmany(64):
|
|
643
|
+
for identifier, page, observed, payload in rows:
|
|
644
|
+
yield identifier.decode("utf-8"), page, observed, payload
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
# ------------------------------------------------------------------------------------------
|
|
648
|
+
# Authoring
|
|
649
|
+
# ------------------------------------------------------------------------------------------
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _author_generation(
|
|
653
|
+
index: _RecordIndex,
|
|
654
|
+
*,
|
|
655
|
+
output: _AuthoringOutput,
|
|
656
|
+
policy: AuthoringPolicy,
|
|
657
|
+
staging_coordinate: dict[str, Any],
|
|
658
|
+
coverage: dict[str, Any],
|
|
659
|
+
limits: AuthoringLimits,
|
|
660
|
+
journal: dict[str, Any],
|
|
661
|
+
deadline: float,
|
|
662
|
+
monotonic: Callable[[], float],
|
|
663
|
+
) -> AuthoringResult:
|
|
664
|
+
shards: list[dict[str, Any]] = list(journal["shards"])
|
|
665
|
+
counts = dict(journal["counts"])
|
|
666
|
+
digests = dict(journal["layer_text_digests"])
|
|
667
|
+
last_id: str | None = journal["last_provider_record_id"]
|
|
668
|
+
buffered: list[dict[str, Any]] = []
|
|
669
|
+
buffered_bytes = 0
|
|
670
|
+
stop: str | None = None
|
|
671
|
+
|
|
672
|
+
def flush() -> None:
|
|
673
|
+
nonlocal buffered, buffered_bytes
|
|
674
|
+
if not buffered:
|
|
675
|
+
return
|
|
676
|
+
payload = {
|
|
677
|
+
"schema_version": AUTHORING_SHARD_SCHEMA,
|
|
678
|
+
"policy_sha256": policy.digest,
|
|
679
|
+
"first_provider_record_id": buffered[0]["provider_record_id"],
|
|
680
|
+
"last_provider_record_id": buffered[-1]["provider_record_id"],
|
|
681
|
+
"items": buffered,
|
|
682
|
+
}
|
|
683
|
+
digest, size = output.write_shard(payload, maximum=limits.max_shard_bytes)
|
|
684
|
+
shards.append(
|
|
685
|
+
{
|
|
686
|
+
"sha256": digest,
|
|
687
|
+
"bytes": size,
|
|
688
|
+
"item_count": len(buffered),
|
|
689
|
+
"entry_count": sum(1 for item in buffered if item["entry"] is not None),
|
|
690
|
+
"first_provider_record_id": payload["first_provider_record_id"],
|
|
691
|
+
"last_provider_record_id": payload["last_provider_record_id"],
|
|
692
|
+
}
|
|
693
|
+
)
|
|
694
|
+
buffered = []
|
|
695
|
+
buffered_bytes = 0
|
|
696
|
+
_refuse_disk(output, index, limits)
|
|
697
|
+
|
|
698
|
+
for identifier, page_sha256, observed_at, payload in index.stream(last_id):
|
|
699
|
+
if counts["records"] >= limits.max_records:
|
|
700
|
+
stop = "AUTHOR_RECORD_LIMIT"
|
|
701
|
+
break
|
|
702
|
+
if monotonic() >= deadline:
|
|
703
|
+
stop = "AUTHOR_WALL_LIMIT"
|
|
704
|
+
break
|
|
705
|
+
authored = author_catalog_record(
|
|
706
|
+
parse_canonical_json(payload),
|
|
707
|
+
policy=policy,
|
|
708
|
+
observed_at=observed_at,
|
|
709
|
+
page_evidence_sha256=page_sha256,
|
|
710
|
+
)
|
|
711
|
+
item = authored.to_dict()
|
|
712
|
+
item_bytes = len(canonical_json_bytes(item))
|
|
713
|
+
if buffered and (
|
|
714
|
+
len(buffered) >= limits.max_shard_entries
|
|
715
|
+
or buffered_bytes + item_bytes + len(buffered) + _SHARD_ENVELOPE_BYTES
|
|
716
|
+
> limits.max_shard_bytes
|
|
717
|
+
):
|
|
718
|
+
flush()
|
|
719
|
+
buffered.append(item)
|
|
720
|
+
buffered_bytes += item_bytes
|
|
721
|
+
counts["records"] += 1
|
|
722
|
+
counts[authored.disposition] += 1
|
|
723
|
+
counts["layer_text_truncated"] += len(authored.truncated_layers)
|
|
724
|
+
if authored.entry is not None:
|
|
725
|
+
for layer, text in authored.layer_texts:
|
|
726
|
+
digests[layer] = _layer_chain(digests[layer], authored.entry.entry_id, layer, text)
|
|
727
|
+
last_id = identifier
|
|
728
|
+
flush()
|
|
729
|
+
|
|
730
|
+
# The journal carries the coverage of the generation it continues, so a resume that restated
|
|
731
|
+
# the rule or the bound is refused against retained evidence rather than against an argument
|
|
732
|
+
# this process happens to hold.
|
|
733
|
+
journal_sha256 = output.publish(
|
|
734
|
+
JOURNAL_FILENAME,
|
|
735
|
+
{
|
|
736
|
+
"schema_version": AUTHORING_JOURNAL_SCHEMA,
|
|
737
|
+
"policy_sha256": policy.digest,
|
|
738
|
+
"staging_checkpoint_sha256": staging_coordinate["checkpoint_sha256"],
|
|
739
|
+
"coverage": coverage,
|
|
740
|
+
"shards": shards,
|
|
741
|
+
"counts": counts,
|
|
742
|
+
"layer_text_digests": digests,
|
|
743
|
+
"last_provider_record_id": last_id,
|
|
744
|
+
"reason_code": stop,
|
|
745
|
+
},
|
|
746
|
+
maximum=MAX_AUTHORING_JOURNAL_BYTES,
|
|
747
|
+
)
|
|
748
|
+
_refuse_disk(output, index, limits)
|
|
749
|
+
# The manifest is written only when the stream ran out, which under a selection means the
|
|
750
|
+
# selected population was authored in full: `counts["records"]` and
|
|
751
|
+
# `coverage["selected_records"]` cannot differ here without one of them having been forged
|
|
752
|
+
# after the fact. That is a reader's question rather than a writer's, and every stage that
|
|
753
|
+
# reads an authored manifest asks it -- see `admit_authored_generation` and the publisher's
|
|
754
|
+
# receipt reconciliation, both of which hold a document they did not write.
|
|
755
|
+
manifest_sha256: str | None = None
|
|
756
|
+
if stop is None:
|
|
757
|
+
manifest_sha256 = output.publish(
|
|
758
|
+
MANIFEST_FILENAME,
|
|
759
|
+
_manifest(policy, staging_coordinate, coverage, counts, digests, shards),
|
|
760
|
+
maximum=MAX_AUTHORING_MANIFEST_BYTES,
|
|
761
|
+
)
|
|
762
|
+
return AuthoringResult(
|
|
763
|
+
status="complete" if stop is None else "incomplete",
|
|
764
|
+
reason_code=stop,
|
|
765
|
+
manifest_sha256=manifest_sha256,
|
|
766
|
+
journal_sha256=journal_sha256,
|
|
767
|
+
coverage=coverage,
|
|
768
|
+
receipt={
|
|
769
|
+
"schema_version": AUTHORING_RECEIPT_SCHEMA,
|
|
770
|
+
"status": "complete" if stop is None else "incomplete",
|
|
771
|
+
"reason_code": stop,
|
|
772
|
+
"policy_id": policy.policy_id,
|
|
773
|
+
"policy_sha256": policy.digest,
|
|
774
|
+
"provider_id": policy.provider_id,
|
|
775
|
+
"harvester_coordinate": policy.harvester_coordinate,
|
|
776
|
+
"staging": staging_coordinate,
|
|
777
|
+
"coverage": coverage,
|
|
778
|
+
"counts": counts,
|
|
779
|
+
"layer_text_digests": digests,
|
|
780
|
+
"shard_count": len(shards),
|
|
781
|
+
"shards": shards,
|
|
782
|
+
"limits": limits.to_dict(),
|
|
783
|
+
"manifest_sha256": manifest_sha256,
|
|
784
|
+
"journal_sha256": journal_sha256,
|
|
785
|
+
"output_bytes": output.bytes_written,
|
|
786
|
+
},
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def _manifest(
|
|
791
|
+
policy: AuthoringPolicy,
|
|
792
|
+
staging: dict[str, Any],
|
|
793
|
+
coverage: dict[str, Any],
|
|
794
|
+
counts: dict[str, int],
|
|
795
|
+
digests: dict[str, str | None],
|
|
796
|
+
shards: list[dict[str, Any]],
|
|
797
|
+
) -> dict[str, Any]:
|
|
798
|
+
body = {
|
|
799
|
+
"schema_version": AUTHORING_MANIFEST_SCHEMA,
|
|
800
|
+
"policy_id": policy.policy_id,
|
|
801
|
+
"policy_sha256": policy.digest,
|
|
802
|
+
"staging": staging,
|
|
803
|
+
"coverage": coverage,
|
|
804
|
+
"counts": counts,
|
|
805
|
+
"layer_text_digests": digests,
|
|
806
|
+
"shards": shards,
|
|
807
|
+
}
|
|
808
|
+
return {**body, "root_sha256": canonical_sha256(body)}
|
|
809
|
+
|
|
810
|
+
|
|
811
|
+
def _staging_coordinate(state: dict[str, Any], pass_map, checkpoint_sha256: str) -> dict[str, Any]:
|
|
812
|
+
indexes = state["evidence_indexes"]
|
|
813
|
+
return {
|
|
814
|
+
"checkpoint_sha256": checkpoint_sha256,
|
|
815
|
+
"config_sha256": state["config_sha256"],
|
|
816
|
+
"endpoint": state["endpoint"],
|
|
817
|
+
"harvester_coordinate": state["harvester_coordinate"],
|
|
818
|
+
"pass_number": state["pass_number"],
|
|
819
|
+
"pass_map_sha256": pass_map.manifest_sha256,
|
|
820
|
+
"pass_map_root_sha256": pass_map.root_sha256,
|
|
821
|
+
"unique_records": pass_map.unique_records,
|
|
822
|
+
"raw_response_index_root_sha256": indexes["response"]["root_sha256"],
|
|
823
|
+
"raw_response_index_count": indexes["response"]["count"],
|
|
824
|
+
"normalized_page_index_root_sha256": indexes["page"]["root_sha256"],
|
|
825
|
+
"normalized_page_index_count": indexes["page"]["count"],
|
|
826
|
+
"record_index_root_sha256": indexes["record"]["root_sha256"],
|
|
827
|
+
"record_index_count": indexes["record"]["count"],
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
def _layer_chain(previous: str | None, entry_id: str, layer: str, text: str) -> str:
|
|
832
|
+
return canonical_sha256(
|
|
833
|
+
{
|
|
834
|
+
"previous_sha256": previous,
|
|
835
|
+
"entry_id": entry_id,
|
|
836
|
+
"layer": layer,
|
|
837
|
+
"text_sha256": sha256_bytes(text.encode("utf-8")),
|
|
838
|
+
}
|
|
839
|
+
)
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _refuse_disk(output: _AuthoringOutput, index: _RecordIndex, limits: AuthoringLimits) -> None:
|
|
843
|
+
if output.bytes_written + index.disk_bytes > limits.max_disk_bytes:
|
|
844
|
+
raise CatalogAuthoringRefused(
|
|
845
|
+
"AUTHOR_DISK_LIMIT", "authoring.output", "authoring exceeds its local disk bound"
|
|
846
|
+
)
|
|
847
|
+
|
|
848
|
+
|
|
849
|
+
def _require_bounds_admit_retained(journal: dict[str, Any], limits: AuthoringLimits) -> None:
|
|
850
|
+
"""A resume may not narrow a bound the generation it continues has already passed.
|
|
851
|
+
|
|
852
|
+
The receipt an authoring stage writes states the bounds of the invocation that *finished* the
|
|
853
|
+
generation, and a publication reconciles those bounds against every shard in the manifest --
|
|
854
|
+
including the ones earlier invocations wrote. Narrowing a bound mid-generation would produce
|
|
855
|
+
a complete generation that no publication can accept, and it would say so hours later, at the
|
|
856
|
+
publisher, about work that is already durable. It is refused here instead, where the
|
|
857
|
+
narrowing happened and before any further record is authored.
|
|
858
|
+
|
|
859
|
+
A fresh authoring resumes an empty journal, so nothing here constrains a first invocation.
|
|
860
|
+
"""
|
|
861
|
+
|
|
862
|
+
shards = journal["shards"]
|
|
863
|
+
if (
|
|
864
|
+
limits.max_records < journal["counts"]["records"]
|
|
865
|
+
or any(limits.max_shard_entries < descriptor["item_count"] for descriptor in shards)
|
|
866
|
+
or any(limits.max_shard_bytes < descriptor["bytes"] for descriptor in shards)
|
|
867
|
+
):
|
|
868
|
+
raise CatalogAuthoringRefused(
|
|
869
|
+
"AUTHOR_LIMIT",
|
|
870
|
+
"limits",
|
|
871
|
+
"a resume may not narrow a bound the retained generation has already passed",
|
|
872
|
+
)
|
|
873
|
+
|
|
874
|
+
|
|
875
|
+
def _require_coverage_matches_retained(journal: dict[str, Any], coverage: dict[str, Any]) -> None:
|
|
876
|
+
"""A resume continues one coverage claim; it never restates it.
|
|
877
|
+
|
|
878
|
+
``_require_bounds_admit_retained`` lets a resume widen a *bound* because a bound is a
|
|
879
|
+
permission to spend and a wider one strands nothing. Coverage is not that. It is the claim
|
|
880
|
+
the generation makes about which records it holds, and shards written under one claim plus
|
|
881
|
+
shards written under another are a generation that describes neither. So this is equality,
|
|
882
|
+
in both directions, and it is checked against the retained journal rather than against a
|
|
883
|
+
caller's argument.
|
|
884
|
+
|
|
885
|
+
A fresh authoring resumes an empty journal, so nothing here constrains a first invocation.
|
|
886
|
+
"""
|
|
887
|
+
|
|
888
|
+
retained = journal["coverage"]
|
|
889
|
+
if retained is not None and retained != coverage:
|
|
890
|
+
raise CatalogAuthoringRefused(
|
|
891
|
+
"AUTHOR_COVERAGE_MISMATCH",
|
|
892
|
+
"authoring.coverage",
|
|
893
|
+
"the retained partial generation was authored under another stated coverage",
|
|
894
|
+
)
|
|
895
|
+
|
|
896
|
+
|
|
897
|
+
def _empty_journal() -> dict[str, Any]:
|
|
898
|
+
return {
|
|
899
|
+
"coverage": None,
|
|
900
|
+
"shards": [],
|
|
901
|
+
"counts": {
|
|
902
|
+
"records": 0,
|
|
903
|
+
"authored": 0,
|
|
904
|
+
"flagged": 0,
|
|
905
|
+
"skipped": 0,
|
|
906
|
+
"failed": 0,
|
|
907
|
+
"layer_text_truncated": 0,
|
|
908
|
+
},
|
|
909
|
+
"layer_text_digests": dict.fromkeys(EMBEDDING_LAYERS),
|
|
910
|
+
"last_provider_record_id": None,
|
|
911
|
+
}
|
|
912
|
+
|
|
913
|
+
|
|
914
|
+
def _resume_point(
|
|
915
|
+
output: _AuthoringOutput,
|
|
916
|
+
expected_journal_sha256: str | None,
|
|
917
|
+
policy: AuthoringPolicy,
|
|
918
|
+
expected_fill_checkpoint_sha256: str,
|
|
919
|
+
selection: AuthoringSelection,
|
|
920
|
+
) -> dict[str, Any]:
|
|
921
|
+
manifest = output.read(MANIFEST_FILENAME, maximum=MAX_AUTHORING_MANIFEST_BYTES)
|
|
922
|
+
if manifest is not None:
|
|
923
|
+
raise CatalogAuthoringRefused(
|
|
924
|
+
"AUTHOR_OUTPUT_EXISTS",
|
|
925
|
+
"authoring.output",
|
|
926
|
+
"a published generation is immutable; author into a new output root",
|
|
927
|
+
)
|
|
928
|
+
raw = output.read(JOURNAL_FILENAME, maximum=MAX_AUTHORING_JOURNAL_BYTES)
|
|
929
|
+
if expected_journal_sha256 is None:
|
|
930
|
+
if raw is not None:
|
|
931
|
+
raise CatalogAuthoringRefused(
|
|
932
|
+
"AUTHOR_OUTPUT_EXISTS",
|
|
933
|
+
"authoring.output",
|
|
934
|
+
"a partial generation exists; resume it with its exact journal digest",
|
|
935
|
+
)
|
|
936
|
+
return _empty_journal()
|
|
937
|
+
if raw is None or sha256_bytes(raw) != expected_journal_sha256:
|
|
938
|
+
raise CatalogAuthoringRefused(
|
|
939
|
+
"AUTHOR_JOURNAL_MISMATCH",
|
|
940
|
+
"authoring.journal",
|
|
941
|
+
"no retained journal matches the caller's exact digest",
|
|
942
|
+
)
|
|
943
|
+
try:
|
|
944
|
+
journal = parse_canonical_json(raw)
|
|
945
|
+
except CanonicalJSONError as error:
|
|
946
|
+
raise CatalogAuthoringRefused(
|
|
947
|
+
"AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "journal is not canonical"
|
|
948
|
+
) from error
|
|
949
|
+
if (
|
|
950
|
+
not isinstance(journal, dict)
|
|
951
|
+
or journal.get("schema_version") != AUTHORING_JOURNAL_SCHEMA
|
|
952
|
+
or not isinstance(journal.get("shards"), list)
|
|
953
|
+
or not isinstance(journal.get("counts"), dict)
|
|
954
|
+
or not isinstance(journal.get("layer_text_digests"), dict)
|
|
955
|
+
or set(journal["layer_text_digests"]) != set(EMBEDDING_LAYERS)
|
|
956
|
+
or set(journal["counts"]) != set(_empty_journal()["counts"])
|
|
957
|
+
or not isinstance(journal.get("last_provider_record_id"), (str, type(None)))
|
|
958
|
+
or not _is_digest(journal.get("policy_sha256"))
|
|
959
|
+
or not _is_digest(journal.get("staging_checkpoint_sha256"))
|
|
960
|
+
or not coverage_is_valid(journal.get("coverage"))
|
|
961
|
+
):
|
|
962
|
+
raise CatalogAuthoringRefused(
|
|
963
|
+
"AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "journal contract differs"
|
|
964
|
+
)
|
|
965
|
+
if (
|
|
966
|
+
journal["policy_sha256"] != policy.digest
|
|
967
|
+
or journal["staging_checkpoint_sha256"] != expected_fill_checkpoint_sha256
|
|
968
|
+
):
|
|
969
|
+
raise CatalogAuthoringRefused(
|
|
970
|
+
"AUTHOR_JOURNAL_COORDINATE",
|
|
971
|
+
"authoring.journal",
|
|
972
|
+
"the retained partial generation was authored from another policy or checkpoint",
|
|
973
|
+
)
|
|
974
|
+
# The rule and the bound are the caller's half of the claim and are answerable here, before
|
|
975
|
+
# the staging root is opened or one record is re-read. The corpus and selected populations
|
|
976
|
+
# are the generation's half; they are recomputed from immutable staging and checked against
|
|
977
|
+
# this same journal once the index exists.
|
|
978
|
+
if (journal["coverage"]["rule"], journal["coverage"]["bound"]) != (
|
|
979
|
+
selection.rule,
|
|
980
|
+
selection.bound,
|
|
981
|
+
):
|
|
982
|
+
raise CatalogAuthoringRefused(
|
|
983
|
+
"AUTHOR_COVERAGE_MISMATCH",
|
|
984
|
+
"authoring.journal.coverage",
|
|
985
|
+
"the retained partial generation was authored under another stated coverage",
|
|
986
|
+
)
|
|
987
|
+
for descriptor in journal["shards"]:
|
|
988
|
+
output.verify_shard(descriptor)
|
|
989
|
+
return journal
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
def _scratch_root(output_root: Path, staging_root: Path) -> str | None:
|
|
993
|
+
for candidate in (Path(output_root).parent, Path(staging_root).parent):
|
|
994
|
+
if candidate.is_dir():
|
|
995
|
+
return str(candidate)
|
|
996
|
+
return None
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def _utc_second(epoch_seconds: int) -> str:
|
|
1000
|
+
try:
|
|
1001
|
+
return datetime.fromtimestamp(epoch_seconds, UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
1002
|
+
except (OverflowError, OSError, ValueError) as error:
|
|
1003
|
+
raise CatalogAuthoringRefused(
|
|
1004
|
+
"AUTHOR_STAGING_CORRUPT",
|
|
1005
|
+
"fill.staging.response",
|
|
1006
|
+
"retained response observation is outside the representable range",
|
|
1007
|
+
) from error
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
def _is_digest(value: Any) -> bool:
|
|
1011
|
+
return (
|
|
1012
|
+
isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
|
|
1013
|
+
)
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
# ------------------------------------------------------------------------------------------
|
|
1017
|
+
# Durable, descriptor-rooted, no-follow output
|
|
1018
|
+
# ------------------------------------------------------------------------------------------
|
|
1019
|
+
|
|
1020
|
+
|
|
1021
|
+
class _AuthoringOutput:
|
|
1022
|
+
"""One locally confined output root whose members install atomically and read back exactly."""
|
|
1023
|
+
|
|
1024
|
+
def __init__(self, root: Path, *, root_descriptor: int | None = None) -> None:
|
|
1025
|
+
self.root = Path(root)
|
|
1026
|
+
self._retained_root_descriptor = root_descriptor
|
|
1027
|
+
self.bytes_written = 0
|
|
1028
|
+
self._root_fd = -1
|
|
1029
|
+
self._shards_fd = -1
|
|
1030
|
+
self._owned: list[int] = []
|
|
1031
|
+
|
|
1032
|
+
@contextlib.contextmanager
|
|
1033
|
+
def opened(self) -> Iterator[None]:
|
|
1034
|
+
try:
|
|
1035
|
+
if self._retained_root_descriptor is None:
|
|
1036
|
+
self.root.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
1037
|
+
self._root_fd = self._open_directory(self.root)
|
|
1038
|
+
(self.root / SHARDS_DIRNAME).mkdir(exist_ok=True, mode=0o700)
|
|
1039
|
+
self._shards_fd = self._open_directory(self.root / SHARDS_DIRNAME)
|
|
1040
|
+
else:
|
|
1041
|
+
self._root_fd = os.dup(self._retained_root_descriptor)
|
|
1042
|
+
self._owned.append(self._root_fd)
|
|
1043
|
+
self._require_private_directory(self._root_fd)
|
|
1044
|
+
try:
|
|
1045
|
+
os.mkdir(SHARDS_DIRNAME, mode=0o700, dir_fd=self._root_fd)
|
|
1046
|
+
os.fsync(self._root_fd)
|
|
1047
|
+
except FileExistsError:
|
|
1048
|
+
pass
|
|
1049
|
+
self._shards_fd = os.open(SHARDS_DIRNAME, _OPEN_DIRECTORY, dir_fd=self._root_fd)
|
|
1050
|
+
self._owned.append(self._shards_fd)
|
|
1051
|
+
self._require_private_directory(self._shards_fd)
|
|
1052
|
+
except CatalogAuthoringRefused:
|
|
1053
|
+
self._release()
|
|
1054
|
+
raise
|
|
1055
|
+
except OSError as error:
|
|
1056
|
+
self._release()
|
|
1057
|
+
raise CatalogAuthoringRefused(
|
|
1058
|
+
"AUTHOR_OUTPUT_IO", "authoring.output", "authoring output root is unusable"
|
|
1059
|
+
) from error
|
|
1060
|
+
try:
|
|
1061
|
+
yield
|
|
1062
|
+
finally:
|
|
1063
|
+
self._release()
|
|
1064
|
+
|
|
1065
|
+
def _release(self) -> None:
|
|
1066
|
+
for descriptor in reversed(self._owned):
|
|
1067
|
+
os.close(descriptor)
|
|
1068
|
+
self._owned = []
|
|
1069
|
+
self._root_fd = -1
|
|
1070
|
+
self._shards_fd = -1
|
|
1071
|
+
|
|
1072
|
+
def _open_directory(self, path: Path) -> int:
|
|
1073
|
+
descriptor = os.open(path, _OPEN_DIRECTORY)
|
|
1074
|
+
self._owned.append(descriptor)
|
|
1075
|
+
self._require_private_directory(descriptor)
|
|
1076
|
+
return descriptor
|
|
1077
|
+
|
|
1078
|
+
@staticmethod
|
|
1079
|
+
def _require_private_directory(descriptor: int) -> None:
|
|
1080
|
+
info = os.fstat(descriptor)
|
|
1081
|
+
if not stat.S_ISDIR(info.st_mode) or stat.S_IMODE(info.st_mode) & 0o077:
|
|
1082
|
+
raise CatalogAuthoringRefused(
|
|
1083
|
+
"AUTHOR_OUTPUT_PATH",
|
|
1084
|
+
"authoring.output",
|
|
1085
|
+
"authoring output components must be private directories",
|
|
1086
|
+
)
|
|
1087
|
+
|
|
1088
|
+
def read(self, name: str, *, maximum: int) -> bytes | None:
|
|
1089
|
+
try:
|
|
1090
|
+
return _read_regular_at(self._root_fd, name, maximum=maximum)
|
|
1091
|
+
except FileNotFoundError:
|
|
1092
|
+
return None
|
|
1093
|
+
|
|
1094
|
+
def write_shard(self, payload: dict[str, Any], *, maximum: int) -> tuple[str, int]:
|
|
1095
|
+
raw = canonical_json_bytes(payload)
|
|
1096
|
+
if len(raw) > maximum:
|
|
1097
|
+
raise CatalogAuthoringRefused(
|
|
1098
|
+
"AUTHOR_SHARD_LIMIT", "authoring.shard", "shard exceeds its byte bound"
|
|
1099
|
+
)
|
|
1100
|
+
digest = sha256_bytes(raw)
|
|
1101
|
+
self._install_immutable(f"{digest}.json", raw)
|
|
1102
|
+
return digest, len(raw)
|
|
1103
|
+
|
|
1104
|
+
def verify_shard(self, descriptor: Any) -> None:
|
|
1105
|
+
if not shard_descriptor_is_valid(descriptor):
|
|
1106
|
+
raise CatalogAuthoringRefused(
|
|
1107
|
+
"AUTHOR_JOURNAL_CORRUPT", "authoring.journal", "shard descriptor is invalid"
|
|
1108
|
+
)
|
|
1109
|
+
try:
|
|
1110
|
+
raw = _read_regular_at(
|
|
1111
|
+
self._shards_fd, f"{descriptor['sha256']}.json", maximum=descriptor["bytes"]
|
|
1112
|
+
)
|
|
1113
|
+
except OSError as error:
|
|
1114
|
+
raise CatalogAuthoringRefused(
|
|
1115
|
+
"AUTHOR_OUTPUT_READBACK",
|
|
1116
|
+
"authoring.shard",
|
|
1117
|
+
"a committed shard is absent or unreadable",
|
|
1118
|
+
) from error
|
|
1119
|
+
if len(raw) != descriptor["bytes"] or sha256_bytes(raw) != descriptor["sha256"]:
|
|
1120
|
+
raise CatalogAuthoringRefused(
|
|
1121
|
+
"AUTHOR_OUTPUT_READBACK", "authoring.shard", "a committed shard differs"
|
|
1122
|
+
)
|
|
1123
|
+
|
|
1124
|
+
def _install_immutable(self, name: str, raw: bytes) -> None:
|
|
1125
|
+
try:
|
|
1126
|
+
existing = _read_regular_at(self._shards_fd, name, maximum=len(raw))
|
|
1127
|
+
except FileNotFoundError:
|
|
1128
|
+
existing = None
|
|
1129
|
+
except OSError as error:
|
|
1130
|
+
raise CatalogAuthoringRefused(
|
|
1131
|
+
"AUTHOR_OUTPUT_IO", "authoring.shard", "immutable shard lookup failed"
|
|
1132
|
+
) from error
|
|
1133
|
+
if existing is not None:
|
|
1134
|
+
if existing != raw:
|
|
1135
|
+
raise CatalogAuthoringRefused(
|
|
1136
|
+
"AUTHOR_OUTPUT_READBACK", "authoring.shard", "digest path bytes differ"
|
|
1137
|
+
)
|
|
1138
|
+
return
|
|
1139
|
+
temporary = f".publish.{os.getpid()}.{secrets.token_hex(16)}.tmp"
|
|
1140
|
+
try:
|
|
1141
|
+
_write_regular_at(self._shards_fd, temporary, raw)
|
|
1142
|
+
with contextlib.suppress(FileExistsError):
|
|
1143
|
+
os.link(
|
|
1144
|
+
temporary,
|
|
1145
|
+
name,
|
|
1146
|
+
src_dir_fd=self._shards_fd,
|
|
1147
|
+
dst_dir_fd=self._shards_fd,
|
|
1148
|
+
follow_symlinks=False,
|
|
1149
|
+
)
|
|
1150
|
+
except OSError as error:
|
|
1151
|
+
raise CatalogAuthoringRefused(
|
|
1152
|
+
"AUTHOR_OUTPUT_IO", "authoring.shard", "immutable shard publication failed"
|
|
1153
|
+
) from error
|
|
1154
|
+
finally:
|
|
1155
|
+
with contextlib.suppress(FileNotFoundError):
|
|
1156
|
+
os.unlink(temporary, dir_fd=self._shards_fd)
|
|
1157
|
+
os.fsync(self._shards_fd)
|
|
1158
|
+
if _read_regular_at(self._shards_fd, name, maximum=len(raw)) != raw:
|
|
1159
|
+
raise CatalogAuthoringRefused(
|
|
1160
|
+
"AUTHOR_OUTPUT_READBACK", "authoring.shard", "immutable shard readback differs"
|
|
1161
|
+
)
|
|
1162
|
+
self.bytes_written += len(raw)
|
|
1163
|
+
|
|
1164
|
+
def publish(self, name: str, payload: dict[str, Any], *, maximum: int) -> str:
|
|
1165
|
+
raw = canonical_json_bytes(payload)
|
|
1166
|
+
if len(raw) > maximum:
|
|
1167
|
+
raise CatalogAuthoringRefused(
|
|
1168
|
+
"AUTHOR_SHARD_LIMIT", f"authoring.{name}", "document exceeds its byte bound"
|
|
1169
|
+
)
|
|
1170
|
+
temporary = f".{name}.{os.getpid()}.{secrets.token_hex(8)}.tmp"
|
|
1171
|
+
try:
|
|
1172
|
+
_write_regular_at(self._root_fd, temporary, raw)
|
|
1173
|
+
os.replace(temporary, name, src_dir_fd=self._root_fd, dst_dir_fd=self._root_fd)
|
|
1174
|
+
os.fsync(self._root_fd)
|
|
1175
|
+
except OSError as error:
|
|
1176
|
+
raise CatalogAuthoringRefused(
|
|
1177
|
+
"AUTHOR_OUTPUT_IO", f"authoring.{name}", "document publication failed"
|
|
1178
|
+
) from error
|
|
1179
|
+
finally:
|
|
1180
|
+
with contextlib.suppress(FileNotFoundError):
|
|
1181
|
+
os.unlink(temporary, dir_fd=self._root_fd)
|
|
1182
|
+
if _read_regular_at(self._root_fd, name, maximum=maximum) != raw:
|
|
1183
|
+
raise CatalogAuthoringRefused(
|
|
1184
|
+
"AUTHOR_OUTPUT_READBACK", f"authoring.{name}", "document readback differs"
|
|
1185
|
+
)
|
|
1186
|
+
self.bytes_written += len(raw)
|
|
1187
|
+
return sha256_bytes(raw)
|
|
1188
|
+
|
|
1189
|
+
|
|
1190
|
+
def _read_regular_at(parent_fd: int, name: str, *, maximum: int) -> bytes:
|
|
1191
|
+
try:
|
|
1192
|
+
raw = read_bounded_at(parent_fd, name, maximum=maximum)
|
|
1193
|
+
except BoundedReadFailure as error:
|
|
1194
|
+
raise CatalogAuthoringRefused(
|
|
1195
|
+
"AUTHOR_OUTPUT_PATH",
|
|
1196
|
+
"authoring.output",
|
|
1197
|
+
"output members must be stable bounded single-link regular files",
|
|
1198
|
+
) from error
|
|
1199
|
+
assert raw is not None
|
|
1200
|
+
return raw
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def _write_regular_at(parent_fd: int, name: str, raw: bytes) -> None:
|
|
1204
|
+
flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0)
|
|
1205
|
+
descriptor = os.open(name, flags, 0o600, dir_fd=parent_fd)
|
|
1206
|
+
try:
|
|
1207
|
+
written = 0
|
|
1208
|
+
while written < len(raw):
|
|
1209
|
+
count = os.write(descriptor, raw[written:])
|
|
1210
|
+
if count < 1:
|
|
1211
|
+
raise CatalogAuthoringRefused(
|
|
1212
|
+
"AUTHOR_OUTPUT_IO", "authoring.output", "write made no progress"
|
|
1213
|
+
)
|
|
1214
|
+
written += count
|
|
1215
|
+
os.fsync(descriptor)
|
|
1216
|
+
finally:
|
|
1217
|
+
os.close(descriptor)
|