mostlyright-data 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mostlyright/data_harness/__init__.py +158 -0
- mostlyright/data_harness/acquisition/__init__.py +55 -0
- mostlyright/data_harness/acquisition/http.py +2773 -0
- mostlyright/data_harness/acquisition/parsing.py +809 -0
- mostlyright/data_harness/acquisition/ranges.py +495 -0
- mostlyright/data_harness/acquisition/result_download.py +360 -0
- mostlyright/data_harness/acquisition/retention_admission.py +248 -0
- mostlyright/data_harness/acquisition/sandbox.py +4888 -0
- mostlyright/data_harness/acquisition/url_policy.py +530 -0
- mostlyright/data_harness/agent_runtime.py +2743 -0
- mostlyright/data_harness/assets/logo-ink.svg +31 -0
- mostlyright/data_harness/backends/__init__.py +28 -0
- mostlyright/data_harness/backends/pandas_backend.py +350 -0
- mostlyright/data_harness/backends/polars_backend.py +366 -0
- mostlyright/data_harness/backends/protocol.py +124 -0
- mostlyright/data_harness/backends/reference.py +83 -0
- mostlyright/data_harness/backends/registry.py +55 -0
- mostlyright/data_harness/backends/restrictions.py +126 -0
- mostlyright/data_harness/canonical.py +333 -0
- mostlyright/data_harness/catalog_job.py +625 -0
- mostlyright/data_harness/cli.py +5398 -0
- mostlyright/data_harness/contracts.py +53 -0
- mostlyright/data_harness/coordinator.py +1307 -0
- mostlyright/data_harness/deploy.py +924 -0
- mostlyright/data_harness/deploy_target.py +312 -0
- mostlyright/data_harness/deployment_evidence.py +1067 -0
- mostlyright/data_harness/event_presentation.py +576 -0
- mostlyright/data_harness/events.py +2152 -0
- mostlyright/data_harness/fast_delimited.py +239 -0
- mostlyright/data_harness/fleet.py +237 -0
- mostlyright/data_harness/formats.py +236 -0
- mostlyright/data_harness/governors.py +1163 -0
- mostlyright/data_harness/hosted_bootstrap.py +972 -0
- mostlyright/data_harness/hosted_crawler.py +1115 -0
- mostlyright/data_harness/hosted_crawler_container_smoke.py +351 -0
- mostlyright/data_harness/hosted_crawler_fetch.py +423 -0
- mostlyright/data_harness/hosted_crawler_job.py +1277 -0
- mostlyright/data_harness/hosted_crawler_protocol.py +676 -0
- mostlyright/data_harness/hosted_dataset.py +1500 -0
- mostlyright/data_harness/hosted_deploy.py +3037 -0
- mostlyright/data_harness/hosted_handoff.py +62 -0
- mostlyright/data_harness/hosted_ingestion_contract.py +504 -0
- mostlyright/data_harness/hosted_ingestion_job.py +356 -0
- mostlyright/data_harness/hosted_ingestion_job_smoke.py +40 -0
- mostlyright/data_harness/hosted_session_container_smoke.py +194 -0
- mostlyright/data_harness/hosted_session_worker.py +3554 -0
- mostlyright/data_harness/hosted_session_worker_job_smoke.py +46 -0
- mostlyright/data_harness/hosted_worker.py +6784 -0
- mostlyright/data_harness/ingestion/__init__.py +56 -0
- mostlyright/data_harness/ingestion/contracts.py +461 -0
- mostlyright/data_harness/ingestion/faults.py +42 -0
- mostlyright/data_harness/ingestion/gcs_store.py +1162 -0
- mostlyright/data_harness/ingestion/spool.py +130 -0
- mostlyright/data_harness/ingestion/store.py +885 -0
- mostlyright/data_harness/key_seam.py +434 -0
- mostlyright/data_harness/linux_process_boundary.py +262 -0
- mostlyright/data_harness/local_contracts.py +2880 -0
- mostlyright/data_harness/local_search/__init__.py +5 -0
- mostlyright/data_harness/local_search/build_index.py +1087 -0
- mostlyright/data_harness/local_search/contracts.py +920 -0
- mostlyright/data_harness/local_search/query_trace.py +266 -0
- mostlyright/data_harness/local_search/retrieval.py +700 -0
- mostlyright/data_harness/local_search/sealed.py +474 -0
- mostlyright/data_harness/local_search/service.py +784 -0
- mostlyright/data_harness/nbrender/CONTRACT.md +212 -0
- mostlyright/data_harness/nbrender/__init__.py +12 -0
- mostlyright/data_harness/nbrender/chrome.py +359 -0
- mostlyright/data_harness/nbrender/code_body.py +266 -0
- mostlyright/data_harness/nbrender/document.py +407 -0
- mostlyright/data_harness/nbrender/frame.py +275 -0
- mostlyright/data_harness/nbrender/interactive.py +337 -0
- mostlyright/data_harness/nbrender/markdown_body.py +477 -0
- mostlyright/data_harness/nbrender/mr_components.py +134 -0
- mostlyright/data_harness/nbrender/outputs_data.py +595 -0
- mostlyright/data_harness/nbrender/outputs_rich.py +906 -0
- mostlyright/data_harness/nbrender/outputs_source.py +260 -0
- mostlyright/data_harness/nbrender/outputs_stage.py +176 -0
- mostlyright/data_harness/nbrender/outputs_text.py +400 -0
- mostlyright/data_harness/nbrender/parse.py +394 -0
- mostlyright/data_harness/nbrender/status.py +40 -0
- mostlyright/data_harness/nbrender/tokens.py +1295 -0
- mostlyright/data_harness/notebook.py +1710 -0
- mostlyright/data_harness/offline.py +2049 -0
- mostlyright/data_harness/operation_registry.py +1007 -0
- mostlyright/data_harness/operator_setup.py +239 -0
- mostlyright/data_harness/pipeline.py +6428 -0
- mostlyright/data_harness/plan_graph.py +2026 -0
- mostlyright/data_harness/preparation/__init__.py +104 -0
- mostlyright/data_harness/preparation/contracts.py +1017 -0
- mostlyright/data_harness/preparation/engine.py +221 -0
- mostlyright/data_harness/preparation/errors.py +14 -0
- mostlyright/data_harness/preparation/gates.py +751 -0
- mostlyright/data_harness/preparation/joins.py +574 -0
- mostlyright/data_harness/preparation/profile.py +384 -0
- mostlyright/data_harness/preparation/table.py +217 -0
- mostlyright/data_harness/preparation/transforms.py +568 -0
- mostlyright/data_harness/progress_events.py +534 -0
- mostlyright/data_harness/readers/__init__.py +46 -0
- mostlyright/data_harness/readers/containers.py +963 -0
- mostlyright/data_harness/readers/contracts.py +542 -0
- mostlyright/data_harness/readers/delimited.py +257 -0
- mostlyright/data_harness/readers/grib2/__init__.py +33 -0
- mostlyright/data_harness/readers/grib2/admission.py +722 -0
- mostlyright/data_harness/readers/grib2/decode.py +1009 -0
- mostlyright/data_harness/readers/grib2/geometry.py +1133 -0
- mostlyright/data_harness/readers/grib2/portable_math.py +501 -0
- mostlyright/data_harness/readers/json_tabular.py +485 -0
- mostlyright/data_harness/readers/registry.py +514 -0
- mostlyright/data_harness/readers/samples/README.md +110 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.0.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/cities.csv.gz +0 -0
- mostlyright/data_harness/readers/samples/archive.gzip/1.1.0/cities_one_stream/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.0.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/cities.tar +0 -0
- mostlyright/data_harness/readers/samples/archive.tar/1.1.0/cities_beside_a_directory_entry/expected.json +24 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/cities.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.0.0/cities_beside_a_second_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.1.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/dwd-station.zip +0 -0
- mostlyright/data_harness/readers/samples/archive.zip/1.2.0/dwd_semicolon_station_member/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/cities.csv +3 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/an_ordinary_comma_separated_table/expected.json +23 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/cities.tsv +5 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.0.0/quoted_fields_holding_the_delimiter/expected.json +25 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/expected.json +30 -0
- mostlyright/data_harness/readers/samples/delimited_text/1.1.0/an_hourly_observation_table_served_as_plain_text/observations.csv +5 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/expected.json +44 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.0.0/nested_hourly_observations/stations.json +1 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/expected.json +48 -0
- mostlyright/data_harness/readers/samples/json.tabular/1.1.0/an_observation_stream_served_as_plain_text/observations.ndjson +4 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/cities.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/an_ordinary_table_beside_a_second_sheet/expected.json +24 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.0.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/expected.json +27 -0
- mostlyright/data_harness/readers/samples/spreadsheet.xlsx/1.1.0/shares_the_workbook_had_already_computed/shares.xlsx +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/README.md +20 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/expected.json +55 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/gfs_2m_temperature/gfs-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/1.0.0/hrrr_categorical_rain/hrrr-categorical-rain.grib2 +0 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/expected.json +54 -0
- mostlyright/data_harness/readers/samples/weather.grib2/2.0.0/hrrr_2m_temperature/hrrr-2m-temperature.grib2 +0 -0
- mostlyright/data_harness/readers/samples.py +582 -0
- mostlyright/data_harness/readers/spreadsheet.py +803 -0
- mostlyright/data_harness/readers/tabular.py +510 -0
- mostlyright/data_harness/recipe.py +5321 -0
- mostlyright/data_harness/repair/__init__.py +78 -0
- mostlyright/data_harness/repair/adapters.py +274 -0
- mostlyright/data_harness/repair/contracts.py +872 -0
- mostlyright/data_harness/repair/coordinator.py +1099 -0
- mostlyright/data_harness/repair/errors.py +16 -0
- mostlyright/data_harness/review.py +2533 -0
- mostlyright/data_harness/rowset.py +283 -0
- mostlyright/data_harness/serving.py +1975 -0
- mostlyright/data_harness/serving_edge.py +590 -0
- mostlyright/data_harness/serving_http.py +1031 -0
- mostlyright/data_harness/session_probes.py +759 -0
- mostlyright/data_harness/signing.py +101 -0
- mostlyright/data_harness/source_discovery.py +898 -0
- mostlyright/data_harness/sources/__init__.py +209 -0
- mostlyright/data_harness/sources/_adapter_steps.py +213 -0
- mostlyright/data_harness/sources/adapters.py +1214 -0
- mostlyright/data_harness/sources/cadence.py +1428 -0
- mostlyright/data_harness/sources/cadence_emission.py +453 -0
- mostlyright/data_harness/sources/cadence_history.py +546 -0
- mostlyright/data_harness/sources/catalog/__init__.py +17 -0
- mostlyright/data_harness/sources/catalog/admission.py +477 -0
- mostlyright/data_harness/sources/catalog/authoring.py +1701 -0
- mostlyright/data_harness/sources/catalog/authoring_policy.py +701 -0
- mostlyright/data_harness/sources/catalog/authoring_shards.py +1217 -0
- mostlyright/data_harness/sources/catalog/bounded_io.py +231 -0
- mostlyright/data_harness/sources/catalog/channel.py +523 -0
- mostlyright/data_harness/sources/catalog/channel_client.py +296 -0
- mostlyright/data_harness/sources/catalog/contracts.py +825 -0
- mostlyright/data_harness/sources/catalog/coverage.py +137 -0
- mostlyright/data_harness/sources/catalog/delta.py +1340 -0
- mostlyright/data_harness/sources/catalog/embedding.py +532 -0
- mostlyright/data_harness/sources/catalog/entry_v2.py +1182 -0
- mostlyright/data_harness/sources/catalog/fill.py +3889 -0
- mostlyright/data_harness/sources/catalog/fill_partitions.py +459 -0
- mostlyright/data_harness/sources/catalog/fill_staging.py +1105 -0
- mostlyright/data_harness/sources/catalog/gating.py +374 -0
- mostlyright/data_harness/sources/catalog/generation_receipt.py +1607 -0
- mostlyright/data_harness/sources/catalog/harvest/__init__.py +7 -0
- mostlyright/data_harness/sources/catalog/harvest/ckan.py +384 -0
- mostlyright/data_harness/sources/catalog/harvest/datagov_v4.py +798 -0
- mostlyright/data_harness/sources/catalog/harvest/protocol.py +964 -0
- mostlyright/data_harness/sources/catalog/harvest/sdmx.py +445 -0
- mostlyright/data_harness/sources/catalog/harvest/stac.py +384 -0
- mostlyright/data_harness/sources/catalog/health.py +447 -0
- mostlyright/data_harness/sources/catalog/hosted_catalog.py +105 -0
- mostlyright/data_harness/sources/catalog/identity_history.py +1549 -0
- mostlyright/data_harness/sources/catalog/neural.py +1618 -0
- mostlyright/data_harness/sources/catalog/packed_catalog.py +2345 -0
- mostlyright/data_harness/sources/catalog/packed_retrieval.py +1517 -0
- mostlyright/data_harness/sources/catalog/packed_writer.py +2802 -0
- mostlyright/data_harness/sources/catalog/query_trace.py +1037 -0
- mostlyright/data_harness/sources/catalog/recommend.py +171 -0
- mostlyright/data_harness/sources/catalog/retrieval.py +230 -0
- mostlyright/data_harness/sources/catalog/retrieval_manifest.py +995 -0
- mostlyright/data_harness/sources/catalog/rights_decisions.py +254 -0
- mostlyright/data_harness/sources/catalog/sealed.py +560 -0
- mostlyright/data_harness/sources/catalog/search.py +230 -0
- mostlyright/data_harness/sources/catalog/streaming_delta.py +1097 -0
- mostlyright/data_harness/sources/catalog/update.py +891 -0
- mostlyright/data_harness/sources/collections.py +815 -0
- mostlyright/data_harness/sources/contracts.py +2223 -0
- mostlyright/data_harness/sources/deletion.py +761 -0
- mostlyright/data_harness/sources/fitness.py +162 -0
- mostlyright/data_harness/sources/governance.py +163 -0
- mostlyright/data_harness/sources/hosted.py +173 -0
- mostlyright/data_harness/sources/integration.py +218 -0
- mostlyright/data_harness/sources/range_reader.py +418 -0
- mostlyright/data_harness/sources/registry.py +514 -0
- mostlyright/data_harness/sources/rights_rule.py +59 -0
- mostlyright/data_harness/sources/source_cadence_vectors.v1.json +1 -0
- mostlyright/data_harness/sources/sports.py +521 -0
- mostlyright/data_harness/sources/stream.py +524 -0
- mostlyright/data_harness/sources/stream_connector.py +418 -0
- mostlyright/data_harness/sources/stream_recorder.py +1404 -0
- mostlyright/data_harness/studio_boundary.py +2019 -0
- mostlyright/data_harness/thin/__init__.py +37 -0
- mostlyright/data_harness/thin/acquire.py +1137 -0
- mostlyright/data_harness/thin/acquire_cancel.py +579 -0
- mostlyright/data_harness/thin/approvals.py +617 -0
- mostlyright/data_harness/thin/commands.py +406 -0
- mostlyright/data_harness/thin/download.py +194 -0
- mostlyright/data_harness/thin/narrative.py +589 -0
- mostlyright/data_harness/thin/parity.py +1070 -0
- mostlyright/data_harness/thin/propose.py +2759 -0
- mostlyright/data_harness/thin/research.py +1663 -0
- mostlyright/data_harness/thin/router.py +924 -0
- mostlyright/data_harness/thin/runs.py +519 -0
- mostlyright/data_harness/thin/session.py +281 -0
- mostlyright/data_harness/thin/stream.py +501 -0
- mostlyright/data_harness/thin/transport.py +187 -0
- mostlyright/data_harness/thin/vocabulary.py +368 -0
- mostlyright/data_harness/thin/workers.py +164 -0
- mostlyright/data_harness/ucum/TABLE-PIN.json +40 -0
- mostlyright/data_harness/ucum/ucum-subset.v1.json +632 -0
- mostlyright/data_harness/unit_flow.py +927 -0
- mostlyright/data_harness/units.py +572 -0
- mostlyright/data_harness/ux/__init__.py +9 -0
- mostlyright/data_harness/ux/approve.py +485 -0
- mostlyright/data_harness/ux/author_yaml.py +597 -0
- mostlyright/data_harness/ux/cloud_auth.py +447 -0
- mostlyright/data_harness/ux/commands/__init__.py +260 -0
- mostlyright/data_harness/ux/commands/approve.py +136 -0
- mostlyright/data_harness/ux/commands/auth.py +744 -0
- mostlyright/data_harness/ux/commands/author.py +79 -0
- mostlyright/data_harness/ux/commands/catalog_author.py +403 -0
- mostlyright/data_harness/ux/commands/catalog_fill.py +523 -0
- mostlyright/data_harness/ux/commands/catalog_harvest.py +545 -0
- mostlyright/data_harness/ux/commands/catalog_publish.py +1838 -0
- mostlyright/data_harness/ux/commands/catalog_search.py +71 -0
- mostlyright/data_harness/ux/commands/catalog_update.py +437 -0
- mostlyright/data_harness/ux/commands/deploy.py +134 -0
- mostlyright/data_harness/ux/commands/deploy_dataset.py +98 -0
- mostlyright/data_harness/ux/commands/deploy_plan.py +105 -0
- mostlyright/data_harness/ux/commands/deploy_status.py +104 -0
- mostlyright/data_harness/ux/commands/diff.py +74 -0
- mostlyright/data_harness/ux/commands/index.py +84 -0
- mostlyright/data_harness/ux/commands/inventory.py +47 -0
- mostlyright/data_harness/ux/commands/list_builds.py +143 -0
- mostlyright/data_harness/ux/commands/login.py +63 -0
- mostlyright/data_harness/ux/commands/peek.py +236 -0
- mostlyright/data_harness/ux/commands/plan_check.py +90 -0
- mostlyright/data_harness/ux/commands/preflight.py +97 -0
- mostlyright/data_harness/ux/commands/record.py +107 -0
- mostlyright/data_harness/ux/commands/review_setup.py +47 -0
- mostlyright/data_harness/ux/commands/search.py +440 -0
- mostlyright/data_harness/ux/commands/show.py +61 -0
- mostlyright/data_harness/ux/commands/whoami.py +37 -0
- mostlyright/data_harness/ux/credential_native.py +551 -0
- mostlyright/data_harness/ux/credential_store.py +1055 -0
- mostlyright/data_harness/ux/credentials.py +631 -0
- mostlyright/data_harness/ux/diffing.py +444 -0
- mostlyright/data_harness/ux/headline.py +671 -0
- mostlyright/data_harness/ux/hosted_acquisition.py +974 -0
- mostlyright/data_harness/ux/hosted_run_status.py +619 -0
- mostlyright/data_harness/ux/inventory.py +427 -0
- mostlyright/data_harness/ux/local_review.py +375 -0
- mostlyright/data_harness/ux/login.py +691 -0
- mostlyright/data_harness/ux/path_kind.py +147 -0
- mostlyright/data_harness/ux/peek.py +1000 -0
- mostlyright/data_harness/ux/plain_file.py +178 -0
- mostlyright/data_harness/ux/plan_check.py +311 -0
- mostlyright/data_harness/ux/preflight.py +918 -0
- mostlyright/data_harness/ux/readers.py +1124 -0
- mostlyright/data_harness/ux/remediation.py +2195 -0
- mostlyright/data_harness/ux/render.py +657 -0
- mostlyright/data_harness/ux/workload.py +1077 -0
- mostlyright/data_harness/viewer.py +3713 -0
- mostlyright/data_harness/visual_run/__init__.py +83 -0
- mostlyright/data_harness/visual_run/authoring.py +235 -0
- mostlyright/data_harness/visual_run/contracts.py +673 -0
- mostlyright/data_harness/visual_run/materialize.py +486 -0
- mostlyright/data_harness/visual_run/observations.py +874 -0
- mostlyright/data_harness/visual_run/query.py +259 -0
- mostlyright/data_harness/visual_run/reducer.py +280 -0
- mostlyright/data_harness/visual_run/sdk.py +892 -0
- mostlyright/data_harness/visual_run/store.py +584 -0
- mostlyright/data_harness/visual_run/transport.py +239 -0
- mostlyright/data_harness/watch.py +2999 -0
- mostlyright_data-0.9.0.dist-info/METADATA +607 -0
- mostlyright_data-0.9.0.dist-info/RECORD +314 -0
- mostlyright_data-0.9.0.dist-info/WHEEL +4 -0
- mostlyright_data-0.9.0.dist-info/entry_points.txt +12 -0
|
@@ -0,0 +1,2802 @@
|
|
|
1
|
+
"""The streaming packed publisher, and the only place a packed generation's work is counted.
|
|
2
|
+
|
|
3
|
+
This module turns one classified sweep into one packed generation. Its inputs are exactly the
|
|
4
|
+
delta's classification shards, the successor identity generation
|
|
5
|
+
behind them, and the authored shards the classification was computed from -- and its output is the
|
|
6
|
+
range-local schema in :mod:`packed_catalog`.
|
|
7
|
+
|
|
8
|
+
Four properties are load-bearing.
|
|
9
|
+
|
|
10
|
+
*One pack in, one pack out.* The three publishable classification streams are already ascending by
|
|
11
|
+
provider record, so publication is a merge rather than a join: at most one shard per classification,
|
|
12
|
+
one authored shard, one identity pack, one predecessor range and one range under construction are
|
|
13
|
+
ever resident. Nothing is indexed and nothing is sorted.
|
|
14
|
+
|
|
15
|
+
*Encoding happens exactly once per nonempty fresh range, layer and backend.* A range that contains
|
|
16
|
+
at least one new or changed entry issues one bounded ``encode_many`` invocation per layer per
|
|
17
|
+
backend, carrying exactly that range's fresh texts; an unchanged entry's four layer vectors are read
|
|
18
|
+
back out of the predecessor's packs instead. So the totals are equations rather than observations
|
|
19
|
+
after the fact: ``batch_invocations = nonempty_fresh_range_count * 4 * backend_count`` and
|
|
20
|
+
``encoded_layer_items = fresh_entries * 4 * backend_count``, and this module refuses to install a
|
|
21
|
+
head whose counters do not satisfy them.
|
|
22
|
+
|
|
23
|
+
*The counters are produced, never accepted.* There is no parameter through which a caller can hand
|
|
24
|
+
this function a work total. Every number comes from a seam-minted :class:`BatchEncodeResult` this
|
|
25
|
+
module summed in order, and when a publication spans several invocations the head carries each
|
|
26
|
+
invocation's record beside the ordered sum of them.
|
|
27
|
+
|
|
28
|
+
*A publication is complete, resumable, or nothing.* Members are immutable and content-addressed;
|
|
29
|
+
after each range is durable, a typed-incomplete work journal binds the output root, the ordered
|
|
30
|
+
completed-range member digests and this invocation's counter record. An explicit resume validates
|
|
31
|
+
the journal's own digest, re-reads every already-emitted member and refuses one whose bytes moved,
|
|
32
|
+
and continues from the first missing one without re-encoding a pre-head complete range. Before the
|
|
33
|
+
head is installed, a terminal record captures the exact final journal and packed-receipt inputs; the
|
|
34
|
+
head binds that record's digest. Recovering that mutable terminal state deterministically replays
|
|
35
|
+
the generation before completing its evidence transaction. A crash therefore leaves either
|
|
36
|
+
ordinary range work to resume or a head-bound state that can be verified and completed exactly.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
import contextlib
|
|
42
|
+
import os
|
|
43
|
+
import stat
|
|
44
|
+
import time
|
|
45
|
+
from collections.abc import Callable, Iterator, Mapping, Sequence
|
|
46
|
+
from dataclasses import dataclass, field
|
|
47
|
+
from pathlib import Path
|
|
48
|
+
from typing import Any
|
|
49
|
+
|
|
50
|
+
from mostlyright.data_harness.canonical import (
|
|
51
|
+
CanonicalJSONError,
|
|
52
|
+
canonical_json_bytes,
|
|
53
|
+
canonical_sha256,
|
|
54
|
+
parse_canonical_json,
|
|
55
|
+
sha256_bytes,
|
|
56
|
+
)
|
|
57
|
+
from mostlyright.data_harness.sources.catalog.admission import admit_public_fact_bytes
|
|
58
|
+
from mostlyright.data_harness.sources.catalog.authoring_policy import (
|
|
59
|
+
AUTHORING_DISPOSITIONS,
|
|
60
|
+
AUTHORING_REASON_CODES,
|
|
61
|
+
MAX_LAYER_TEXT_BYTES,
|
|
62
|
+
)
|
|
63
|
+
from mostlyright.data_harness.sources.catalog.authoring_shards import (
|
|
64
|
+
AUTHORING_MANIFEST_SCHEMA,
|
|
65
|
+
AUTHORING_SHARD_SCHEMA,
|
|
66
|
+
MANIFEST_FILENAME,
|
|
67
|
+
MAX_AUTHORING_MANIFEST_BYTES,
|
|
68
|
+
MAX_AUTHORING_SHARD_BYTES,
|
|
69
|
+
SHARDS_DIRNAME,
|
|
70
|
+
shard_descriptor_is_valid,
|
|
71
|
+
)
|
|
72
|
+
from mostlyright.data_harness.sources.catalog.bounded_io import (
|
|
73
|
+
BoundedReadFailure,
|
|
74
|
+
read_bounded_at,
|
|
75
|
+
read_bounded_path,
|
|
76
|
+
)
|
|
77
|
+
from mostlyright.data_harness.sources.catalog.contracts import EMBEDDING_LAYERS
|
|
78
|
+
from mostlyright.data_harness.sources.catalog.coverage import coverage_is_valid
|
|
79
|
+
from mostlyright.data_harness.sources.catalog.embedding import (
|
|
80
|
+
BOUNDED_SCALAR_ADAPTER,
|
|
81
|
+
BatchEncodeResult,
|
|
82
|
+
EmbeddingBackend,
|
|
83
|
+
)
|
|
84
|
+
from mostlyright.data_harness.sources.catalog.entry_v2 import catalog_entry_v2_from_dict
|
|
85
|
+
from mostlyright.data_harness.sources.catalog.identity_history import (
|
|
86
|
+
IdentityHistory,
|
|
87
|
+
IdentityHistoryStore,
|
|
88
|
+
derive_entry_id,
|
|
89
|
+
)
|
|
90
|
+
from mostlyright.data_harness.sources.catalog.packed_catalog import (
|
|
91
|
+
COUNTER_MEMBERS,
|
|
92
|
+
MAX_BACKEND_DIMENSION,
|
|
93
|
+
MAX_FACTS_MEMBER_BYTES,
|
|
94
|
+
MAX_HISTORY_MEMBER_BYTES,
|
|
95
|
+
MAX_PACKED_BACKENDS,
|
|
96
|
+
MAX_PACKED_HEAD_BYTES,
|
|
97
|
+
MAX_POSTING_SEGMENT_BYTES,
|
|
98
|
+
MAX_RANGE_DESCRIPTORS,
|
|
99
|
+
MAX_RANGE_MANIFEST_BYTES,
|
|
100
|
+
MAX_VECTOR_MEMBER_BYTES,
|
|
101
|
+
MEMBER_HEADER_BYTES,
|
|
102
|
+
PACKED_HEAD_FILENAME,
|
|
103
|
+
RANGE_ENTRIES,
|
|
104
|
+
BuiltPackedRange,
|
|
105
|
+
CatalogPackedRefused,
|
|
106
|
+
PackedBackend,
|
|
107
|
+
PackedDirectory,
|
|
108
|
+
PackedRangeDescriptor,
|
|
109
|
+
PackedRangeInput,
|
|
110
|
+
VerifiedPackedGeneration,
|
|
111
|
+
bound_segment_bytes,
|
|
112
|
+
build_packed_head,
|
|
113
|
+
build_packed_range,
|
|
114
|
+
encode_layer_batch,
|
|
115
|
+
member_descriptor_from_dict,
|
|
116
|
+
parse_vector_member,
|
|
117
|
+
range_binding_sha256,
|
|
118
|
+
range_descriptor_from_dict,
|
|
119
|
+
ranges_chain_sha256,
|
|
120
|
+
verify_packed_generation,
|
|
121
|
+
verify_packed_range,
|
|
122
|
+
)
|
|
123
|
+
from mostlyright.data_harness.sources.catalog.streaming_delta import (
|
|
124
|
+
DELTA_IDENTITY_DIRNAME,
|
|
125
|
+
DELTA_MANIFEST_FILENAME,
|
|
126
|
+
DELTA_MANIFEST_SCHEMA,
|
|
127
|
+
DELTA_SHARDS_DIRNAME,
|
|
128
|
+
MAX_DELTA_MANIFEST_BYTES,
|
|
129
|
+
MAX_DELTA_SHARD_BYTES,
|
|
130
|
+
)
|
|
131
|
+
from mostlyright.data_harness.sources.contracts import SourceContractError
|
|
132
|
+
|
|
133
|
+
PACKED_JOURNAL_SCHEMA = "harness-catalog-packed-journal.v1"
|
|
134
|
+
PACKED_JOURNAL_FILENAME = "packed-journal.json"
|
|
135
|
+
PACKED_RECEIPT_SCHEMA = "mr-data-catalog-packed.v1"
|
|
136
|
+
PACKED_TERMINAL_SCHEMA = "harness-catalog-packed-terminal.v1"
|
|
137
|
+
PACKED_TERMINAL_FILENAME = "packed-terminal.json"
|
|
138
|
+
MAX_PACKED_JOURNAL_BYTES = 8 * 1024 * 1024
|
|
139
|
+
MAX_PACKED_TERMINAL_BYTES = 16 * 1024 * 1024
|
|
140
|
+
|
|
141
|
+
#: The three classifications that put an entry in a packed generation. ``absent`` describes a
|
|
142
|
+
#: predecessor record that is gone, and the three non-admitted dispositions never had an entry.
|
|
143
|
+
PUBLISHABLE_CLASSIFICATIONS = ("changed", "new", "unchanged")
|
|
144
|
+
FRESH_CLASSIFICATIONS = ("changed", "new")
|
|
145
|
+
|
|
146
|
+
#: Each member kind's own byte ceiling, so a journal reverification allocates a kind's exact bound
|
|
147
|
+
#: rather than the largest bound any kind has.
|
|
148
|
+
MEMBER_MAXIMUM_BYTES = {
|
|
149
|
+
"facts": MAX_FACTS_MEMBER_BYTES,
|
|
150
|
+
"history": MAX_HISTORY_MEMBER_BYTES,
|
|
151
|
+
"vector": MAX_VECTOR_MEMBER_BYTES,
|
|
152
|
+
"posting": MAX_POSTING_SEGMENT_BYTES,
|
|
153
|
+
"bound": bound_segment_bytes(dimensions=MAX_BACKEND_DIMENSION),
|
|
154
|
+
"range": MAX_RANGE_MANIFEST_BYTES,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
_OPEN_MEMBER = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
|
|
158
|
+
_OPEN_DIRECTORY = _OPEN_MEMBER | getattr(os, "O_DIRECTORY", 0)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _refuse(code: str, path: str, detail: str) -> CatalogPackedRefused:
|
|
162
|
+
return CatalogPackedRefused(code, path, detail)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _is_digest(value: Any) -> bool:
|
|
166
|
+
return (
|
|
167
|
+
isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value)
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
@dataclass(frozen=True)
|
|
172
|
+
class PackedLimits:
|
|
173
|
+
"""Every bound one publication may consume, stated before it consumes any of them."""
|
|
174
|
+
|
|
175
|
+
max_ranges: int = MAX_RANGE_DESCRIPTORS
|
|
176
|
+
max_entries: int = MAX_RANGE_DESCRIPTORS * RANGE_ENTRIES
|
|
177
|
+
max_terms_per_segment: int | None = None
|
|
178
|
+
max_disk_bytes: int = 64 * 1024 * 1024 * 1024
|
|
179
|
+
max_wall_seconds: int = 24 * 60 * 60
|
|
180
|
+
|
|
181
|
+
def __post_init__(self) -> None:
|
|
182
|
+
if type(self.max_ranges) is not int or not 1 <= self.max_ranges <= MAX_RANGE_DESCRIPTORS:
|
|
183
|
+
raise _refuse(
|
|
184
|
+
"PACKED_LIMIT",
|
|
185
|
+
"limits.max_ranges",
|
|
186
|
+
f"a generation holds 1 to {MAX_RANGE_DESCRIPTORS} ranges",
|
|
187
|
+
)
|
|
188
|
+
if type(self.max_entries) is not int or self.max_entries < 1:
|
|
189
|
+
raise _refuse("PACKED_LIMIT", "limits.max_entries", "must be a positive integer")
|
|
190
|
+
if type(self.max_disk_bytes) is not int or self.max_disk_bytes < 1:
|
|
191
|
+
raise _refuse("PACKED_LIMIT", "limits.max_disk_bytes", "must be a positive integer")
|
|
192
|
+
if type(self.max_wall_seconds) is not int or self.max_wall_seconds < 1:
|
|
193
|
+
raise _refuse("PACKED_LIMIT", "limits.max_wall_seconds", "must be a positive integer")
|
|
194
|
+
|
|
195
|
+
def to_dict(self) -> dict[str, Any]:
|
|
196
|
+
return {
|
|
197
|
+
"max_ranges": self.max_ranges,
|
|
198
|
+
"max_entries": self.max_entries,
|
|
199
|
+
"max_terms_per_segment": self.max_terms_per_segment,
|
|
200
|
+
"max_disk_bytes": self.max_disk_bytes,
|
|
201
|
+
"max_wall_seconds": self.max_wall_seconds,
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@dataclass(frozen=True)
|
|
206
|
+
class PackedRangeProgress:
|
|
207
|
+
"""One range that is durable on disk, handed to the caller's observer after it is installed."""
|
|
208
|
+
|
|
209
|
+
range_index: int
|
|
210
|
+
first_key: str
|
|
211
|
+
last_key: str
|
|
212
|
+
entry_count: int
|
|
213
|
+
fresh_entries: int
|
|
214
|
+
reused_entries: int
|
|
215
|
+
range_binding_sha256: str
|
|
216
|
+
posting_backend_coordinate: str
|
|
217
|
+
manifest_sha256: str
|
|
218
|
+
member_sha256s: tuple[tuple[str, str], ...]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@dataclass(frozen=True)
|
|
222
|
+
class PackedPublicationResult:
|
|
223
|
+
"""One publication's exact durable outcome and its non-forgeable work evidence."""
|
|
224
|
+
|
|
225
|
+
status: str
|
|
226
|
+
head_sha256: str | None
|
|
227
|
+
journal_sha256: str
|
|
228
|
+
counters: dict[str, int]
|
|
229
|
+
counts: dict[str, int]
|
|
230
|
+
receipt: dict[str, Any]
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# ------------------------------------------------------------------------------------------
|
|
234
|
+
# The publisher's only door to encoding
|
|
235
|
+
# ------------------------------------------------------------------------------------------
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
@dataclass
|
|
239
|
+
class _Counters:
|
|
240
|
+
"""Ordered sums of seam-minted records. Nothing here is ever set from an argument."""
|
|
241
|
+
|
|
242
|
+
batch_invocations: int = 0
|
|
243
|
+
encoded_layer_items: int = 0
|
|
244
|
+
backend_scalar_invocations: int = 0
|
|
245
|
+
encoded_utf8_bytes: int = 0
|
|
246
|
+
reused_layer_items: int = 0
|
|
247
|
+
|
|
248
|
+
def absorb(self, record: BatchEncodeResult) -> None:
|
|
249
|
+
self.batch_invocations += record.batch_invocations
|
|
250
|
+
self.encoded_layer_items += len(record.vectors)
|
|
251
|
+
self.backend_scalar_invocations += record.scalar_invocations
|
|
252
|
+
self.encoded_utf8_bytes += record.encoded_utf8_bytes
|
|
253
|
+
|
|
254
|
+
def absorb_record(self, record: Mapping[str, int]) -> None:
|
|
255
|
+
for name in COUNTER_MEMBERS:
|
|
256
|
+
setattr(self, name, getattr(self, name) + record[name])
|
|
257
|
+
|
|
258
|
+
def to_dict(self) -> dict[str, int]:
|
|
259
|
+
return {name: getattr(self, name) for name in COUNTER_MEMBERS}
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
class _CountingSeam:
|
|
263
|
+
"""One backend behind one counted door.
|
|
264
|
+
|
|
265
|
+
Every vector this publisher writes for a fresh entry comes through here, and every invocation
|
|
266
|
+
leaves its own seam-minted record behind. There is deliberately no other method: a second
|
|
267
|
+
encoding path would be a second accounting path.
|
|
268
|
+
"""
|
|
269
|
+
|
|
270
|
+
def __init__(self, backend: EmbeddingBackend, counters: _Counters) -> None:
|
|
271
|
+
self._backend = backend
|
|
272
|
+
self._counters = counters
|
|
273
|
+
self.coordinate = backend.descriptor.coordinate
|
|
274
|
+
self.dimensions = backend.descriptor.dimensions
|
|
275
|
+
|
|
276
|
+
def encode_many(self, texts: Sequence[str]) -> tuple[tuple[int, ...], ...]:
|
|
277
|
+
vectors, record = encode_layer_batch(
|
|
278
|
+
self._backend,
|
|
279
|
+
texts,
|
|
280
|
+
entry_count=len(texts),
|
|
281
|
+
dimensions=self.dimensions,
|
|
282
|
+
)
|
|
283
|
+
if record.backend_coordinate != self.coordinate:
|
|
284
|
+
raise _refuse(
|
|
285
|
+
"PACKED_COUNTER_MISMATCH",
|
|
286
|
+
"packed.encode.record",
|
|
287
|
+
"a counter record names another backend than the one that was asked",
|
|
288
|
+
)
|
|
289
|
+
if record.adapter_mode != BOUNDED_SCALAR_ADAPTER:
|
|
290
|
+
raise _refuse(
|
|
291
|
+
"PACKED_COUNTER_ADAPTER",
|
|
292
|
+
"packed.encode.record.adapter_mode",
|
|
293
|
+
f"the only encoding adapter is {BOUNDED_SCALAR_ADAPTER!r}",
|
|
294
|
+
)
|
|
295
|
+
self._counters.absorb(record)
|
|
296
|
+
return vectors
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
# ------------------------------------------------------------------------------------------
|
|
300
|
+
# Bounded inputs
|
|
301
|
+
# ------------------------------------------------------------------------------------------
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _read_bounded(path: Path, *, maximum: int) -> bytes | None:
|
|
305
|
+
try:
|
|
306
|
+
return read_bounded_path(path, maximum=maximum, missing_ok=True)
|
|
307
|
+
except BoundedReadFailure as error:
|
|
308
|
+
raise _refuse(
|
|
309
|
+
"PACKED_INPUT_PATH",
|
|
310
|
+
"packed.input",
|
|
311
|
+
"input members must be stable bounded single-link regular files",
|
|
312
|
+
) from error
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _read_bounded_at(parent_descriptor: int, name: str, *, maximum: int) -> bytes | None:
|
|
316
|
+
try:
|
|
317
|
+
return read_bounded_at(parent_descriptor, name, maximum=maximum, missing_ok=True)
|
|
318
|
+
except BoundedReadFailure as error:
|
|
319
|
+
raise _refuse(
|
|
320
|
+
"PACKED_INPUT_PATH",
|
|
321
|
+
"packed.input",
|
|
322
|
+
"input members must be stable bounded single-link regular files",
|
|
323
|
+
) from error
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
@contextlib.contextmanager
|
|
327
|
+
def _opened_directory_at(parent_descriptor: int, name: str) -> Iterator[int]:
|
|
328
|
+
descriptor = -1
|
|
329
|
+
try:
|
|
330
|
+
descriptor = os.open(name, _OPEN_DIRECTORY, dir_fd=parent_descriptor)
|
|
331
|
+
info = os.fstat(descriptor)
|
|
332
|
+
if not stat.S_ISDIR(info.st_mode):
|
|
333
|
+
raise OSError
|
|
334
|
+
except OSError as error:
|
|
335
|
+
if descriptor >= 0:
|
|
336
|
+
os.close(descriptor)
|
|
337
|
+
raise _refuse(
|
|
338
|
+
"PACKED_INPUT_PATH", "packed.input", "input directory is unreadable"
|
|
339
|
+
) from error
|
|
340
|
+
try:
|
|
341
|
+
yield descriptor
|
|
342
|
+
finally:
|
|
343
|
+
os.close(descriptor)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _read_manifest(
|
|
347
|
+
path: Path,
|
|
348
|
+
*,
|
|
349
|
+
expected_sha256: str,
|
|
350
|
+
maximum: int,
|
|
351
|
+
schema: str,
|
|
352
|
+
label: str,
|
|
353
|
+
parent_descriptor: int | None = None,
|
|
354
|
+
name: str | None = None,
|
|
355
|
+
) -> dict[str, Any]:
|
|
356
|
+
raw = (
|
|
357
|
+
_read_bounded(path, maximum=maximum)
|
|
358
|
+
if parent_descriptor is None
|
|
359
|
+
else _read_bounded_at(parent_descriptor, name or path.name, maximum=maximum)
|
|
360
|
+
)
|
|
361
|
+
if raw is None:
|
|
362
|
+
raise _refuse("PACKED_INPUT_MANIFEST", label, "no readable manifest at this root")
|
|
363
|
+
if sha256_bytes(raw) != expected_sha256:
|
|
364
|
+
raise _refuse(
|
|
365
|
+
"PACKED_INPUT_DIGEST", label, "the retained generation is not the caller's exact one"
|
|
366
|
+
)
|
|
367
|
+
try:
|
|
368
|
+
manifest = parse_canonical_json(raw)
|
|
369
|
+
except CanonicalJSONError as error:
|
|
370
|
+
raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest is not canonical") from error
|
|
371
|
+
if (
|
|
372
|
+
not isinstance(manifest, dict)
|
|
373
|
+
or manifest.get("schema_version") != schema
|
|
374
|
+
or not _is_digest(manifest.get("root_sha256"))
|
|
375
|
+
):
|
|
376
|
+
raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest contract differs")
|
|
377
|
+
body = {key: value for key, value in manifest.items() if key != "root_sha256"}
|
|
378
|
+
if canonical_sha256(body) != manifest["root_sha256"]:
|
|
379
|
+
raise _refuse("PACKED_INPUT_MANIFEST", label, "manifest root digest differs")
|
|
380
|
+
return manifest
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
class _ClassificationStream:
|
|
384
|
+
"""The three publishable classification streams, merged into one ascending record stream."""
|
|
385
|
+
|
|
386
|
+
def __init__(
|
|
387
|
+
self,
|
|
388
|
+
root: Path,
|
|
389
|
+
manifest: Mapping[str, Any],
|
|
390
|
+
*,
|
|
391
|
+
shards_descriptor: int | None = None,
|
|
392
|
+
) -> None:
|
|
393
|
+
self._root = Path(root) / DELTA_SHARDS_DIRNAME
|
|
394
|
+
self._shards_descriptor = shards_descriptor
|
|
395
|
+
shards = manifest["outputs"]["shards"]
|
|
396
|
+
if not isinstance(shards, list):
|
|
397
|
+
raise _refuse(
|
|
398
|
+
"PACKED_INPUT_MANIFEST", "delta.manifest.outputs.shards", "shard list differs"
|
|
399
|
+
)
|
|
400
|
+
self._descriptors: dict[str, list[Mapping[str, Any]]] = {
|
|
401
|
+
name: [] for name in PUBLISHABLE_CLASSIFICATIONS
|
|
402
|
+
}
|
|
403
|
+
for descriptor in shards:
|
|
404
|
+
if (
|
|
405
|
+
not isinstance(descriptor, dict)
|
|
406
|
+
or not _is_digest(descriptor.get("sha256"))
|
|
407
|
+
or type(descriptor.get("bytes")) is not int
|
|
408
|
+
or not 1 <= descriptor["bytes"] <= MAX_DELTA_SHARD_BYTES
|
|
409
|
+
or type(descriptor.get("item_count")) is not int
|
|
410
|
+
or descriptor["item_count"] < 1
|
|
411
|
+
or not isinstance(descriptor.get("classification"), str)
|
|
412
|
+
):
|
|
413
|
+
raise _refuse(
|
|
414
|
+
"PACKED_INPUT_MANIFEST",
|
|
415
|
+
"delta.manifest.outputs.shards[]",
|
|
416
|
+
"shard descriptor is invalid",
|
|
417
|
+
)
|
|
418
|
+
if descriptor["classification"] in self._descriptors:
|
|
419
|
+
self._descriptors[descriptor["classification"]].append(descriptor)
|
|
420
|
+
self._iterators = {name: self._items(name) for name in PUBLISHABLE_CLASSIFICATIONS}
|
|
421
|
+
self._peeked: dict[str, dict[str, Any] | None] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS)
|
|
422
|
+
self._exhausted: dict[str, bool] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS, False)
|
|
423
|
+
self._loaded: dict[str, bool] = dict.fromkeys(PUBLISHABLE_CLASSIFICATIONS, False)
|
|
424
|
+
self.max_resident_packs = 0
|
|
425
|
+
|
|
426
|
+
def _observe_residency(self) -> None:
|
|
427
|
+
resident = sum(1 for value in self._loaded.values() if value)
|
|
428
|
+
self.max_resident_packs = max(self.max_resident_packs, resident)
|
|
429
|
+
|
|
430
|
+
def _items(self, classification: str) -> Iterator[dict[str, Any]]:
|
|
431
|
+
last: bytes | None = None
|
|
432
|
+
for descriptor in self._descriptors[classification]:
|
|
433
|
+
name = f"{descriptor['sha256']}.json"
|
|
434
|
+
raw = (
|
|
435
|
+
_read_bounded(
|
|
436
|
+
self._root / name,
|
|
437
|
+
maximum=MAX_DELTA_SHARD_BYTES,
|
|
438
|
+
)
|
|
439
|
+
if self._shards_descriptor is None
|
|
440
|
+
else _read_bounded_at(
|
|
441
|
+
self._shards_descriptor,
|
|
442
|
+
name,
|
|
443
|
+
maximum=MAX_DELTA_SHARD_BYTES,
|
|
444
|
+
)
|
|
445
|
+
)
|
|
446
|
+
if (
|
|
447
|
+
raw is None
|
|
448
|
+
or len(raw) != descriptor["bytes"]
|
|
449
|
+
or sha256_bytes(raw) != descriptor["sha256"]
|
|
450
|
+
):
|
|
451
|
+
raise _refuse(
|
|
452
|
+
"PACKED_INPUT_SHARD", "delta.shard", "a committed delta shard differs"
|
|
453
|
+
)
|
|
454
|
+
try:
|
|
455
|
+
payload = parse_canonical_json(raw)
|
|
456
|
+
except CanonicalJSONError as error:
|
|
457
|
+
raise _refuse(
|
|
458
|
+
"PACKED_INPUT_SHARD", "delta.shard", "shard is not canonical"
|
|
459
|
+
) from error
|
|
460
|
+
if (
|
|
461
|
+
not isinstance(payload, dict)
|
|
462
|
+
or not isinstance(payload.get("items"), list)
|
|
463
|
+
or len(payload["items"]) != descriptor["item_count"]
|
|
464
|
+
or payload.get("classification") != classification
|
|
465
|
+
):
|
|
466
|
+
raise _refuse("PACKED_INPUT_SHARD", "delta.shard", "shard contract differs")
|
|
467
|
+
self._loaded[classification] = True
|
|
468
|
+
self._observe_residency()
|
|
469
|
+
for item in payload["items"]:
|
|
470
|
+
if (
|
|
471
|
+
not isinstance(item, dict)
|
|
472
|
+
or not isinstance(item.get("provider_record_id"), str)
|
|
473
|
+
or not isinstance(item.get("entry_id"), str)
|
|
474
|
+
or item.get("classification") != classification
|
|
475
|
+
):
|
|
476
|
+
raise _refuse(
|
|
477
|
+
"PACKED_INPUT_SHARD", "delta.shard.items[]", "a delta item contract differs"
|
|
478
|
+
)
|
|
479
|
+
key = item["provider_record_id"].encode("utf-8")
|
|
480
|
+
if last is not None and key <= last:
|
|
481
|
+
raise _refuse(
|
|
482
|
+
"PACKED_INPUT_ORDER",
|
|
483
|
+
"delta.shards",
|
|
484
|
+
"a classification stream must ascend by provider record",
|
|
485
|
+
)
|
|
486
|
+
last = key
|
|
487
|
+
yield item
|
|
488
|
+
self._loaded[classification] = False
|
|
489
|
+
self._observe_residency()
|
|
490
|
+
|
|
491
|
+
def _peek(self, classification: str) -> dict[str, Any] | None:
|
|
492
|
+
if self._peeked[classification] is None and not self._exhausted[classification]:
|
|
493
|
+
self._peeked[classification] = next(self._iterators[classification], None)
|
|
494
|
+
if self._peeked[classification] is None:
|
|
495
|
+
self._exhausted[classification] = True
|
|
496
|
+
return self._peeked[classification]
|
|
497
|
+
|
|
498
|
+
def __iter__(self) -> Iterator[dict[str, Any]]:
|
|
499
|
+
last: bytes | None = None
|
|
500
|
+
while True:
|
|
501
|
+
live = {
|
|
502
|
+
name: item
|
|
503
|
+
for name in PUBLISHABLE_CLASSIFICATIONS
|
|
504
|
+
if (item := self._peek(name)) is not None
|
|
505
|
+
}
|
|
506
|
+
if not live:
|
|
507
|
+
return
|
|
508
|
+
key = min(item["provider_record_id"] for item in live.values())
|
|
509
|
+
owners = [name for name, item in live.items() if item["provider_record_id"] == key]
|
|
510
|
+
if len(owners) != 1:
|
|
511
|
+
raise _refuse(
|
|
512
|
+
"PACKED_INPUT_ORDER",
|
|
513
|
+
"delta.shards",
|
|
514
|
+
"one provider record may hold only one publishable classification",
|
|
515
|
+
)
|
|
516
|
+
encoded = key.encode("utf-8")
|
|
517
|
+
if last is not None and encoded <= last:
|
|
518
|
+
raise _refuse(
|
|
519
|
+
"PACKED_INPUT_ORDER", "delta.shards", "the merged stream must ascend by key"
|
|
520
|
+
)
|
|
521
|
+
last = encoded
|
|
522
|
+
item = self._peeked[owners[0]]
|
|
523
|
+
self._peeked[owners[0]] = None
|
|
524
|
+
assert item is not None
|
|
525
|
+
yield item
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
class _AuthoredStream:
|
|
529
|
+
"""One authored generation as one ascending record stream, one shard resident."""
|
|
530
|
+
|
|
531
|
+
def __init__(self, root: Path, manifest: Mapping[str, Any]) -> None:
|
|
532
|
+
self._root = Path(root)
|
|
533
|
+
self._manifest = manifest
|
|
534
|
+
if not isinstance(manifest.get("shards"), list):
|
|
535
|
+
raise _refuse(
|
|
536
|
+
"PACKED_INPUT_MANIFEST", "authoring.manifest.shards", "shard list differs"
|
|
537
|
+
)
|
|
538
|
+
self._iterator = self._records()
|
|
539
|
+
self._peeked: dict[str, Any] | None = None
|
|
540
|
+
self._exhausted = False
|
|
541
|
+
self.max_resident_shards = 0
|
|
542
|
+
|
|
543
|
+
def _records(self) -> Iterator[dict[str, Any]]:
|
|
544
|
+
last: bytes | None = None
|
|
545
|
+
for descriptor in self._manifest["shards"]:
|
|
546
|
+
if not shard_descriptor_is_valid(descriptor):
|
|
547
|
+
raise _refuse(
|
|
548
|
+
"PACKED_INPUT_MANIFEST",
|
|
549
|
+
"authoring.manifest.shards[]",
|
|
550
|
+
"shard descriptor is invalid",
|
|
551
|
+
)
|
|
552
|
+
raw = _read_bounded(
|
|
553
|
+
self._root / SHARDS_DIRNAME / f"{descriptor['sha256']}.json",
|
|
554
|
+
maximum=MAX_AUTHORING_SHARD_BYTES,
|
|
555
|
+
)
|
|
556
|
+
if (
|
|
557
|
+
raw is None
|
|
558
|
+
or len(raw) != descriptor["bytes"]
|
|
559
|
+
or sha256_bytes(raw) != descriptor["sha256"]
|
|
560
|
+
):
|
|
561
|
+
raise _refuse(
|
|
562
|
+
"PACKED_INPUT_SHARD", "authoring.shard", "a committed authored shard differs"
|
|
563
|
+
)
|
|
564
|
+
try:
|
|
565
|
+
payload = parse_canonical_json(raw)
|
|
566
|
+
except CanonicalJSONError as error:
|
|
567
|
+
raise _refuse(
|
|
568
|
+
"PACKED_INPUT_SHARD", "authoring.shard", "shard is not canonical"
|
|
569
|
+
) from error
|
|
570
|
+
if (
|
|
571
|
+
not isinstance(payload, dict)
|
|
572
|
+
or payload.get("schema_version") != AUTHORING_SHARD_SCHEMA
|
|
573
|
+
or payload.get("policy_sha256") != self._manifest.get("policy_sha256")
|
|
574
|
+
or not isinstance(payload.get("items"), list)
|
|
575
|
+
or len(payload["items"]) != descriptor["item_count"]
|
|
576
|
+
or payload.get("first_provider_record_id") != descriptor["first_provider_record_id"]
|
|
577
|
+
or payload.get("last_provider_record_id") != descriptor["last_provider_record_id"]
|
|
578
|
+
or not payload["items"]
|
|
579
|
+
or not isinstance(payload["items"][0], dict)
|
|
580
|
+
or not isinstance(payload["items"][-1], dict)
|
|
581
|
+
or payload["items"][0].get("provider_record_id")
|
|
582
|
+
!= descriptor["first_provider_record_id"]
|
|
583
|
+
or payload["items"][-1].get("provider_record_id")
|
|
584
|
+
!= descriptor["last_provider_record_id"]
|
|
585
|
+
or sum(
|
|
586
|
+
1
|
|
587
|
+
for item in payload["items"]
|
|
588
|
+
if isinstance(item, dict) and item.get("entry") is not None
|
|
589
|
+
)
|
|
590
|
+
!= descriptor["entry_count"]
|
|
591
|
+
):
|
|
592
|
+
raise _refuse("PACKED_INPUT_SHARD", "authoring.shard", "shard contract differs")
|
|
593
|
+
self.max_resident_shards = max(self.max_resident_shards, 1)
|
|
594
|
+
for item in payload["items"]:
|
|
595
|
+
if not isinstance(item, dict) or not isinstance(
|
|
596
|
+
item.get("provider_record_id"), str
|
|
597
|
+
):
|
|
598
|
+
raise _refuse(
|
|
599
|
+
"PACKED_INPUT_SHARD",
|
|
600
|
+
"authoring.shard.items[]",
|
|
601
|
+
"an authored item contract differs",
|
|
602
|
+
)
|
|
603
|
+
key = item["provider_record_id"].encode("utf-8")
|
|
604
|
+
if last is not None and key <= last:
|
|
605
|
+
raise _refuse(
|
|
606
|
+
"PACKED_INPUT_ORDER",
|
|
607
|
+
"authoring.shards",
|
|
608
|
+
"an authored generation must ascend by provider record",
|
|
609
|
+
)
|
|
610
|
+
last = key
|
|
611
|
+
yield item
|
|
612
|
+
|
|
613
|
+
def advance_to(self, record_id: str) -> dict[str, Any]:
|
|
614
|
+
while True:
|
|
615
|
+
if self._peeked is None and not self._exhausted:
|
|
616
|
+
self._peeked = next(self._iterator, None)
|
|
617
|
+
if self._peeked is None:
|
|
618
|
+
self._exhausted = True
|
|
619
|
+
if self._peeked is None:
|
|
620
|
+
raise _refuse(
|
|
621
|
+
"PACKED_INPUT_AUTHORING",
|
|
622
|
+
"authoring.shards",
|
|
623
|
+
"a classified record has no authored entry in this generation",
|
|
624
|
+
)
|
|
625
|
+
found = self._peeked["provider_record_id"]
|
|
626
|
+
if found == record_id:
|
|
627
|
+
item = self._peeked
|
|
628
|
+
self._peeked = None
|
|
629
|
+
return item
|
|
630
|
+
if found.encode("utf-8") > record_id.encode("utf-8"):
|
|
631
|
+
raise _refuse(
|
|
632
|
+
"PACKED_INPUT_AUTHORING",
|
|
633
|
+
"authoring.shards",
|
|
634
|
+
"a classified record has no authored entry in this generation",
|
|
635
|
+
)
|
|
636
|
+
self._peeked = None
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
class _HistoryStream:
|
|
640
|
+
"""The successor identity generation as one ascending stream, one pack resident."""
|
|
641
|
+
|
|
642
|
+
def __init__(self, store: IdentityHistoryStore) -> None:
|
|
643
|
+
self._store = store
|
|
644
|
+
self._iterator = store.stream()
|
|
645
|
+
self._peeked: IdentityHistory | None = None
|
|
646
|
+
self._exhausted = False
|
|
647
|
+
self.max_resident_packs = 0
|
|
648
|
+
|
|
649
|
+
def advance_to(self, record_id: str) -> IdentityHistory:
|
|
650
|
+
while True:
|
|
651
|
+
if self._peeked is None and not self._exhausted:
|
|
652
|
+
self._peeked = next(self._iterator, None)
|
|
653
|
+
self.max_resident_packs = max(self.max_resident_packs, self._store.resident_packs)
|
|
654
|
+
if self._peeked is None:
|
|
655
|
+
self._exhausted = True
|
|
656
|
+
if self._peeked is None:
|
|
657
|
+
raise _refuse(
|
|
658
|
+
"PACKED_INPUT_IDENTITY",
|
|
659
|
+
"identity.records",
|
|
660
|
+
"a classified record has no identity history in this generation",
|
|
661
|
+
)
|
|
662
|
+
found = self._peeked.identity.provider_record_id
|
|
663
|
+
if found == record_id:
|
|
664
|
+
history = self._peeked
|
|
665
|
+
self._peeked = None
|
|
666
|
+
return history
|
|
667
|
+
if found.encode("utf-8") > record_id.encode("utf-8"):
|
|
668
|
+
raise _refuse(
|
|
669
|
+
"PACKED_INPUT_IDENTITY",
|
|
670
|
+
"identity.records",
|
|
671
|
+
"a classified record has no identity history in this generation",
|
|
672
|
+
)
|
|
673
|
+
self._peeked = None
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
class _PredecessorPacks:
|
|
677
|
+
"""One predecessor packed generation, read one range at a time in ascending key order."""
|
|
678
|
+
|
|
679
|
+
def __init__(self, directory: PackedDirectory, head: Mapping[str, Any]) -> None:
|
|
680
|
+
self._directory = directory
|
|
681
|
+
self._descriptors = [
|
|
682
|
+
range_descriptor_from_dict(value, path=f"predecessor.head.ranges[{index}]")
|
|
683
|
+
for index, value in enumerate(head["ranges"])
|
|
684
|
+
]
|
|
685
|
+
self._backends = {item["coordinate"]: item["dimensions"] for item in head["backends"]}
|
|
686
|
+
self._position = 0
|
|
687
|
+
self._resident: int | None = None
|
|
688
|
+
self._ordinals: dict[str, int] = {}
|
|
689
|
+
self._vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
|
|
690
|
+
self.max_resident_ranges = 0
|
|
691
|
+
|
|
692
|
+
def require_backends(self, coordinates: Sequence[str]) -> None:
|
|
693
|
+
missing = [name for name in coordinates if name not in self._backends]
|
|
694
|
+
if missing:
|
|
695
|
+
raise _refuse(
|
|
696
|
+
"PACKED_REUSE_BACKEND",
|
|
697
|
+
"predecessor.backends",
|
|
698
|
+
"the predecessor generation has no packs for a backend this publication reuses",
|
|
699
|
+
)
|
|
700
|
+
|
|
701
|
+
def lookup(self, key: str) -> dict[str, dict[str, tuple[int, ...]]]:
|
|
702
|
+
encoded = key.encode("utf-8")
|
|
703
|
+
while self._position < len(self._descriptors):
|
|
704
|
+
if encoded <= self._descriptors[self._position].last_key.encode("utf-8"):
|
|
705
|
+
break
|
|
706
|
+
self._position += 1
|
|
707
|
+
if self._position >= len(self._descriptors):
|
|
708
|
+
raise _refuse(
|
|
709
|
+
"PACKED_REUSE_MISSING",
|
|
710
|
+
"predecessor.ranges",
|
|
711
|
+
"an unchanged entry has no predecessor pack to reuse",
|
|
712
|
+
)
|
|
713
|
+
if self._resident != self._position:
|
|
714
|
+
self._load(self._position)
|
|
715
|
+
ordinal = self._ordinals.get(key)
|
|
716
|
+
if ordinal is None:
|
|
717
|
+
raise _refuse(
|
|
718
|
+
"PACKED_REUSE_MISSING",
|
|
719
|
+
"predecessor.ranges",
|
|
720
|
+
"an unchanged entry has no predecessor pack to reuse",
|
|
721
|
+
)
|
|
722
|
+
return {
|
|
723
|
+
coordinate: {layer: vectors[ordinal] for layer, vectors in layers.items()}
|
|
724
|
+
for coordinate, layers in self._vectors.items()
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
def _load(self, position: int) -> None:
|
|
728
|
+
descriptor = self._descriptors[position]
|
|
729
|
+
raw = self._directory.read_member(
|
|
730
|
+
descriptor.manifest_sha256, kind="range", maximum=MAX_RANGE_MANIFEST_BYTES
|
|
731
|
+
)
|
|
732
|
+
if raw is None or sha256_bytes(raw) != descriptor.manifest_sha256:
|
|
733
|
+
raise _refuse(
|
|
734
|
+
"PACKED_REUSE_MEMBER",
|
|
735
|
+
"predecessor.range.manifest",
|
|
736
|
+
"a predecessor range manifest is not the manifest its head binds",
|
|
737
|
+
)
|
|
738
|
+
manifest = parse_canonical_json(raw)
|
|
739
|
+
members = [
|
|
740
|
+
member_descriptor_from_dict(value, path="predecessor.range.members[]")
|
|
741
|
+
for value in manifest["members"]
|
|
742
|
+
]
|
|
743
|
+
facts = self._member(_only(members, "facts"), maximum=MAX_FACTS_MEMBER_BYTES)
|
|
744
|
+
payload = parse_canonical_json(facts)
|
|
745
|
+
self._ordinals = {item["key"]: ordinal for ordinal, item in enumerate(payload["entries"])}
|
|
746
|
+
vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
|
|
747
|
+
for member in members:
|
|
748
|
+
if member.kind != "vector":
|
|
749
|
+
continue
|
|
750
|
+
raw_member = self._member(member, maximum=MAX_VECTOR_MEMBER_BYTES)
|
|
751
|
+
parsed = parse_vector_member(raw_member, descriptor=member)
|
|
752
|
+
vectors.setdefault(member.backend_coordinate or "", {})[parsed.layer] = parsed.vectors
|
|
753
|
+
self._vectors = vectors
|
|
754
|
+
self._resident = position
|
|
755
|
+
self.max_resident_ranges = max(self.max_resident_ranges, 1)
|
|
756
|
+
|
|
757
|
+
def _member(self, descriptor: Any, *, maximum: int) -> bytes:
|
|
758
|
+
raw = self._directory.read_member(descriptor.sha256, kind=descriptor.kind, maximum=maximum)
|
|
759
|
+
if raw is None or len(raw) != descriptor.bytes or sha256_bytes(raw) != descriptor.sha256:
|
|
760
|
+
raise _refuse(
|
|
761
|
+
"PACKED_REUSE_MEMBER",
|
|
762
|
+
"predecessor.range.members[]",
|
|
763
|
+
"a predecessor member is not the member its descriptor binds",
|
|
764
|
+
)
|
|
765
|
+
return raw
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
def _only(descriptors: Sequence[Any], kind: str) -> Any:
|
|
769
|
+
found = [descriptor for descriptor in descriptors if descriptor.kind == kind]
|
|
770
|
+
if len(found) != 1:
|
|
771
|
+
raise _refuse(
|
|
772
|
+
"PACKED_REUSE_MEMBER",
|
|
773
|
+
"predecessor.range.members[]",
|
|
774
|
+
f"a predecessor range carries exactly one {kind} member",
|
|
775
|
+
)
|
|
776
|
+
return found[0]
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
# ------------------------------------------------------------------------------------------
|
|
780
|
+
# The work journal
|
|
781
|
+
# ------------------------------------------------------------------------------------------
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
#: Exactly what one completed-range record carries. A resume reads nothing else from it, and a
|
|
785
|
+
#: record carrying anything else is a record this publisher did not write.
|
|
786
|
+
_COMPLETED_RANGE_MEMBERS = frozenset(
|
|
787
|
+
{
|
|
788
|
+
"range_index",
|
|
789
|
+
"first_key",
|
|
790
|
+
"last_key",
|
|
791
|
+
"entry_count",
|
|
792
|
+
"facts_sha256",
|
|
793
|
+
"range_binding_sha256",
|
|
794
|
+
"range_manifest_sha256",
|
|
795
|
+
"range_manifest_bytes",
|
|
796
|
+
"member_sha256s",
|
|
797
|
+
"vector_payload_sha256s",
|
|
798
|
+
}
|
|
799
|
+
)
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
@dataclass
|
|
803
|
+
class _Journal:
|
|
804
|
+
"""The typed work journal an interruption leaves behind and a resume validates."""
|
|
805
|
+
|
|
806
|
+
output_root: str
|
|
807
|
+
generation: dict[str, Any]
|
|
808
|
+
completed_ranges: list[dict[str, Any]] = field(default_factory=list)
|
|
809
|
+
invocations: list[dict[str, Any]] = field(default_factory=list)
|
|
810
|
+
|
|
811
|
+
def body(self, *, status: str, head_sha256: str | None) -> dict[str, Any]:
|
|
812
|
+
return {
|
|
813
|
+
"schema_version": PACKED_JOURNAL_SCHEMA,
|
|
814
|
+
"status": status,
|
|
815
|
+
"output_root": self.output_root,
|
|
816
|
+
"generation": self.generation,
|
|
817
|
+
"completed_ranges": list(self.completed_ranges),
|
|
818
|
+
"invocations": list(self.invocations),
|
|
819
|
+
"head_sha256": head_sha256,
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
def document(self, *, status: str, head_sha256: str | None) -> dict[str, Any]:
|
|
823
|
+
body = self.body(status=status, head_sha256=head_sha256)
|
|
824
|
+
return {**body, "root_sha256": canonical_sha256(body)}
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
def _read_journal(directory: PackedDirectory) -> dict[str, Any] | None:
|
|
828
|
+
raw = directory.read(PACKED_JOURNAL_FILENAME, maximum=MAX_PACKED_JOURNAL_BYTES)
|
|
829
|
+
if raw is None:
|
|
830
|
+
return None
|
|
831
|
+
try:
|
|
832
|
+
journal = parse_canonical_json(raw)
|
|
833
|
+
except CanonicalJSONError as error:
|
|
834
|
+
raise _refuse(
|
|
835
|
+
"PACKED_JOURNAL_DIGEST", "packed.journal", "the work journal is not canonical"
|
|
836
|
+
) from error
|
|
837
|
+
if (
|
|
838
|
+
not isinstance(journal, dict)
|
|
839
|
+
or journal.get("schema_version") != PACKED_JOURNAL_SCHEMA
|
|
840
|
+
or not _is_digest(journal.get("root_sha256"))
|
|
841
|
+
or not isinstance(journal.get("completed_ranges"), list)
|
|
842
|
+
or not isinstance(journal.get("invocations"), list)
|
|
843
|
+
or not isinstance(journal.get("generation"), dict)
|
|
844
|
+
):
|
|
845
|
+
raise _refuse(
|
|
846
|
+
"PACKED_JOURNAL_DIGEST", "packed.journal", "the work journal contract differs"
|
|
847
|
+
)
|
|
848
|
+
body = {key: value for key, value in journal.items() if key != "root_sha256"}
|
|
849
|
+
if canonical_sha256(body) != journal["root_sha256"]:
|
|
850
|
+
raise _refuse(
|
|
851
|
+
"PACKED_JOURNAL_DIGEST",
|
|
852
|
+
"packed.journal.root_sha256",
|
|
853
|
+
"the work journal does not reproduce its own digest",
|
|
854
|
+
)
|
|
855
|
+
# The journal's digest is its own, not a secret's, so a rewriter can mint a consistent one.
|
|
856
|
+
# Every record it carries is therefore re-derived rather than believed: the counter records
|
|
857
|
+
# below only survive because the equations at the end of a publication must still hold.
|
|
858
|
+
for position, record in enumerate(journal["invocations"]):
|
|
859
|
+
if not isinstance(record, dict) or set(record) != set(COUNTER_MEMBERS):
|
|
860
|
+
raise _refuse(
|
|
861
|
+
"PACKED_JOURNAL_COUNTER",
|
|
862
|
+
f"packed.journal.invocations[{position}]",
|
|
863
|
+
f"one invocation record carries exactly {list(COUNTER_MEMBERS)}",
|
|
864
|
+
)
|
|
865
|
+
for name in COUNTER_MEMBERS:
|
|
866
|
+
if type(record[name]) is not int or record[name] < 0:
|
|
867
|
+
raise _refuse(
|
|
868
|
+
"PACKED_JOURNAL_COUNTER",
|
|
869
|
+
f"packed.journal.invocations[{position}].{name}",
|
|
870
|
+
"a work counter is a non-negative integer",
|
|
871
|
+
)
|
|
872
|
+
return journal
|
|
873
|
+
|
|
874
|
+
|
|
875
|
+
def _resume_from(
|
|
876
|
+
directory: PackedDirectory,
|
|
877
|
+
journal: dict[str, Any],
|
|
878
|
+
*,
|
|
879
|
+
output_root: Path,
|
|
880
|
+
generation: Mapping[str, Any],
|
|
881
|
+
) -> _Journal:
|
|
882
|
+
if journal["status"] != "incomplete" or journal["head_sha256"] is not None:
|
|
883
|
+
raise _refuse(
|
|
884
|
+
"PACKED_JOURNAL_STATUS",
|
|
885
|
+
"packed.journal.status",
|
|
886
|
+
"only a typed-incomplete journal may be resumed",
|
|
887
|
+
)
|
|
888
|
+
if journal["output_root"] != os.path.abspath(os.fspath(output_root)):
|
|
889
|
+
raise _refuse(
|
|
890
|
+
"PACKED_JOURNAL_ROOT",
|
|
891
|
+
"packed.journal.output_root",
|
|
892
|
+
"the journal was written for another output root",
|
|
893
|
+
)
|
|
894
|
+
if journal["generation"] != dict(generation):
|
|
895
|
+
raise _refuse(
|
|
896
|
+
"PACKED_JOURNAL_GENERATION",
|
|
897
|
+
"packed.journal.generation",
|
|
898
|
+
"the journal was written for another generation coordinate",
|
|
899
|
+
)
|
|
900
|
+
for position, completed in enumerate(journal["completed_ranges"]):
|
|
901
|
+
if (
|
|
902
|
+
not isinstance(completed, dict)
|
|
903
|
+
or set(completed) != _COMPLETED_RANGE_MEMBERS
|
|
904
|
+
or completed["range_index"] != position
|
|
905
|
+
or not _is_digest(completed["range_manifest_sha256"])
|
|
906
|
+
or not _is_digest(completed["facts_sha256"])
|
|
907
|
+
or not _is_digest(completed["range_binding_sha256"])
|
|
908
|
+
or not isinstance(completed["member_sha256s"], list)
|
|
909
|
+
or not isinstance(completed["vector_payload_sha256s"], list)
|
|
910
|
+
or type(completed["entry_count"]) is not int
|
|
911
|
+
):
|
|
912
|
+
raise _refuse(
|
|
913
|
+
"PACKED_JOURNAL_RANGE",
|
|
914
|
+
f"packed.journal.completed_ranges[{position}]",
|
|
915
|
+
"a completed range record contract differs",
|
|
916
|
+
)
|
|
917
|
+
payloads: list[str] = []
|
|
918
|
+
for entry in [
|
|
919
|
+
["range", completed["range_manifest_sha256"]],
|
|
920
|
+
*completed["member_sha256s"],
|
|
921
|
+
]:
|
|
922
|
+
if not isinstance(entry, list) or len(entry) != 2:
|
|
923
|
+
raise _refuse(
|
|
924
|
+
"PACKED_JOURNAL_RANGE",
|
|
925
|
+
f"packed.journal.completed_ranges[{position}].member_sha256s[]",
|
|
926
|
+
"a completed member record is exactly one kind and one digest",
|
|
927
|
+
)
|
|
928
|
+
kind, sha256 = entry
|
|
929
|
+
if kind not in MEMBER_MAXIMUM_BYTES or not _is_digest(sha256):
|
|
930
|
+
raise _refuse(
|
|
931
|
+
"PACKED_JOURNAL_RANGE",
|
|
932
|
+
f"packed.journal.completed_ranges[{position}].member_sha256s[]",
|
|
933
|
+
"a completed member record names a known kind and an exact digest",
|
|
934
|
+
)
|
|
935
|
+
raw = directory.read_member(sha256, kind=kind, maximum=MEMBER_MAXIMUM_BYTES[kind])
|
|
936
|
+
if raw is None or sha256_bytes(raw) != sha256:
|
|
937
|
+
raise _refuse(
|
|
938
|
+
"PACKED_JOURNAL_MEMBER",
|
|
939
|
+
f"packed.journal.completed_ranges[{position}]",
|
|
940
|
+
"an already-emitted immutable member does not read back exactly",
|
|
941
|
+
)
|
|
942
|
+
if kind == "vector":
|
|
943
|
+
payloads.append(sha256_bytes(raw[MEMBER_HEADER_BYTES:]))
|
|
944
|
+
# The receipt states these, so they are recomputed from the member bytes this loop just
|
|
945
|
+
# read rather than carried over from a journal that a rewriter could have restated.
|
|
946
|
+
if payloads != list(completed["vector_payload_sha256s"]):
|
|
947
|
+
raise _refuse(
|
|
948
|
+
"PACKED_JOURNAL_MEMBER",
|
|
949
|
+
f"packed.journal.completed_ranges[{position}].vector_payload_sha256s",
|
|
950
|
+
"a completed range's vector payload digests are not the digests of its own packs",
|
|
951
|
+
)
|
|
952
|
+
return _Journal(
|
|
953
|
+
output_root=journal["output_root"],
|
|
954
|
+
generation=dict(journal["generation"]),
|
|
955
|
+
completed_ranges=list(journal["completed_ranges"]),
|
|
956
|
+
invocations=list(journal["invocations"]),
|
|
957
|
+
)
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
def _verify_recovery_ranges(
|
|
961
|
+
directory: PackedDirectory,
|
|
962
|
+
installed: VerifiedPackedGeneration,
|
|
963
|
+
*,
|
|
964
|
+
completed_ranges: Sequence[Mapping[str, Any]],
|
|
965
|
+
delta_root: Path,
|
|
966
|
+
delta_manifest: Mapping[str, Any],
|
|
967
|
+
delta_shards_descriptor: int | None,
|
|
968
|
+
identity_descriptor: int | None,
|
|
969
|
+
authoring_root: Path,
|
|
970
|
+
authoring_manifest: Mapping[str, Any],
|
|
971
|
+
identity_manifest_sha256: str,
|
|
972
|
+
packed_backends: Sequence[PackedBackend],
|
|
973
|
+
seam_backends: Sequence[EmbeddingBackend],
|
|
974
|
+
posting_backend: PackedBackend,
|
|
975
|
+
predecessor: _PredecessorPacks | None,
|
|
976
|
+
limits: PackedLimits,
|
|
977
|
+
deadline: float,
|
|
978
|
+
) -> tuple[dict[str, int], dict[str, int]]:
|
|
979
|
+
"""Replay the exact retained generation and bind every installed range to those inputs."""
|
|
980
|
+
|
|
981
|
+
classification = _ClassificationStream(
|
|
982
|
+
delta_root, delta_manifest, shards_descriptor=delta_shards_descriptor
|
|
983
|
+
)
|
|
984
|
+
authored = _AuthoredStream(authoring_root, authoring_manifest)
|
|
985
|
+
identity = _HistoryStream(
|
|
986
|
+
IdentityHistoryStore.open(
|
|
987
|
+
delta_root / DELTA_IDENTITY_DIRNAME,
|
|
988
|
+
expected_manifest_sha256=identity_manifest_sha256,
|
|
989
|
+
root_descriptor=identity_descriptor,
|
|
990
|
+
)
|
|
991
|
+
)
|
|
992
|
+
counters = _Counters()
|
|
993
|
+
seams = {
|
|
994
|
+
backend.descriptor.coordinate: _CountingSeam(backend, counters) for backend in seam_backends
|
|
995
|
+
}
|
|
996
|
+
counts = {
|
|
997
|
+
"ranges": 0,
|
|
998
|
+
"entries": 0,
|
|
999
|
+
"fresh_ranges": 0,
|
|
1000
|
+
"new": 0,
|
|
1001
|
+
"changed": 0,
|
|
1002
|
+
"unchanged": 0,
|
|
1003
|
+
"members": 0,
|
|
1004
|
+
}
|
|
1005
|
+
for range_index, entries in enumerate(_ranges(classification, authored, identity)):
|
|
1006
|
+
if range_index >= limits.max_ranges:
|
|
1007
|
+
raise _refuse(
|
|
1008
|
+
"PACKED_RANGE_LIMIT",
|
|
1009
|
+
"packed.terminal",
|
|
1010
|
+
"terminal recovery exceeds the declared range bound",
|
|
1011
|
+
)
|
|
1012
|
+
if time.monotonic() >= deadline:
|
|
1013
|
+
raise _refuse(
|
|
1014
|
+
"PACKED_WALL_LIMIT",
|
|
1015
|
+
"packed.terminal",
|
|
1016
|
+
"terminal recovery exceeded its declared wall bound",
|
|
1017
|
+
)
|
|
1018
|
+
if range_index >= len(completed_ranges) or range_index >= len(installed.head["ranges"]):
|
|
1019
|
+
raise _refuse(
|
|
1020
|
+
"PACKED_TERMINAL_RANGE",
|
|
1021
|
+
"packed.terminal.journal.completed_ranges",
|
|
1022
|
+
"the terminal record omits a range produced by the retained inputs",
|
|
1023
|
+
)
|
|
1024
|
+
range_input = PackedRangeInput(
|
|
1025
|
+
range_index=range_index, entries=tuple(item.to_entry() for item in entries)
|
|
1026
|
+
)
|
|
1027
|
+
completed = completed_ranges[range_index]
|
|
1028
|
+
_verify_completed_range(directory, completed, range_input)
|
|
1029
|
+
|
|
1030
|
+
fresh = tuple(
|
|
1031
|
+
position
|
|
1032
|
+
for position, item in enumerate(entries)
|
|
1033
|
+
if item.classification in FRESH_CLASSIFICATIONS
|
|
1034
|
+
)
|
|
1035
|
+
fresh_positions = set(fresh)
|
|
1036
|
+
reused = tuple(
|
|
1037
|
+
position for position in range(len(entries)) if position not in fresh_positions
|
|
1038
|
+
)
|
|
1039
|
+
if reused and predecessor is None:
|
|
1040
|
+
raise _refuse(
|
|
1041
|
+
"PACKED_PREDECESSOR_REQUIRED",
|
|
1042
|
+
"predecessor",
|
|
1043
|
+
"an unchanged entry can only be verified against the packs it reuses",
|
|
1044
|
+
)
|
|
1045
|
+
recovered: dict[int, dict[str, dict[str, tuple[int, ...]]]] = {}
|
|
1046
|
+
for position in reused:
|
|
1047
|
+
assert predecessor is not None
|
|
1048
|
+
recovered[position] = predecessor.lookup(entries[position].key)
|
|
1049
|
+
|
|
1050
|
+
# All terminal files are locally rewriteable. For a fresh entry no retained artifact other
|
|
1051
|
+
# than a deterministic backend replay can prove that a vector came from its exact layer
|
|
1052
|
+
# text, so recovery performs that replay before it mutates the journal.
|
|
1053
|
+
vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
|
|
1054
|
+
for backend in packed_backends:
|
|
1055
|
+
seam = seams[backend.coordinate]
|
|
1056
|
+
layers: dict[str, tuple[tuple[int, ...], ...]] = {}
|
|
1057
|
+
for layer in EMBEDDING_LAYERS:
|
|
1058
|
+
column: list[tuple[int, ...] | None] = [None] * len(entries)
|
|
1059
|
+
if fresh:
|
|
1060
|
+
encoded = seam.encode_many(
|
|
1061
|
+
tuple(entries[position].layer_texts[layer] for position in fresh)
|
|
1062
|
+
)
|
|
1063
|
+
for position, vector in zip(fresh, encoded, strict=True):
|
|
1064
|
+
column[position] = vector
|
|
1065
|
+
for position in reused:
|
|
1066
|
+
vector = recovered[position].get(backend.coordinate, {}).get(layer)
|
|
1067
|
+
if vector is None or len(vector) != backend.dimensions:
|
|
1068
|
+
raise _refuse(
|
|
1069
|
+
"PACKED_REUSE_MISSING",
|
|
1070
|
+
"predecessor.ranges",
|
|
1071
|
+
"an unchanged entry has no reusable layer vector for this backend",
|
|
1072
|
+
)
|
|
1073
|
+
column[position] = vector
|
|
1074
|
+
layers[layer] = tuple(value for value in column if value is not None)
|
|
1075
|
+
if len(layers[layer]) != len(entries): # pragma: no cover - filled above
|
|
1076
|
+
raise _refuse(
|
|
1077
|
+
"PACKED_RANGE_ENTRY",
|
|
1078
|
+
"packed.range.vectors",
|
|
1079
|
+
"a layer member does not cover every entry in its range",
|
|
1080
|
+
)
|
|
1081
|
+
vectors[backend.coordinate] = layers
|
|
1082
|
+
|
|
1083
|
+
built = build_packed_range(
|
|
1084
|
+
range_input,
|
|
1085
|
+
vectors,
|
|
1086
|
+
backends=packed_backends,
|
|
1087
|
+
posting_backend_coordinate=posting_backend.coordinate,
|
|
1088
|
+
max_terms_per_segment=limits.max_terms_per_segment,
|
|
1089
|
+
)
|
|
1090
|
+
expected_completed = _completed_range_record(built)
|
|
1091
|
+
if dict(completed) != expected_completed or installed.head["ranges"][range_index] != (
|
|
1092
|
+
built.descriptor.to_dict()
|
|
1093
|
+
):
|
|
1094
|
+
raise _refuse(
|
|
1095
|
+
"PACKED_TERMINAL_RANGE",
|
|
1096
|
+
f"packed.terminal.journal.completed_ranges[{range_index}]",
|
|
1097
|
+
"an installed range is not the exact range reproduced from the retained inputs",
|
|
1098
|
+
)
|
|
1099
|
+
counts["ranges"] += 1
|
|
1100
|
+
counts["entries"] += len(entries)
|
|
1101
|
+
counts["fresh_ranges"] += 1 if fresh else 0
|
|
1102
|
+
counts["members"] += len(built.members)
|
|
1103
|
+
counters.reused_layer_items += len(reused) * len(EMBEDDING_LAYERS) * len(packed_backends)
|
|
1104
|
+
for item in entries:
|
|
1105
|
+
counts[item.classification] += 1
|
|
1106
|
+
|
|
1107
|
+
if len(completed_ranges) != counts["ranges"] or counts["entries"] > limits.max_entries:
|
|
1108
|
+
raise _refuse(
|
|
1109
|
+
"PACKED_TERMINAL_RANGE",
|
|
1110
|
+
"packed.terminal.journal.completed_ranges",
|
|
1111
|
+
"the terminal range list differs from the retained generation",
|
|
1112
|
+
)
|
|
1113
|
+
if (
|
|
1114
|
+
counts["ranges"] != installed.range_count
|
|
1115
|
+
or counts["entries"] != installed.entry_count
|
|
1116
|
+
or counts["members"] != installed.member_count
|
|
1117
|
+
or counts["fresh_ranges"] != installed.fresh_range_count
|
|
1118
|
+
or {name: counts[name] for name in PUBLISHABLE_CLASSIFICATIONS}
|
|
1119
|
+
!= dict(installed.classification_counts)
|
|
1120
|
+
):
|
|
1121
|
+
raise _refuse(
|
|
1122
|
+
"PACKED_TERMINAL_COUNTER",
|
|
1123
|
+
"packed.terminal.receipt.counts",
|
|
1124
|
+
"terminal recovery inputs do not reproduce the installed generation",
|
|
1125
|
+
)
|
|
1126
|
+
_verify_counter_equations(counters, counts, backend_count=len(packed_backends))
|
|
1127
|
+
return counts, counters.to_dict()
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
def _canonical_residency(
|
|
1131
|
+
counts: Mapping[str, int], generation: Mapping[str, Any]
|
|
1132
|
+
) -> dict[str, int]:
|
|
1133
|
+
"""State deterministic streaming bounds, not unauthenticatable process observations."""
|
|
1134
|
+
|
|
1135
|
+
return {
|
|
1136
|
+
"max_resident_classification_packs": sum(
|
|
1137
|
+
1 for name in PUBLISHABLE_CLASSIFICATIONS if counts[name] > 0
|
|
1138
|
+
),
|
|
1139
|
+
"max_resident_authoring_shards": 1,
|
|
1140
|
+
"max_resident_identity_packs": 1,
|
|
1141
|
+
"max_resident_output_ranges": 1,
|
|
1142
|
+
"max_resident_predecessor_ranges": (
|
|
1143
|
+
0 if generation["predecessor_head_sha256"] is None else 1
|
|
1144
|
+
),
|
|
1145
|
+
}
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
def _verified_receipt_outputs(installed: VerifiedPackedGeneration) -> dict[str, Any]:
|
|
1149
|
+
"""Derive every receipt output from the generation's independently verified bytes."""
|
|
1150
|
+
|
|
1151
|
+
return {
|
|
1152
|
+
"range_count": installed.range_count,
|
|
1153
|
+
"entry_count": installed.entry_count,
|
|
1154
|
+
"member_count": installed.member_count,
|
|
1155
|
+
"posting_term_count": installed.posting_term_count,
|
|
1156
|
+
"ranges": list(installed.head["ranges"]),
|
|
1157
|
+
"ranges_chain_sha256": installed.head["ranges_chain_sha256"],
|
|
1158
|
+
"vector_member_sha256s": list(installed.vector_payload_sha256s),
|
|
1159
|
+
"posting_member_sha256s": list(installed.posting_member_sha256s),
|
|
1160
|
+
"bound_member_sha256s": list(installed.bound_member_sha256s),
|
|
1161
|
+
}
|
|
1162
|
+
|
|
1163
|
+
|
|
1164
|
+
def _verify_recovery_evidence(
|
|
1165
|
+
installed: VerifiedPackedGeneration,
|
|
1166
|
+
*,
|
|
1167
|
+
terminal: Mapping[str, Any],
|
|
1168
|
+
journal_template: Mapping[str, Any],
|
|
1169
|
+
journal: Mapping[str, Any],
|
|
1170
|
+
expected_counts: Mapping[str, int],
|
|
1171
|
+
expected_counters: Mapping[str, int],
|
|
1172
|
+
delta_manifest: Mapping[str, Any],
|
|
1173
|
+
packed_backends: Sequence[PackedBackend],
|
|
1174
|
+
posting_backend: PackedBackend,
|
|
1175
|
+
limits: PackedLimits,
|
|
1176
|
+
) -> None:
|
|
1177
|
+
"""Rebuild every terminal receipt claim from caller inputs and verified member bytes."""
|
|
1178
|
+
|
|
1179
|
+
receipt = terminal.get("receipt")
|
|
1180
|
+
if not isinstance(receipt, Mapping):
|
|
1181
|
+
raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "receipt differs")
|
|
1182
|
+
invocations = receipt.get("invocations")
|
|
1183
|
+
canonical_invocations = [dict(expected_counters)]
|
|
1184
|
+
expected_outputs = _verified_receipt_outputs(installed)
|
|
1185
|
+
expected_receipt = {
|
|
1186
|
+
"schema_version": PACKED_RECEIPT_SCHEMA,
|
|
1187
|
+
"status": "complete",
|
|
1188
|
+
"provider_id": terminal["generation"]["provider_id"],
|
|
1189
|
+
"adapter_mode": BOUNDED_SCALAR_ADAPTER,
|
|
1190
|
+
# Rebuilt from the *installed head*, not carried over from the retained receipt, so a
|
|
1191
|
+
# recovery reconciles the two documents against each other rather than believing either.
|
|
1192
|
+
# `_parse_head` has already required the head's statement to be a coverage at all.
|
|
1193
|
+
"coverage": dict(installed.head["coverage"]),
|
|
1194
|
+
"inputs": {
|
|
1195
|
+
"delta_manifest_sha256": terminal["generation"]["delta_manifest_sha256"],
|
|
1196
|
+
"delta_root_sha256": delta_manifest["root_sha256"],
|
|
1197
|
+
"authoring_manifest_sha256": terminal["generation"]["authoring_manifest_sha256"],
|
|
1198
|
+
"identity_manifest_sha256": terminal["generation"]["identity_manifest_sha256"],
|
|
1199
|
+
"predecessor_head_sha256": terminal["generation"]["predecessor_head_sha256"],
|
|
1200
|
+
"publication_binding_sha256": terminal["generation"].get("publication_binding_sha256"),
|
|
1201
|
+
},
|
|
1202
|
+
"backends": [backend.to_dict() for backend in packed_backends],
|
|
1203
|
+
"posting_backend_coordinate": posting_backend.coordinate,
|
|
1204
|
+
"counts": dict(expected_counts),
|
|
1205
|
+
"counters": dict(expected_counters),
|
|
1206
|
+
"invocations": canonical_invocations,
|
|
1207
|
+
"outputs": expected_outputs,
|
|
1208
|
+
"residency": _canonical_residency(expected_counts, terminal["generation"]),
|
|
1209
|
+
"limits": limits.to_dict(),
|
|
1210
|
+
}
|
|
1211
|
+
if (
|
|
1212
|
+
dict(receipt) != expected_receipt
|
|
1213
|
+
or installed.head.get("provider_id") != terminal["generation"]["provider_id"]
|
|
1214
|
+
or installed.head.get("generation_sha256")
|
|
1215
|
+
!= terminal["generation"]["delta_manifest_sha256"]
|
|
1216
|
+
or installed.head.get("predecessor_head_sha256")
|
|
1217
|
+
!= terminal["generation"]["predecessor_head_sha256"]
|
|
1218
|
+
or installed.head.get("backends") != expected_receipt["backends"]
|
|
1219
|
+
or installed.head.get("posting_backend_coordinate") != posting_backend.coordinate
|
|
1220
|
+
or installed.head.get("counters") != dict(expected_counters)
|
|
1221
|
+
or invocations != canonical_invocations
|
|
1222
|
+
or installed.head.get("invocations") != canonical_invocations
|
|
1223
|
+
or journal_template.get("invocations") != canonical_invocations
|
|
1224
|
+
or journal.get("invocations") != canonical_invocations
|
|
1225
|
+
):
|
|
1226
|
+
raise _refuse(
|
|
1227
|
+
"PACKED_TERMINAL_COUNTER",
|
|
1228
|
+
"packed.terminal.receipt",
|
|
1229
|
+
"terminal work evidence does not reproduce from the installed generation",
|
|
1230
|
+
)
|
|
1231
|
+
|
|
1232
|
+
|
|
1233
|
+
def _recover_terminal_publication(
|
|
1234
|
+
directory: PackedDirectory,
|
|
1235
|
+
*,
|
|
1236
|
+
output_root: Path,
|
|
1237
|
+
generation: Mapping[str, Any],
|
|
1238
|
+
head_raw: bytes,
|
|
1239
|
+
delta_root: Path,
|
|
1240
|
+
delta_manifest: Mapping[str, Any],
|
|
1241
|
+
delta_shards_descriptor: int | None,
|
|
1242
|
+
identity_descriptor: int | None,
|
|
1243
|
+
authoring_root: Path,
|
|
1244
|
+
authoring_manifest: Mapping[str, Any],
|
|
1245
|
+
packed_backends: Sequence[PackedBackend],
|
|
1246
|
+
seam_backends: Sequence[EmbeddingBackend],
|
|
1247
|
+
posting_backend: PackedBackend,
|
|
1248
|
+
predecessor: _PredecessorPacks | None,
|
|
1249
|
+
limits: PackedLimits,
|
|
1250
|
+
deadline: float,
|
|
1251
|
+
) -> PackedPublicationResult:
|
|
1252
|
+
"""Finish the evidence transaction a hard kill left after installing the packed head."""
|
|
1253
|
+
|
|
1254
|
+
head_sha256 = sha256_bytes(head_raw)
|
|
1255
|
+
installed = verify_packed_generation(
|
|
1256
|
+
output_root,
|
|
1257
|
+
expected_head_sha256=head_sha256,
|
|
1258
|
+
root_descriptor=directory.root_descriptor,
|
|
1259
|
+
)
|
|
1260
|
+
terminal_raw = directory.read(PACKED_TERMINAL_FILENAME, maximum=MAX_PACKED_TERMINAL_BYTES)
|
|
1261
|
+
if terminal_raw is None or sha256_bytes(terminal_raw) != installed.head["terminal_sha256"]:
|
|
1262
|
+
raise _refuse(
|
|
1263
|
+
"PACKED_TERMINAL_DIGEST",
|
|
1264
|
+
"packed.terminal",
|
|
1265
|
+
"the published head does not have its exact terminal record",
|
|
1266
|
+
)
|
|
1267
|
+
terminal = _parse_terminal(terminal_raw)
|
|
1268
|
+
if terminal["generation"] != dict(generation):
|
|
1269
|
+
raise _refuse(
|
|
1270
|
+
"PACKED_TERMINAL_GENERATION",
|
|
1271
|
+
"packed.terminal.generation",
|
|
1272
|
+
"the published terminal record belongs to another generation",
|
|
1273
|
+
)
|
|
1274
|
+
template = terminal["journal"]
|
|
1275
|
+
if (
|
|
1276
|
+
set(template) != {"output_root", "generation", "completed_ranges", "invocations"}
|
|
1277
|
+
or template["output_root"] != os.path.abspath(os.fspath(output_root))
|
|
1278
|
+
or template["generation"] != dict(generation)
|
|
1279
|
+
or not isinstance(template["completed_ranges"], list)
|
|
1280
|
+
or not isinstance(template["invocations"], list)
|
|
1281
|
+
):
|
|
1282
|
+
raise _refuse(
|
|
1283
|
+
"PACKED_TERMINAL_SCHEMA",
|
|
1284
|
+
"packed.terminal.journal",
|
|
1285
|
+
"the terminal journal template differs from this generation",
|
|
1286
|
+
)
|
|
1287
|
+
existing = _read_journal(directory)
|
|
1288
|
+
if existing is None:
|
|
1289
|
+
raise _refuse(
|
|
1290
|
+
"PACKED_JOURNAL_MISSING",
|
|
1291
|
+
"packed.journal",
|
|
1292
|
+
"terminal recovery requires the publication journal",
|
|
1293
|
+
)
|
|
1294
|
+
if existing["completed_ranges"] != template["completed_ranges"]:
|
|
1295
|
+
raise _refuse(
|
|
1296
|
+
"PACKED_JOURNAL_DIGEST",
|
|
1297
|
+
"packed.journal.completed_ranges",
|
|
1298
|
+
"the publication journal differs from the head-bound terminal ranges",
|
|
1299
|
+
)
|
|
1300
|
+
expected_counts, expected_counters = _verify_recovery_ranges(
|
|
1301
|
+
directory,
|
|
1302
|
+
installed,
|
|
1303
|
+
completed_ranges=existing["completed_ranges"],
|
|
1304
|
+
delta_root=delta_root,
|
|
1305
|
+
delta_manifest=delta_manifest,
|
|
1306
|
+
delta_shards_descriptor=delta_shards_descriptor,
|
|
1307
|
+
identity_descriptor=identity_descriptor,
|
|
1308
|
+
authoring_root=authoring_root,
|
|
1309
|
+
authoring_manifest=authoring_manifest,
|
|
1310
|
+
identity_manifest_sha256=generation["identity_manifest_sha256"],
|
|
1311
|
+
packed_backends=packed_backends,
|
|
1312
|
+
seam_backends=seam_backends,
|
|
1313
|
+
posting_backend=posting_backend,
|
|
1314
|
+
predecessor=predecessor,
|
|
1315
|
+
limits=limits,
|
|
1316
|
+
deadline=deadline,
|
|
1317
|
+
)
|
|
1318
|
+
_verify_recovery_evidence(
|
|
1319
|
+
installed,
|
|
1320
|
+
terminal=terminal,
|
|
1321
|
+
journal_template=template,
|
|
1322
|
+
journal=existing,
|
|
1323
|
+
expected_counts=expected_counts,
|
|
1324
|
+
expected_counters=expected_counters,
|
|
1325
|
+
delta_manifest=delta_manifest,
|
|
1326
|
+
packed_backends=packed_backends,
|
|
1327
|
+
posting_backend=posting_backend,
|
|
1328
|
+
limits=limits,
|
|
1329
|
+
)
|
|
1330
|
+
final_journal = _Journal(
|
|
1331
|
+
output_root=template["output_root"],
|
|
1332
|
+
generation=dict(template["generation"]),
|
|
1333
|
+
completed_ranges=[dict(item) for item in template["completed_ranges"]],
|
|
1334
|
+
invocations=[dict(item) for item in template["invocations"]],
|
|
1335
|
+
).document(status="complete", head_sha256=head_sha256)
|
|
1336
|
+
journal_sha256 = sha256_bytes(canonical_json_bytes(final_journal))
|
|
1337
|
+
if existing.get("status") == "incomplete" and existing.get("head_sha256") is None:
|
|
1338
|
+
incomplete = _Journal(
|
|
1339
|
+
output_root=template["output_root"],
|
|
1340
|
+
generation=dict(template["generation"]),
|
|
1341
|
+
completed_ranges=[dict(item) for item in template["completed_ranges"]],
|
|
1342
|
+
invocations=[dict(item) for item in template["invocations"]],
|
|
1343
|
+
).document(status="incomplete", head_sha256=None)
|
|
1344
|
+
if existing != incomplete:
|
|
1345
|
+
raise _refuse(
|
|
1346
|
+
"PACKED_JOURNAL_DIGEST",
|
|
1347
|
+
"packed.journal",
|
|
1348
|
+
"the incomplete journal differs from the head-bound terminal record",
|
|
1349
|
+
)
|
|
1350
|
+
if (
|
|
1351
|
+
directory.publish(
|
|
1352
|
+
PACKED_JOURNAL_FILENAME, final_journal, maximum=MAX_PACKED_JOURNAL_BYTES
|
|
1353
|
+
)
|
|
1354
|
+
!= journal_sha256
|
|
1355
|
+
):
|
|
1356
|
+
raise _refuse(
|
|
1357
|
+
"PACKED_JOURNAL_DIGEST",
|
|
1358
|
+
"packed.journal",
|
|
1359
|
+
"the completed journal differs from the head-bound terminal record",
|
|
1360
|
+
)
|
|
1361
|
+
elif existing != final_journal:
|
|
1362
|
+
raise _refuse(
|
|
1363
|
+
"PACKED_JOURNAL_DIGEST",
|
|
1364
|
+
"packed.journal",
|
|
1365
|
+
"the complete journal differs from the head-bound terminal record",
|
|
1366
|
+
)
|
|
1367
|
+
receipt = _receipt_from_terminal(
|
|
1368
|
+
terminal, head_sha256=head_sha256, journal_sha256=journal_sha256
|
|
1369
|
+
)
|
|
1370
|
+
verify_packed_receipt(receipt, root=output_root, root_descriptor=directory.root_descriptor)
|
|
1371
|
+
counts = receipt.get("counts")
|
|
1372
|
+
counters = receipt.get("counters")
|
|
1373
|
+
if not isinstance(counts, dict) or not isinstance(counters, dict):
|
|
1374
|
+
raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "work totals differ")
|
|
1375
|
+
return PackedPublicationResult(
|
|
1376
|
+
status="complete",
|
|
1377
|
+
head_sha256=head_sha256,
|
|
1378
|
+
journal_sha256=journal_sha256,
|
|
1379
|
+
counters=dict(counters),
|
|
1380
|
+
counts=dict(counts),
|
|
1381
|
+
receipt=receipt,
|
|
1382
|
+
)
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
# ------------------------------------------------------------------------------------------
|
|
1386
|
+
# Publication
|
|
1387
|
+
# ------------------------------------------------------------------------------------------
|
|
1388
|
+
|
|
1389
|
+
|
|
1390
|
+
def publish_packed_generation(
|
|
1391
|
+
*,
|
|
1392
|
+
delta_root: Path,
|
|
1393
|
+
expected_delta_manifest_sha256: str,
|
|
1394
|
+
authoring_root: Path,
|
|
1395
|
+
expected_authoring_manifest_sha256: str,
|
|
1396
|
+
output_root: Path,
|
|
1397
|
+
output_descriptor: int | None = None,
|
|
1398
|
+
delta_descriptor: int | None = None,
|
|
1399
|
+
backends: Sequence[EmbeddingBackend],
|
|
1400
|
+
predecessor_root: Path | None = None,
|
|
1401
|
+
expected_predecessor_head_sha256: str | None = None,
|
|
1402
|
+
expected_predecessor_generation_sha256: str | None = None,
|
|
1403
|
+
limits: PackedLimits | None = None,
|
|
1404
|
+
resume: bool = False,
|
|
1405
|
+
publication_binding_sha256: str | None = None,
|
|
1406
|
+
after_range: Callable[[PackedRangeProgress], None] | None = None,
|
|
1407
|
+
) -> PackedPublicationResult:
|
|
1408
|
+
"""Publish one classified sweep as one packed generation, counting its own work exactly."""
|
|
1409
|
+
|
|
1410
|
+
bounds = PackedLimits() if limits is None else limits
|
|
1411
|
+
if not isinstance(bounds, PackedLimits):
|
|
1412
|
+
raise _refuse("PACKED_LIMIT", "limits", "must be PackedLimits")
|
|
1413
|
+
if type(resume) is not bool:
|
|
1414
|
+
raise _refuse("PACKED_CONTRACT", "resume", "must be a boolean")
|
|
1415
|
+
if publication_binding_sha256 is not None and not _is_digest(publication_binding_sha256):
|
|
1416
|
+
raise _refuse(
|
|
1417
|
+
"PACKED_CONTRACT",
|
|
1418
|
+
"publication_binding_sha256",
|
|
1419
|
+
"must be a lowercase SHA-256 when the caller binds external generation evidence",
|
|
1420
|
+
)
|
|
1421
|
+
packed_backends, seam_backends = _resolve_backends(backends)
|
|
1422
|
+
posting_backend = _posting_backend(packed_backends)
|
|
1423
|
+
predecessor_coordinates = (
|
|
1424
|
+
predecessor_root,
|
|
1425
|
+
expected_predecessor_head_sha256,
|
|
1426
|
+
expected_predecessor_generation_sha256,
|
|
1427
|
+
)
|
|
1428
|
+
if any(value is None for value in predecessor_coordinates) and not all(
|
|
1429
|
+
value is None for value in predecessor_coordinates
|
|
1430
|
+
):
|
|
1431
|
+
raise _refuse(
|
|
1432
|
+
"PACKED_PREDECESSOR_REQUIRED",
|
|
1433
|
+
"predecessor",
|
|
1434
|
+
"a predecessor is named by its root, exact head digest, and exact generation digest, "
|
|
1435
|
+
"or not at all",
|
|
1436
|
+
)
|
|
1437
|
+
if expected_predecessor_generation_sha256 is not None and not _is_digest(
|
|
1438
|
+
expected_predecessor_generation_sha256
|
|
1439
|
+
):
|
|
1440
|
+
raise _refuse(
|
|
1441
|
+
"PACKED_PREDECESSOR_GENERATION",
|
|
1442
|
+
"predecessor.generation_sha256",
|
|
1443
|
+
"must be a lowercase SHA-256 digest",
|
|
1444
|
+
)
|
|
1445
|
+
deadline = time.monotonic() + bounds.max_wall_seconds
|
|
1446
|
+
|
|
1447
|
+
delta_manifest = _read_manifest(
|
|
1448
|
+
Path(delta_root) / DELTA_MANIFEST_FILENAME,
|
|
1449
|
+
expected_sha256=expected_delta_manifest_sha256,
|
|
1450
|
+
maximum=MAX_DELTA_MANIFEST_BYTES,
|
|
1451
|
+
schema=DELTA_MANIFEST_SCHEMA,
|
|
1452
|
+
label="delta.manifest",
|
|
1453
|
+
parent_descriptor=delta_descriptor,
|
|
1454
|
+
name=DELTA_MANIFEST_FILENAME,
|
|
1455
|
+
)
|
|
1456
|
+
authoring_manifest = _read_manifest(
|
|
1457
|
+
Path(authoring_root) / MANIFEST_FILENAME,
|
|
1458
|
+
expected_sha256=expected_authoring_manifest_sha256,
|
|
1459
|
+
maximum=MAX_AUTHORING_MANIFEST_BYTES,
|
|
1460
|
+
schema=AUTHORING_MANIFEST_SCHEMA,
|
|
1461
|
+
label="authoring.manifest",
|
|
1462
|
+
)
|
|
1463
|
+
if delta_manifest["inputs"]["authoring_manifest_sha256"] != expected_authoring_manifest_sha256:
|
|
1464
|
+
raise _refuse(
|
|
1465
|
+
"PACKED_INPUT_GENERATION",
|
|
1466
|
+
"authoring.manifest",
|
|
1467
|
+
"the classification was computed from another authored generation",
|
|
1468
|
+
)
|
|
1469
|
+
admit_authored_generation(Path(authoring_root), authoring_manifest)
|
|
1470
|
+
# Validate every classification descriptor while output is still absent. Shard bytes stay
|
|
1471
|
+
# lazy and bounded, but no caller-controlled size can defer a fixed-ceiling refusal until
|
|
1472
|
+
# publication has opened or mutated its output root.
|
|
1473
|
+
_ClassificationStream(Path(delta_root), delta_manifest)
|
|
1474
|
+
identity_manifest_sha256 = delta_manifest["outputs"]["identity_manifest_sha256"]
|
|
1475
|
+
provider_id = delta_manifest["inputs"]["provider_id"]
|
|
1476
|
+
|
|
1477
|
+
generation = {
|
|
1478
|
+
"authoring_manifest_sha256": expected_authoring_manifest_sha256,
|
|
1479
|
+
"backends": [backend.coordinate for backend in packed_backends],
|
|
1480
|
+
"delta_manifest_sha256": expected_delta_manifest_sha256,
|
|
1481
|
+
"identity_manifest_sha256": identity_manifest_sha256,
|
|
1482
|
+
"posting_backend_coordinate": posting_backend.coordinate,
|
|
1483
|
+
"predecessor_head_sha256": expected_predecessor_head_sha256,
|
|
1484
|
+
"provider_id": provider_id,
|
|
1485
|
+
}
|
|
1486
|
+
if publication_binding_sha256 is not None:
|
|
1487
|
+
generation["publication_binding_sha256"] = publication_binding_sha256
|
|
1488
|
+
|
|
1489
|
+
# Authenticate the complete predecessor coordinate before the output root is opened. The
|
|
1490
|
+
# predecessor is opened again under the publication stack below so its descriptor remains
|
|
1491
|
+
# retained while vectors are read; both reads require the same exact head bytes and generation.
|
|
1492
|
+
if predecessor_root is not None:
|
|
1493
|
+
predecessor_preflight = PackedDirectory(Path(predecessor_root))
|
|
1494
|
+
with predecessor_preflight.opened():
|
|
1495
|
+
_read_predecessor_head(
|
|
1496
|
+
predecessor_preflight,
|
|
1497
|
+
expected_predecessor_head_sha256,
|
|
1498
|
+
expected_predecessor_generation_sha256,
|
|
1499
|
+
)
|
|
1500
|
+
|
|
1501
|
+
directory = PackedDirectory(Path(output_root), root_descriptor=output_descriptor)
|
|
1502
|
+
with directory.opened(), contextlib.ExitStack() as input_stack:
|
|
1503
|
+
delta_shards_descriptor = None
|
|
1504
|
+
identity_descriptor = None
|
|
1505
|
+
if delta_descriptor is not None:
|
|
1506
|
+
delta_shards_descriptor = input_stack.enter_context(
|
|
1507
|
+
_opened_directory_at(delta_descriptor, DELTA_SHARDS_DIRNAME)
|
|
1508
|
+
)
|
|
1509
|
+
identity_descriptor = input_stack.enter_context(
|
|
1510
|
+
_opened_directory_at(delta_descriptor, DELTA_IDENTITY_DIRNAME)
|
|
1511
|
+
)
|
|
1512
|
+
head_raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
|
|
1513
|
+
if head_raw is not None and resume:
|
|
1514
|
+
with contextlib.ExitStack() as stack:
|
|
1515
|
+
predecessor: _PredecessorPacks | None = None
|
|
1516
|
+
if predecessor_root is not None:
|
|
1517
|
+
predecessor_directory = PackedDirectory(Path(predecessor_root))
|
|
1518
|
+
stack.enter_context(predecessor_directory.opened())
|
|
1519
|
+
predecessor_head = _read_predecessor_head(
|
|
1520
|
+
predecessor_directory,
|
|
1521
|
+
expected_predecessor_head_sha256,
|
|
1522
|
+
expected_predecessor_generation_sha256,
|
|
1523
|
+
)
|
|
1524
|
+
predecessor = _PredecessorPacks(predecessor_directory, predecessor_head)
|
|
1525
|
+
predecessor.require_backends(
|
|
1526
|
+
[backend.coordinate for backend in packed_backends]
|
|
1527
|
+
)
|
|
1528
|
+
return _recover_terminal_publication(
|
|
1529
|
+
directory,
|
|
1530
|
+
output_root=Path(output_root),
|
|
1531
|
+
generation=generation,
|
|
1532
|
+
head_raw=head_raw,
|
|
1533
|
+
delta_root=Path(delta_root),
|
|
1534
|
+
delta_manifest=delta_manifest,
|
|
1535
|
+
delta_shards_descriptor=delta_shards_descriptor,
|
|
1536
|
+
identity_descriptor=identity_descriptor,
|
|
1537
|
+
authoring_root=Path(authoring_root),
|
|
1538
|
+
authoring_manifest=authoring_manifest,
|
|
1539
|
+
packed_backends=packed_backends,
|
|
1540
|
+
seam_backends=seam_backends,
|
|
1541
|
+
posting_backend=posting_backend,
|
|
1542
|
+
predecessor=predecessor,
|
|
1543
|
+
limits=bounds,
|
|
1544
|
+
deadline=deadline,
|
|
1545
|
+
)
|
|
1546
|
+
if head_raw is not None:
|
|
1547
|
+
raise _refuse(
|
|
1548
|
+
"PACKED_OUTPUT_EXISTS",
|
|
1549
|
+
"packed.output",
|
|
1550
|
+
"a published generation is immutable; publish into a new output root",
|
|
1551
|
+
)
|
|
1552
|
+
existing = _read_journal(directory)
|
|
1553
|
+
if resume:
|
|
1554
|
+
if existing is None:
|
|
1555
|
+
raise _refuse(
|
|
1556
|
+
"PACKED_JOURNAL_MISSING",
|
|
1557
|
+
"packed.journal",
|
|
1558
|
+
"an explicit resume needs the work journal its interruption left behind",
|
|
1559
|
+
)
|
|
1560
|
+
journal = _resume_from(
|
|
1561
|
+
directory,
|
|
1562
|
+
existing,
|
|
1563
|
+
output_root=Path(output_root),
|
|
1564
|
+
generation=generation,
|
|
1565
|
+
)
|
|
1566
|
+
else:
|
|
1567
|
+
if existing is not None:
|
|
1568
|
+
raise _refuse(
|
|
1569
|
+
"PACKED_OUTPUT_EXISTS",
|
|
1570
|
+
"packed.output",
|
|
1571
|
+
"this root holds an in-flight publication; resume it explicitly",
|
|
1572
|
+
)
|
|
1573
|
+
journal = _Journal(
|
|
1574
|
+
output_root=os.path.abspath(os.fspath(output_root)), generation=dict(generation)
|
|
1575
|
+
)
|
|
1576
|
+
|
|
1577
|
+
with contextlib.ExitStack() as stack:
|
|
1578
|
+
predecessor: _PredecessorPacks | None = None
|
|
1579
|
+
if predecessor_root is not None:
|
|
1580
|
+
predecessor_directory = PackedDirectory(Path(predecessor_root))
|
|
1581
|
+
stack.enter_context(predecessor_directory.opened())
|
|
1582
|
+
head = _read_predecessor_head(
|
|
1583
|
+
predecessor_directory,
|
|
1584
|
+
expected_predecessor_head_sha256,
|
|
1585
|
+
expected_predecessor_generation_sha256,
|
|
1586
|
+
)
|
|
1587
|
+
predecessor = _PredecessorPacks(predecessor_directory, head)
|
|
1588
|
+
predecessor.require_backends([backend.coordinate for backend in packed_backends])
|
|
1589
|
+
return _publish(
|
|
1590
|
+
directory=directory,
|
|
1591
|
+
journal=journal,
|
|
1592
|
+
delta_root=Path(delta_root),
|
|
1593
|
+
delta_manifest=delta_manifest,
|
|
1594
|
+
delta_shards_descriptor=delta_shards_descriptor,
|
|
1595
|
+
identity_descriptor=identity_descriptor,
|
|
1596
|
+
authoring_root=Path(authoring_root),
|
|
1597
|
+
authoring_manifest=authoring_manifest,
|
|
1598
|
+
identity_manifest_sha256=identity_manifest_sha256,
|
|
1599
|
+
packed_backends=packed_backends,
|
|
1600
|
+
seam_backends=seam_backends,
|
|
1601
|
+
posting_backend=posting_backend,
|
|
1602
|
+
predecessor=predecessor,
|
|
1603
|
+
provider_id=provider_id,
|
|
1604
|
+
generation=generation,
|
|
1605
|
+
limits=bounds,
|
|
1606
|
+
deadline=deadline,
|
|
1607
|
+
after_range=after_range,
|
|
1608
|
+
)
|
|
1609
|
+
|
|
1610
|
+
|
|
1611
|
+
def _resolve_backends(
|
|
1612
|
+
backends: Sequence[EmbeddingBackend],
|
|
1613
|
+
) -> tuple[tuple[PackedBackend, ...], tuple[EmbeddingBackend, ...]]:
|
|
1614
|
+
if not isinstance(backends, Sequence) or not 1 <= len(backends) <= MAX_PACKED_BACKENDS:
|
|
1615
|
+
raise _refuse(
|
|
1616
|
+
"PACKED_BACKEND",
|
|
1617
|
+
"backends",
|
|
1618
|
+
f"a packed generation carries 1 to {MAX_PACKED_BACKENDS} backends",
|
|
1619
|
+
)
|
|
1620
|
+
resolved = []
|
|
1621
|
+
for backend in backends:
|
|
1622
|
+
descriptor = backend.descriptor
|
|
1623
|
+
resolved.append(
|
|
1624
|
+
(
|
|
1625
|
+
PackedBackend(
|
|
1626
|
+
coordinate=descriptor.coordinate,
|
|
1627
|
+
dimensions=descriptor.dimensions,
|
|
1628
|
+
quantization=descriptor.quantization,
|
|
1629
|
+
query_safe=descriptor.query_safe,
|
|
1630
|
+
),
|
|
1631
|
+
backend,
|
|
1632
|
+
)
|
|
1633
|
+
)
|
|
1634
|
+
resolved.sort(key=lambda pair: pair[0].coordinate)
|
|
1635
|
+
coordinates = [packed.coordinate for packed, _backend in resolved]
|
|
1636
|
+
if len(set(coordinates)) != len(coordinates):
|
|
1637
|
+
raise _refuse("PACKED_BACKEND", "backends", "backends must be distinct coordinates")
|
|
1638
|
+
return tuple(packed for packed, _ in resolved), tuple(backend for _, backend in resolved)
|
|
1639
|
+
|
|
1640
|
+
|
|
1641
|
+
def admit_authored_generation(root: Path, manifest: Mapping[str, Any]) -> None:
|
|
1642
|
+
"""Reconcile and scan every authored record before opening the output directory."""
|
|
1643
|
+
|
|
1644
|
+
expected_counts = {
|
|
1645
|
+
"records": 0,
|
|
1646
|
+
"authored": 0,
|
|
1647
|
+
"flagged": 0,
|
|
1648
|
+
"skipped": 0,
|
|
1649
|
+
"failed": 0,
|
|
1650
|
+
"layer_text_truncated": 0,
|
|
1651
|
+
}
|
|
1652
|
+
expected_digests: dict[str, str | None] = dict.fromkeys(EMBEDDING_LAYERS)
|
|
1653
|
+
for item in _AuthoredStream(root, manifest)._records():
|
|
1654
|
+
_admit_authored_item(item)
|
|
1655
|
+
disposition = item["disposition"]
|
|
1656
|
+
expected_counts["records"] += 1
|
|
1657
|
+
expected_counts[disposition] += 1
|
|
1658
|
+
expected_counts["layer_text_truncated"] += len(item["layer_text_truncated"])
|
|
1659
|
+
entry = item["entry"]
|
|
1660
|
+
if entry is not None:
|
|
1661
|
+
for layer, text in _layer_texts(item).items():
|
|
1662
|
+
expected_digests[layer] = _authored_layer_chain(
|
|
1663
|
+
expected_digests[layer], entry["entry_id"], layer, text
|
|
1664
|
+
)
|
|
1665
|
+
staging = manifest.get("staging")
|
|
1666
|
+
coverage = manifest.get("coverage")
|
|
1667
|
+
if (
|
|
1668
|
+
manifest.get("counts") != expected_counts
|
|
1669
|
+
or manifest.get("layer_text_digests") != expected_digests
|
|
1670
|
+
or not isinstance(staging, Mapping)
|
|
1671
|
+
or type(staging.get("record_index_count")) is not int
|
|
1672
|
+
or staging["record_index_count"] < expected_counts["records"]
|
|
1673
|
+
):
|
|
1674
|
+
raise _refuse(
|
|
1675
|
+
"PACKED_INPUT_MANIFEST",
|
|
1676
|
+
"authoring.manifest",
|
|
1677
|
+
"authored manifest totals do not reproduce from its exact shard records",
|
|
1678
|
+
)
|
|
1679
|
+
# This used to be `staging["unique_records"] != expected_counts["records"]`: the sweep found
|
|
1680
|
+
# exactly as many records as the shards hold. That is the exhaustive case of the two
|
|
1681
|
+
# statements below, and it was the only case the pipeline could publish. Stated as two, a
|
|
1682
|
+
# bounded generation is admissible and still has to account for itself exactly -- the corpus
|
|
1683
|
+
# it claims to have drawn from is the one its own staging coordinate names, and the selection
|
|
1684
|
+
# it claims to have taken is the population its own shards actually contain.
|
|
1685
|
+
if (
|
|
1686
|
+
not coverage_is_valid(coverage)
|
|
1687
|
+
or coverage["corpus_records"] != staging.get("unique_records")
|
|
1688
|
+
or coverage["selected_records"] != expected_counts["records"]
|
|
1689
|
+
):
|
|
1690
|
+
raise _refuse(
|
|
1691
|
+
"PACKED_INPUT_COVERAGE",
|
|
1692
|
+
"authoring.manifest.coverage",
|
|
1693
|
+
"the authored generation's stated coverage does not reproduce from its own staging "
|
|
1694
|
+
"coordinate and shard records",
|
|
1695
|
+
)
|
|
1696
|
+
|
|
1697
|
+
|
|
1698
|
+
def _authored_layer_chain(previous: str | None, entry_id: str, layer: str, text: str) -> str:
|
|
1699
|
+
return canonical_sha256(
|
|
1700
|
+
{
|
|
1701
|
+
"previous_sha256": previous,
|
|
1702
|
+
"entry_id": entry_id,
|
|
1703
|
+
"layer": layer,
|
|
1704
|
+
"text_sha256": sha256_bytes(text.encode("utf-8")),
|
|
1705
|
+
}
|
|
1706
|
+
)
|
|
1707
|
+
|
|
1708
|
+
|
|
1709
|
+
_AUTHORED_ITEM_MEMBERS = frozenset(
|
|
1710
|
+
{
|
|
1711
|
+
"provider_record_id",
|
|
1712
|
+
"disposition",
|
|
1713
|
+
"reason_code",
|
|
1714
|
+
"entry",
|
|
1715
|
+
"layer_texts",
|
|
1716
|
+
"layer_text_truncated",
|
|
1717
|
+
}
|
|
1718
|
+
)
|
|
1719
|
+
|
|
1720
|
+
_REASON_CODES_BY_DISPOSITION = {
|
|
1721
|
+
"authored": frozenset({"AUTHOR_OK"}),
|
|
1722
|
+
"flagged": frozenset(
|
|
1723
|
+
{
|
|
1724
|
+
"AUTHOR_RIGHTS_ABSENT",
|
|
1725
|
+
"AUTHOR_RIGHTS_CONDITIONAL",
|
|
1726
|
+
"AUTHOR_RIGHTS_PROHIBITED",
|
|
1727
|
+
"AUTHOR_RIGHTS_PROSE",
|
|
1728
|
+
"AUTHOR_RIGHTS_UNMAPPED",
|
|
1729
|
+
}
|
|
1730
|
+
),
|
|
1731
|
+
"skipped": frozenset(
|
|
1732
|
+
{
|
|
1733
|
+
"AUTHOR_IDENTIFIER_UNSAFE",
|
|
1734
|
+
"AUTHOR_TITLE_ABSENT",
|
|
1735
|
+
"AUTHOR_TITLE_LIMIT",
|
|
1736
|
+
"AUTHOR_TITLE_UNSAFE",
|
|
1737
|
+
}
|
|
1738
|
+
),
|
|
1739
|
+
"failed": frozenset(
|
|
1740
|
+
{
|
|
1741
|
+
"AUTHOR_ENTRY_CONTRACT",
|
|
1742
|
+
"AUTHOR_RECORD_DIGEST",
|
|
1743
|
+
"AUTHOR_RECORD_FIELDS",
|
|
1744
|
+
"AUTHOR_RECORD_SCHEMA",
|
|
1745
|
+
}
|
|
1746
|
+
),
|
|
1747
|
+
}
|
|
1748
|
+
|
|
1749
|
+
assert frozenset().union(*_REASON_CODES_BY_DISPOSITION.values()) == AUTHORING_REASON_CODES
|
|
1750
|
+
|
|
1751
|
+
|
|
1752
|
+
def _admit_authored_item(item: Mapping[str, Any]) -> None:
|
|
1753
|
+
"""Validate one closed authored-record shape and scan every unconstrained public string."""
|
|
1754
|
+
|
|
1755
|
+
if not isinstance(item, dict) or set(item) != _AUTHORED_ITEM_MEMBERS:
|
|
1756
|
+
raise _refuse(
|
|
1757
|
+
"PACKED_INPUT_ENTRY",
|
|
1758
|
+
"authoring.shard.items[]",
|
|
1759
|
+
"an authored item must carry exactly the authored-record members",
|
|
1760
|
+
)
|
|
1761
|
+
disposition = item["disposition"]
|
|
1762
|
+
if disposition not in AUTHORING_DISPOSITIONS:
|
|
1763
|
+
raise _refuse(
|
|
1764
|
+
"PACKED_INPUT_ENTRY",
|
|
1765
|
+
"authoring.shard.items[].disposition",
|
|
1766
|
+
"an authored item disposition differs",
|
|
1767
|
+
)
|
|
1768
|
+
provider_record_id = item["provider_record_id"]
|
|
1769
|
+
reason_code = item["reason_code"]
|
|
1770
|
+
if not isinstance(provider_record_id, str) or not isinstance(reason_code, str):
|
|
1771
|
+
raise _refuse(
|
|
1772
|
+
"PACKED_INPUT_ENTRY",
|
|
1773
|
+
"authoring.shard.items[]",
|
|
1774
|
+
"an authored item has invalid public strings",
|
|
1775
|
+
)
|
|
1776
|
+
if reason_code not in _REASON_CODES_BY_DISPOSITION[disposition]:
|
|
1777
|
+
raise _refuse(
|
|
1778
|
+
"PACKED_INPUT_ENTRY",
|
|
1779
|
+
"authoring.shard.items[].reason_code",
|
|
1780
|
+
"an authored item reason does not match its disposition",
|
|
1781
|
+
)
|
|
1782
|
+
for field_name, value in (
|
|
1783
|
+
("provider_record_id", provider_record_id),
|
|
1784
|
+
("reason_code", reason_code),
|
|
1785
|
+
):
|
|
1786
|
+
admit_public_fact_bytes(canonical_json_bytes({field_name: value}))
|
|
1787
|
+
|
|
1788
|
+
truncated = item["layer_text_truncated"]
|
|
1789
|
+
if (
|
|
1790
|
+
not isinstance(truncated, list)
|
|
1791
|
+
or any(not isinstance(layer, str) or layer not in EMBEDDING_LAYERS for layer in truncated)
|
|
1792
|
+
or len(set(truncated)) != len(truncated)
|
|
1793
|
+
):
|
|
1794
|
+
raise _refuse(
|
|
1795
|
+
"PACKED_LAYER_TEXTS",
|
|
1796
|
+
"authoring.shard.items[].layer_text_truncated",
|
|
1797
|
+
"truncated layers must be a unique subset of the embedding layers",
|
|
1798
|
+
)
|
|
1799
|
+
|
|
1800
|
+
if disposition in {"skipped", "failed"}:
|
|
1801
|
+
if item["entry"] is not None or item["layer_texts"] != [] or truncated:
|
|
1802
|
+
raise _refuse(
|
|
1803
|
+
"PACKED_INPUT_ENTRY",
|
|
1804
|
+
"authoring.shard.items[]",
|
|
1805
|
+
"a skipped or failed item cannot carry public facts",
|
|
1806
|
+
)
|
|
1807
|
+
return
|
|
1808
|
+
|
|
1809
|
+
entry = item["entry"]
|
|
1810
|
+
if not isinstance(entry, dict):
|
|
1811
|
+
raise _refuse(
|
|
1812
|
+
"PACKED_INPUT_ENTRY",
|
|
1813
|
+
"authoring.shard.items[].entry",
|
|
1814
|
+
"an authored or flagged item must carry one public entry",
|
|
1815
|
+
)
|
|
1816
|
+
_admit_entry_prose(entry, provider_record_id=provider_record_id)
|
|
1817
|
+
for text in _layer_texts(item).values():
|
|
1818
|
+
admit_public_fact_bytes(text.encode("utf-8"))
|
|
1819
|
+
|
|
1820
|
+
|
|
1821
|
+
_PUBLIC_PROSE_FACTS = (
|
|
1822
|
+
"title",
|
|
1823
|
+
"publisher",
|
|
1824
|
+
"description",
|
|
1825
|
+
"spatial_scope",
|
|
1826
|
+
"data_formats",
|
|
1827
|
+
"access_kind",
|
|
1828
|
+
"authentication_required",
|
|
1829
|
+
"rights",
|
|
1830
|
+
"declared_columns",
|
|
1831
|
+
"declared_vocabulary",
|
|
1832
|
+
"declared_row_count",
|
|
1833
|
+
"profiles",
|
|
1834
|
+
)
|
|
1835
|
+
|
|
1836
|
+
|
|
1837
|
+
def _admit_entry_prose(entry: Mapping[str, Any], *, provider_record_id: str) -> None:
|
|
1838
|
+
"""Classify publishable semantic values without treating verified digests as prose."""
|
|
1839
|
+
|
|
1840
|
+
try:
|
|
1841
|
+
captured = catalog_entry_v2_from_dict(entry)
|
|
1842
|
+
except SourceContractError as error:
|
|
1843
|
+
raise _refuse(error.code, error.path, error.detail) from error
|
|
1844
|
+
canonical_entry = captured.to_dict()
|
|
1845
|
+
if canonical_entry != entry:
|
|
1846
|
+
raise _refuse(
|
|
1847
|
+
"PACKED_INPUT_ENTRY",
|
|
1848
|
+
"authoring.shard.items[].entry",
|
|
1849
|
+
"the authored entry is not its strict canonical contract spelling",
|
|
1850
|
+
)
|
|
1851
|
+
identity = captured.provider_record
|
|
1852
|
+
if identity.provider_record_id != provider_record_id:
|
|
1853
|
+
raise _refuse(
|
|
1854
|
+
"PACKED_INPUT_IDENTITY",
|
|
1855
|
+
"authoring.shard.items[].provider_record_id",
|
|
1856
|
+
"the authored item key differs from its provider identity",
|
|
1857
|
+
)
|
|
1858
|
+
if captured.entry_id != derive_entry_id(identity.provider_id, provider_record_id):
|
|
1859
|
+
raise _refuse(
|
|
1860
|
+
"PACKED_INPUT_IDENTITY",
|
|
1861
|
+
"authoring.shard.items[].entry.entry_id",
|
|
1862
|
+
"the entry id is not derived from the authored provider identity",
|
|
1863
|
+
)
|
|
1864
|
+
admit_public_fact_bytes(canonical_json_bytes({"provider_record_id": provider_record_id}))
|
|
1865
|
+
for fact_name in _PUBLIC_PROSE_FACTS:
|
|
1866
|
+
fact = canonical_entry[fact_name]
|
|
1867
|
+
# Keep the field name next to its exact value so assignment-shaped content is still
|
|
1868
|
+
# classified, while authenticated coordinates, digests and evidence identifiers cannot
|
|
1869
|
+
# become accidental prose findings.
|
|
1870
|
+
admit_public_fact_bytes(canonical_json_bytes({fact_name: fact["value"]}))
|
|
1871
|
+
for path, value in _publishable_strings(canonical_entry):
|
|
1872
|
+
admit_public_fact_bytes(canonical_json_bytes({path: value}))
|
|
1873
|
+
|
|
1874
|
+
|
|
1875
|
+
def _publishable_strings(value: Any, *, path: tuple[str, ...] = ()) -> Iterator[tuple[str, str]]:
|
|
1876
|
+
"""Yield source-controlled strings, excluding only validated digests and derived labels."""
|
|
1877
|
+
|
|
1878
|
+
if isinstance(value, dict):
|
|
1879
|
+
for key, item in value.items():
|
|
1880
|
+
if key in {"schema_version", "entry_id"}:
|
|
1881
|
+
continue
|
|
1882
|
+
if key.endswith(("_sha256", "_digest")):
|
|
1883
|
+
if item is not None and not _is_digest(item):
|
|
1884
|
+
raise _refuse(
|
|
1885
|
+
"PACKED_INPUT_ENTRY",
|
|
1886
|
+
".".join((*path, key)),
|
|
1887
|
+
"a digest field is not a lowercase SHA-256 digest",
|
|
1888
|
+
)
|
|
1889
|
+
continue
|
|
1890
|
+
yield from _publishable_strings(item, path=(*path, key))
|
|
1891
|
+
return
|
|
1892
|
+
if isinstance(value, list):
|
|
1893
|
+
for item in value:
|
|
1894
|
+
yield from _publishable_strings(item, path=path)
|
|
1895
|
+
return
|
|
1896
|
+
if isinstance(value, str):
|
|
1897
|
+
yield (".".join(path), value)
|
|
1898
|
+
|
|
1899
|
+
|
|
1900
|
+
def _posting_backend(backends: Sequence[PackedBackend]) -> PackedBackend:
|
|
1901
|
+
query_safe = [backend for backend in backends if backend.query_safe]
|
|
1902
|
+
if len(query_safe) != 1:
|
|
1903
|
+
raise _refuse(
|
|
1904
|
+
"PACKED_POSTING_BACKEND",
|
|
1905
|
+
"backends",
|
|
1906
|
+
"postings are derived from exactly one query-safe backend",
|
|
1907
|
+
)
|
|
1908
|
+
return query_safe[0]
|
|
1909
|
+
|
|
1910
|
+
|
|
1911
|
+
def _read_predecessor_head(
|
|
1912
|
+
directory: PackedDirectory,
|
|
1913
|
+
expected: str | None,
|
|
1914
|
+
expected_generation_sha256: str | None,
|
|
1915
|
+
) -> dict[str, Any]:
|
|
1916
|
+
raw = directory.read(PACKED_HEAD_FILENAME, maximum=MAX_PACKED_HEAD_BYTES)
|
|
1917
|
+
if raw is None:
|
|
1918
|
+
raise _refuse(
|
|
1919
|
+
"PACKED_PREDECESSOR_REQUIRED",
|
|
1920
|
+
"predecessor.head",
|
|
1921
|
+
"no readable packed head at the predecessor root",
|
|
1922
|
+
)
|
|
1923
|
+
if sha256_bytes(raw) != expected:
|
|
1924
|
+
raise _refuse(
|
|
1925
|
+
"PACKED_PREDECESSOR_DIGEST",
|
|
1926
|
+
"predecessor.head",
|
|
1927
|
+
"the predecessor generation is not the caller's exact head",
|
|
1928
|
+
)
|
|
1929
|
+
head = parse_canonical_json(raw)
|
|
1930
|
+
if not isinstance(head, dict) or not isinstance(head.get("ranges"), list):
|
|
1931
|
+
raise _refuse(
|
|
1932
|
+
"PACKED_PREDECESSOR_DIGEST", "predecessor.head", "predecessor head contract differs"
|
|
1933
|
+
)
|
|
1934
|
+
if head.get("generation_sha256") != expected_generation_sha256:
|
|
1935
|
+
raise _refuse(
|
|
1936
|
+
"PACKED_PREDECESSOR_GENERATION",
|
|
1937
|
+
"predecessor.head.generation_sha256",
|
|
1938
|
+
"the predecessor packed head was not published from the named history generation",
|
|
1939
|
+
)
|
|
1940
|
+
return head
|
|
1941
|
+
|
|
1942
|
+
|
|
1943
|
+
def _publish(
|
|
1944
|
+
*,
|
|
1945
|
+
directory: PackedDirectory,
|
|
1946
|
+
journal: _Journal,
|
|
1947
|
+
delta_root: Path,
|
|
1948
|
+
delta_manifest: Mapping[str, Any],
|
|
1949
|
+
delta_shards_descriptor: int | None,
|
|
1950
|
+
identity_descriptor: int | None,
|
|
1951
|
+
authoring_root: Path,
|
|
1952
|
+
authoring_manifest: Mapping[str, Any],
|
|
1953
|
+
identity_manifest_sha256: str,
|
|
1954
|
+
packed_backends: tuple[PackedBackend, ...],
|
|
1955
|
+
seam_backends: tuple[EmbeddingBackend, ...],
|
|
1956
|
+
posting_backend: PackedBackend,
|
|
1957
|
+
predecessor: _PredecessorPacks | None,
|
|
1958
|
+
provider_id: str,
|
|
1959
|
+
generation: Mapping[str, Any],
|
|
1960
|
+
limits: PackedLimits,
|
|
1961
|
+
deadline: float,
|
|
1962
|
+
after_range: Callable[[PackedRangeProgress], None] | None,
|
|
1963
|
+
) -> PackedPublicationResult:
|
|
1964
|
+
counters = _Counters()
|
|
1965
|
+
seams = {
|
|
1966
|
+
backend.descriptor.coordinate: _CountingSeam(backend, counters) for backend in seam_backends
|
|
1967
|
+
}
|
|
1968
|
+
classification = _ClassificationStream(
|
|
1969
|
+
delta_root, delta_manifest, shards_descriptor=delta_shards_descriptor
|
|
1970
|
+
)
|
|
1971
|
+
authored = _AuthoredStream(authoring_root, authoring_manifest)
|
|
1972
|
+
identity = _HistoryStream(
|
|
1973
|
+
IdentityHistoryStore.open(
|
|
1974
|
+
delta_root / DELTA_IDENTITY_DIRNAME,
|
|
1975
|
+
expected_manifest_sha256=identity_manifest_sha256,
|
|
1976
|
+
root_descriptor=identity_descriptor,
|
|
1977
|
+
)
|
|
1978
|
+
)
|
|
1979
|
+
already = len(journal.completed_ranges)
|
|
1980
|
+
counts = {
|
|
1981
|
+
"ranges": 0,
|
|
1982
|
+
"entries": 0,
|
|
1983
|
+
"fresh_ranges": 0,
|
|
1984
|
+
"new": 0,
|
|
1985
|
+
"changed": 0,
|
|
1986
|
+
"unchanged": 0,
|
|
1987
|
+
"members": 0,
|
|
1988
|
+
}
|
|
1989
|
+
for record in journal.invocations:
|
|
1990
|
+
counters.absorb_record(record)
|
|
1991
|
+
invocation_start = counters.to_dict()
|
|
1992
|
+
descriptors: list[PackedRangeDescriptor] = []
|
|
1993
|
+
vector_payload_sha256s: list[str] = []
|
|
1994
|
+
for range_index, entries in enumerate(_ranges(classification, authored, identity)):
|
|
1995
|
+
if range_index >= limits.max_ranges:
|
|
1996
|
+
raise _refuse(
|
|
1997
|
+
"PACKED_RANGE_LIMIT", "packed.ranges", "the generation exceeds its range bound"
|
|
1998
|
+
)
|
|
1999
|
+
range_input = PackedRangeInput(
|
|
2000
|
+
range_index=range_index, entries=tuple(item.to_entry() for item in entries)
|
|
2001
|
+
)
|
|
2002
|
+
fresh = tuple(
|
|
2003
|
+
position
|
|
2004
|
+
for position, item in enumerate(entries)
|
|
2005
|
+
if item.classification in FRESH_CLASSIFICATIONS
|
|
2006
|
+
)
|
|
2007
|
+
encoded_positions = set(fresh)
|
|
2008
|
+
reused = tuple(
|
|
2009
|
+
position for position in range(len(entries)) if position not in encoded_positions
|
|
2010
|
+
)
|
|
2011
|
+
# Every range this generation contains is counted here, whether this invocation built it
|
|
2012
|
+
# or a previous one did, so the counter equations hold across a resume as written.
|
|
2013
|
+
counts["ranges"] += 1
|
|
2014
|
+
counts["entries"] += len(entries)
|
|
2015
|
+
counts["fresh_ranges"] += 1 if fresh else 0
|
|
2016
|
+
for item in entries:
|
|
2017
|
+
counts[item.classification] += 1
|
|
2018
|
+
if range_index < already:
|
|
2019
|
+
completed = journal.completed_ranges[range_index]
|
|
2020
|
+
_verify_completed_range(directory, completed, range_input)
|
|
2021
|
+
descriptors.append(_completed_descriptor(completed))
|
|
2022
|
+
vector_payload_sha256s.extend(completed["vector_payload_sha256s"])
|
|
2023
|
+
counts["members"] += len(completed["member_sha256s"])
|
|
2024
|
+
continue
|
|
2025
|
+
if time.monotonic() >= deadline:
|
|
2026
|
+
raise _refuse(
|
|
2027
|
+
"PACKED_WALL_LIMIT", "packed.output", "publication exceeded its wall bound"
|
|
2028
|
+
)
|
|
2029
|
+
if reused and predecessor is None:
|
|
2030
|
+
raise _refuse(
|
|
2031
|
+
"PACKED_PREDECESSOR_REQUIRED",
|
|
2032
|
+
"predecessor",
|
|
2033
|
+
"an unchanged entry can only be published against the packs it reuses",
|
|
2034
|
+
)
|
|
2035
|
+
recovered: dict[int, dict[str, dict[str, tuple[int, ...]]]] = {}
|
|
2036
|
+
for position in reused:
|
|
2037
|
+
assert predecessor is not None
|
|
2038
|
+
recovered[position] = predecessor.lookup(entries[position].key)
|
|
2039
|
+
|
|
2040
|
+
vectors: dict[str, dict[str, tuple[tuple[int, ...], ...]]] = {}
|
|
2041
|
+
for backend in packed_backends:
|
|
2042
|
+
seam = seams[backend.coordinate]
|
|
2043
|
+
layers: dict[str, tuple[tuple[int, ...], ...]] = {}
|
|
2044
|
+
for layer in EMBEDDING_LAYERS:
|
|
2045
|
+
column: list[tuple[int, ...] | None] = [None] * len(entries)
|
|
2046
|
+
if fresh:
|
|
2047
|
+
encoded = seam.encode_many(
|
|
2048
|
+
tuple(entries[position].layer_texts[layer] for position in fresh)
|
|
2049
|
+
)
|
|
2050
|
+
for position, vector in zip(fresh, encoded, strict=True):
|
|
2051
|
+
column[position] = vector
|
|
2052
|
+
for position in reused:
|
|
2053
|
+
vector = recovered[position].get(backend.coordinate, {}).get(layer)
|
|
2054
|
+
if vector is None or len(vector) != backend.dimensions:
|
|
2055
|
+
raise _refuse(
|
|
2056
|
+
"PACKED_REUSE_MISSING",
|
|
2057
|
+
"predecessor.ranges",
|
|
2058
|
+
"an unchanged entry has no reusable layer vector for this backend",
|
|
2059
|
+
)
|
|
2060
|
+
column[position] = vector
|
|
2061
|
+
layers[layer] = tuple(value for value in column if value is not None)
|
|
2062
|
+
if len(layers[layer]) != len(entries):
|
|
2063
|
+
raise _refuse( # pragma: no cover - every position is filled above
|
|
2064
|
+
"PACKED_RANGE_ENTRY",
|
|
2065
|
+
"packed.range.vectors",
|
|
2066
|
+
"a layer member does not cover every entry in its range",
|
|
2067
|
+
)
|
|
2068
|
+
vectors[backend.coordinate] = layers
|
|
2069
|
+
|
|
2070
|
+
built = build_packed_range(
|
|
2071
|
+
range_input,
|
|
2072
|
+
vectors,
|
|
2073
|
+
backends=packed_backends,
|
|
2074
|
+
posting_backend_coordinate=posting_backend.coordinate,
|
|
2075
|
+
max_terms_per_segment=limits.max_terms_per_segment,
|
|
2076
|
+
)
|
|
2077
|
+
member_sha256s = _install_range(directory, built)
|
|
2078
|
+
if directory.bytes_written > limits.max_disk_bytes:
|
|
2079
|
+
raise _refuse(
|
|
2080
|
+
"PACKED_DISK_LIMIT", "packed.output", "publication exceeds its local disk bound"
|
|
2081
|
+
)
|
|
2082
|
+
counters.reused_layer_items += len(reused) * len(EMBEDDING_LAYERS) * len(packed_backends)
|
|
2083
|
+
payloads = _vector_payload_sha256s(built)
|
|
2084
|
+
completed = _completed_range_record(built)
|
|
2085
|
+
if completed["member_sha256s"] != [list(item) for item in member_sha256s]:
|
|
2086
|
+
raise _refuse( # pragma: no cover - installation checks every built descriptor
|
|
2087
|
+
"PACKED_OUTPUT_READBACK",
|
|
2088
|
+
"packed.output.member",
|
|
2089
|
+
"installed member coordinates differ from the range that produced them",
|
|
2090
|
+
)
|
|
2091
|
+
journal.completed_ranges.append(completed)
|
|
2092
|
+
descriptors.append(built.descriptor)
|
|
2093
|
+
vector_payload_sha256s.extend(payloads)
|
|
2094
|
+
counts["members"] += len(built.members)
|
|
2095
|
+
_write_journal(directory, journal, counters, invocation_start, status="incomplete")
|
|
2096
|
+
if after_range is not None:
|
|
2097
|
+
after_range(
|
|
2098
|
+
PackedRangeProgress(
|
|
2099
|
+
range_index=range_index,
|
|
2100
|
+
first_key=built.first_key,
|
|
2101
|
+
last_key=built.last_key,
|
|
2102
|
+
entry_count=built.entry_count,
|
|
2103
|
+
fresh_entries=len(fresh),
|
|
2104
|
+
reused_entries=len(reused),
|
|
2105
|
+
range_binding_sha256=built.range_binding_sha256,
|
|
2106
|
+
posting_backend_coordinate=posting_backend.coordinate,
|
|
2107
|
+
manifest_sha256=built.manifest_sha256,
|
|
2108
|
+
member_sha256s=member_sha256s,
|
|
2109
|
+
)
|
|
2110
|
+
)
|
|
2111
|
+
|
|
2112
|
+
if not descriptors:
|
|
2113
|
+
raise _refuse(
|
|
2114
|
+
"PACKED_INPUT_EMPTY",
|
|
2115
|
+
"delta.shards",
|
|
2116
|
+
"a classification with no publishable entry publishes no generation",
|
|
2117
|
+
)
|
|
2118
|
+
if counts["ranges"] < already:
|
|
2119
|
+
raise _refuse(
|
|
2120
|
+
"PACKED_JOURNAL_RANGE",
|
|
2121
|
+
"packed.journal.completed_ranges",
|
|
2122
|
+
"the journal completed more ranges than this classification produces",
|
|
2123
|
+
)
|
|
2124
|
+
if counts["entries"] > limits.max_entries:
|
|
2125
|
+
raise _refuse(
|
|
2126
|
+
"PACKED_ENTRY_LIMIT", "packed.entries", "the generation exceeds its entry bound"
|
|
2127
|
+
)
|
|
2128
|
+
|
|
2129
|
+
# The independent final reader: every emitted pack is re-read from disk and its accelerators
|
|
2130
|
+
# recomputed, before any head exists to point at them.
|
|
2131
|
+
posting_term_count = 0
|
|
2132
|
+
posting_member_sha256s: list[str] = []
|
|
2133
|
+
bound_member_sha256s: list[str] = []
|
|
2134
|
+
for descriptor in descriptors:
|
|
2135
|
+
verified = verify_packed_range(
|
|
2136
|
+
directory,
|
|
2137
|
+
descriptor.manifest_sha256,
|
|
2138
|
+
backends=packed_backends,
|
|
2139
|
+
posting_backend_coordinate=posting_backend.coordinate,
|
|
2140
|
+
)
|
|
2141
|
+
if (
|
|
2142
|
+
verified.entry_count != descriptor.entry_count
|
|
2143
|
+
or verified.facts_sha256 != descriptor.facts_sha256
|
|
2144
|
+
or verified.first_key != descriptor.first_key
|
|
2145
|
+
or verified.last_key != descriptor.last_key
|
|
2146
|
+
):
|
|
2147
|
+
raise _refuse(
|
|
2148
|
+
"PACKED_RANGE_DIGEST",
|
|
2149
|
+
"packed.range",
|
|
2150
|
+
"an emitted range does not read back as the range its descriptor names",
|
|
2151
|
+
)
|
|
2152
|
+
posting_term_count += verified.posting_term_count
|
|
2153
|
+
posting_member_sha256s.extend(verified.posting_sha256s)
|
|
2154
|
+
bound_member_sha256s.extend(verified.bound_sha256s)
|
|
2155
|
+
|
|
2156
|
+
_verify_counter_equations(counters, counts, backend_count=len(packed_backends))
|
|
2157
|
+
|
|
2158
|
+
# Invocation boundaries are process history and cannot be authenticated after a hard kill.
|
|
2159
|
+
# The durable representation is therefore one canonical aggregate, which is independently
|
|
2160
|
+
# reproducible from the exact inputs and the classifications in the immutable facts members.
|
|
2161
|
+
invocations = [counters.to_dict()]
|
|
2162
|
+
residency = _canonical_residency(counts, generation)
|
|
2163
|
+
terminal = _terminal_document(
|
|
2164
|
+
provider_id=provider_id,
|
|
2165
|
+
generation=generation,
|
|
2166
|
+
coverage=authoring_manifest["coverage"],
|
|
2167
|
+
delta_manifest=delta_manifest,
|
|
2168
|
+
packed_backends=packed_backends,
|
|
2169
|
+
posting_backend=posting_backend,
|
|
2170
|
+
counters=counters,
|
|
2171
|
+
counts=counts,
|
|
2172
|
+
invocations=invocations,
|
|
2173
|
+
descriptors=descriptors,
|
|
2174
|
+
posting_term_count=posting_term_count,
|
|
2175
|
+
vector_payload_sha256s=vector_payload_sha256s,
|
|
2176
|
+
posting_member_sha256s=posting_member_sha256s,
|
|
2177
|
+
bound_member_sha256s=bound_member_sha256s,
|
|
2178
|
+
limits=limits,
|
|
2179
|
+
residency=residency,
|
|
2180
|
+
journal=journal,
|
|
2181
|
+
)
|
|
2182
|
+
terminal_sha256 = sha256_bytes(canonical_json_bytes(terminal))
|
|
2183
|
+
head = build_packed_head(
|
|
2184
|
+
provider_id=provider_id,
|
|
2185
|
+
generation_sha256=generation["delta_manifest_sha256"],
|
|
2186
|
+
predecessor_head_sha256=generation["predecessor_head_sha256"],
|
|
2187
|
+
backends=packed_backends,
|
|
2188
|
+
posting_backend_coordinate=posting_backend.coordinate,
|
|
2189
|
+
ranges=descriptors,
|
|
2190
|
+
coverage=authoring_manifest["coverage"],
|
|
2191
|
+
counters=counters.to_dict(),
|
|
2192
|
+
invocations=invocations,
|
|
2193
|
+
terminal_sha256=terminal_sha256,
|
|
2194
|
+
)
|
|
2195
|
+
head_sha256 = sha256_bytes(canonical_json_bytes(head))
|
|
2196
|
+
complete_journal = _Journal(
|
|
2197
|
+
output_root=journal.output_root,
|
|
2198
|
+
generation=journal.generation,
|
|
2199
|
+
completed_ranges=journal.completed_ranges,
|
|
2200
|
+
invocations=invocations,
|
|
2201
|
+
).document(status="complete", head_sha256=head_sha256)
|
|
2202
|
+
journal_sha256 = sha256_bytes(canonical_json_bytes(complete_journal))
|
|
2203
|
+
receipt = _receipt_from_terminal(
|
|
2204
|
+
terminal, head_sha256=head_sha256, journal_sha256=journal_sha256
|
|
2205
|
+
)
|
|
2206
|
+
|
|
2207
|
+
final_incomplete_journal = _Journal(
|
|
2208
|
+
output_root=journal.output_root,
|
|
2209
|
+
generation=journal.generation,
|
|
2210
|
+
completed_ranges=journal.completed_ranges,
|
|
2211
|
+
invocations=invocations,
|
|
2212
|
+
).document(status="incomplete", head_sha256=None)
|
|
2213
|
+
directory.publish(
|
|
2214
|
+
PACKED_JOURNAL_FILENAME,
|
|
2215
|
+
final_incomplete_journal,
|
|
2216
|
+
maximum=MAX_PACKED_JOURNAL_BYTES,
|
|
2217
|
+
)
|
|
2218
|
+
installed_terminal_sha256 = directory.publish(
|
|
2219
|
+
PACKED_TERMINAL_FILENAME, terminal, maximum=MAX_PACKED_TERMINAL_BYTES
|
|
2220
|
+
)
|
|
2221
|
+
if installed_terminal_sha256 != terminal_sha256:
|
|
2222
|
+
raise _refuse(
|
|
2223
|
+
"PACKED_TERMINAL_DIGEST",
|
|
2224
|
+
"packed.terminal",
|
|
2225
|
+
"installed terminal record differs from the record the head names",
|
|
2226
|
+
)
|
|
2227
|
+
installed_head_sha256 = directory.publish(
|
|
2228
|
+
PACKED_HEAD_FILENAME, head, maximum=MAX_PACKED_HEAD_BYTES
|
|
2229
|
+
)
|
|
2230
|
+
if installed_head_sha256 != head_sha256:
|
|
2231
|
+
raise _refuse(
|
|
2232
|
+
"PACKED_HEAD_MISMATCH", "packed.head", "installed head differs from its coordinate"
|
|
2233
|
+
)
|
|
2234
|
+
installed = verify_packed_generation(
|
|
2235
|
+
directory.root,
|
|
2236
|
+
expected_head_sha256=head_sha256,
|
|
2237
|
+
root_descriptor=directory.root_descriptor,
|
|
2238
|
+
)
|
|
2239
|
+
if (
|
|
2240
|
+
installed.entry_count != counts["entries"]
|
|
2241
|
+
or installed.range_count != counts["ranges"]
|
|
2242
|
+
or installed.member_count != counts["members"]
|
|
2243
|
+
):
|
|
2244
|
+
raise _refuse(
|
|
2245
|
+
"PACKED_HEAD_EXPECTED",
|
|
2246
|
+
"packed.head",
|
|
2247
|
+
"the installed head does not read back as the generation just written",
|
|
2248
|
+
)
|
|
2249
|
+
|
|
2250
|
+
installed_journal_sha256 = directory.publish(
|
|
2251
|
+
PACKED_JOURNAL_FILENAME, complete_journal, maximum=MAX_PACKED_JOURNAL_BYTES
|
|
2252
|
+
)
|
|
2253
|
+
if installed_journal_sha256 != journal_sha256:
|
|
2254
|
+
raise _refuse(
|
|
2255
|
+
"PACKED_JOURNAL_DIGEST",
|
|
2256
|
+
"packed.journal",
|
|
2257
|
+
"installed complete journal differs from the terminal record",
|
|
2258
|
+
)
|
|
2259
|
+
return PackedPublicationResult(
|
|
2260
|
+
status="complete",
|
|
2261
|
+
head_sha256=head_sha256,
|
|
2262
|
+
journal_sha256=journal_sha256,
|
|
2263
|
+
counters=counters.to_dict(),
|
|
2264
|
+
counts=counts,
|
|
2265
|
+
receipt=receipt,
|
|
2266
|
+
)
|
|
2267
|
+
|
|
2268
|
+
|
|
2269
|
+
@dataclass(frozen=True)
|
|
2270
|
+
class _PublishableEntry:
|
|
2271
|
+
"""One classified entry with everything the packed schema needs and nothing it does not."""
|
|
2272
|
+
|
|
2273
|
+
key: str
|
|
2274
|
+
entry_id: str
|
|
2275
|
+
classification: str
|
|
2276
|
+
semantic_facts_digest: str
|
|
2277
|
+
entry: dict[str, Any]
|
|
2278
|
+
history: dict[str, Any]
|
|
2279
|
+
layer_texts: dict[str, str]
|
|
2280
|
+
|
|
2281
|
+
def to_entry(self) -> dict[str, Any]:
|
|
2282
|
+
return {
|
|
2283
|
+
"key": self.key,
|
|
2284
|
+
"entry_id": self.entry_id,
|
|
2285
|
+
"classification": self.classification,
|
|
2286
|
+
"semantic_facts_digest": self.semantic_facts_digest,
|
|
2287
|
+
"entry": self.entry,
|
|
2288
|
+
"history": self.history,
|
|
2289
|
+
}
|
|
2290
|
+
|
|
2291
|
+
|
|
2292
|
+
def _ranges(
|
|
2293
|
+
classification: _ClassificationStream,
|
|
2294
|
+
authored: _AuthoredStream,
|
|
2295
|
+
identity: _HistoryStream,
|
|
2296
|
+
) -> Iterator[list[_PublishableEntry]]:
|
|
2297
|
+
buffer: list[_PublishableEntry] = []
|
|
2298
|
+
for item in classification:
|
|
2299
|
+
buffer.append(_publishable(item, authored, identity))
|
|
2300
|
+
if len(buffer) == RANGE_ENTRIES:
|
|
2301
|
+
yield buffer
|
|
2302
|
+
buffer = []
|
|
2303
|
+
if buffer:
|
|
2304
|
+
yield buffer
|
|
2305
|
+
|
|
2306
|
+
|
|
2307
|
+
def _publishable(
|
|
2308
|
+
item: Mapping[str, Any], authored: _AuthoredStream, identity: _HistoryStream
|
|
2309
|
+
) -> _PublishableEntry:
|
|
2310
|
+
record_id = item["provider_record_id"]
|
|
2311
|
+
authored_item = authored.advance_to(record_id)
|
|
2312
|
+
history = identity.advance_to(record_id)
|
|
2313
|
+
entry = authored_item.get("entry")
|
|
2314
|
+
if not isinstance(entry, dict) or entry.get("entry_id") != item["entry_id"]:
|
|
2315
|
+
raise _refuse(
|
|
2316
|
+
"PACKED_INPUT_ENTRY",
|
|
2317
|
+
"authoring.shard.items[].entry",
|
|
2318
|
+
"a classified entry does not match the authored entry it names",
|
|
2319
|
+
)
|
|
2320
|
+
if history.identity.entry_id != item["entry_id"]:
|
|
2321
|
+
raise _refuse(
|
|
2322
|
+
"PACKED_INPUT_IDENTITY",
|
|
2323
|
+
"identity.records",
|
|
2324
|
+
"a classified entry does not match the identity history it names",
|
|
2325
|
+
)
|
|
2326
|
+
layer_texts = _layer_texts(authored_item)
|
|
2327
|
+
return _PublishableEntry(
|
|
2328
|
+
key=record_id,
|
|
2329
|
+
entry_id=item["entry_id"],
|
|
2330
|
+
classification=item["classification"],
|
|
2331
|
+
semantic_facts_digest=item["semantic_facts_digest"],
|
|
2332
|
+
entry=entry,
|
|
2333
|
+
history=history.to_dict(),
|
|
2334
|
+
layer_texts=layer_texts,
|
|
2335
|
+
)
|
|
2336
|
+
|
|
2337
|
+
|
|
2338
|
+
def _layer_texts(authored_item: Mapping[str, Any]) -> dict[str, str]:
|
|
2339
|
+
"""Read the exact four authored texts whose UTF-8 bytes the encoding seam meters."""
|
|
2340
|
+
|
|
2341
|
+
texts = authored_item.get("layer_texts")
|
|
2342
|
+
if not isinstance(texts, list) or len(texts) != len(EMBEDDING_LAYERS):
|
|
2343
|
+
raise _refuse(
|
|
2344
|
+
"PACKED_LAYER_TEXTS",
|
|
2345
|
+
"authoring.shard.items[].layer_texts",
|
|
2346
|
+
f"a published entry carries exactly the four layer texts {list(EMBEDDING_LAYERS)}",
|
|
2347
|
+
)
|
|
2348
|
+
layer_texts: dict[str, str] = {}
|
|
2349
|
+
for text in texts:
|
|
2350
|
+
if (
|
|
2351
|
+
not isinstance(text, dict)
|
|
2352
|
+
or set(text) != {"layer", "text"}
|
|
2353
|
+
or text.get("layer") not in EMBEDDING_LAYERS
|
|
2354
|
+
or not isinstance(text.get("text"), str)
|
|
2355
|
+
or not text["text"]
|
|
2356
|
+
or len(text["text"].encode("utf-8")) > MAX_LAYER_TEXT_BYTES
|
|
2357
|
+
):
|
|
2358
|
+
raise _refuse(
|
|
2359
|
+
"PACKED_LAYER_TEXTS",
|
|
2360
|
+
"authoring.shard.items[].layer_texts[]",
|
|
2361
|
+
"each layer text names its layer and carries text",
|
|
2362
|
+
)
|
|
2363
|
+
layer_texts[text["layer"]] = text["text"]
|
|
2364
|
+
if set(layer_texts) != set(EMBEDDING_LAYERS) or [text["layer"] for text in texts] != list(
|
|
2365
|
+
EMBEDDING_LAYERS
|
|
2366
|
+
):
|
|
2367
|
+
raise _refuse(
|
|
2368
|
+
"PACKED_LAYER_TEXTS",
|
|
2369
|
+
"authoring.shard.items[].layer_texts",
|
|
2370
|
+
f"a published entry carries exactly the four layer texts {list(EMBEDDING_LAYERS)}",
|
|
2371
|
+
)
|
|
2372
|
+
return layer_texts
|
|
2373
|
+
|
|
2374
|
+
|
|
2375
|
+
def _install_range(
|
|
2376
|
+
directory: PackedDirectory, built: BuiltPackedRange
|
|
2377
|
+
) -> tuple[tuple[str, str], ...]:
|
|
2378
|
+
installed: list[tuple[str, str]] = []
|
|
2379
|
+
for member in built.members:
|
|
2380
|
+
digest, size = directory.write_member_bytes(member.raw, kind=member.descriptor.kind)
|
|
2381
|
+
if digest != member.descriptor.sha256 or size != member.descriptor.bytes:
|
|
2382
|
+
raise _refuse( # pragma: no cover - content addressing makes this unreachable
|
|
2383
|
+
"PACKED_OUTPUT_READBACK",
|
|
2384
|
+
"packed.output.member",
|
|
2385
|
+
"an installed member is not the member it was built as",
|
|
2386
|
+
)
|
|
2387
|
+
installed.append((member.descriptor.kind, digest))
|
|
2388
|
+
manifest_sha256, _size = directory.write_member_bytes(built.manifest_raw, kind="range")
|
|
2389
|
+
if manifest_sha256 != built.manifest_sha256:
|
|
2390
|
+
raise _refuse( # pragma: no cover - content addressing makes this unreachable
|
|
2391
|
+
"PACKED_OUTPUT_READBACK",
|
|
2392
|
+
"packed.output.manifest",
|
|
2393
|
+
"an installed range manifest is not the manifest it was built as",
|
|
2394
|
+
)
|
|
2395
|
+
return tuple(installed)
|
|
2396
|
+
|
|
2397
|
+
|
|
2398
|
+
def _vector_payload_sha256s(built: BuiltPackedRange) -> tuple[str, ...]:
|
|
2399
|
+
"""Digest each layer member's fixed-width vector payload, header excluded, in published order.
|
|
2400
|
+
|
|
2401
|
+
The header binds a member to its range, so two generations that reuse the same vectors publish
|
|
2402
|
+
different member digests: the facts a range carries move even when its vectors do not. The
|
|
2403
|
+
payload is the part reuse is a claim about, so this is the digest a receipt states when it says
|
|
2404
|
+
an unchanged entry's packs are the predecessor's bytes.
|
|
2405
|
+
"""
|
|
2406
|
+
|
|
2407
|
+
return tuple(
|
|
2408
|
+
sha256_bytes(member.raw[MEMBER_HEADER_BYTES:])
|
|
2409
|
+
for member in built.members
|
|
2410
|
+
if member.descriptor.kind == "vector"
|
|
2411
|
+
)
|
|
2412
|
+
|
|
2413
|
+
|
|
2414
|
+
def _completed_range_record(built: BuiltPackedRange) -> dict[str, Any]:
|
|
2415
|
+
"""State the exact durable range record derived from one deterministic build."""
|
|
2416
|
+
|
|
2417
|
+
return {
|
|
2418
|
+
"range_index": built.range_index,
|
|
2419
|
+
"first_key": built.first_key,
|
|
2420
|
+
"last_key": built.last_key,
|
|
2421
|
+
"entry_count": built.entry_count,
|
|
2422
|
+
"facts_sha256": built.facts_sha256,
|
|
2423
|
+
"range_binding_sha256": built.range_binding_sha256,
|
|
2424
|
+
"range_manifest_sha256": built.manifest_sha256,
|
|
2425
|
+
"range_manifest_bytes": len(built.manifest_raw),
|
|
2426
|
+
"member_sha256s": [
|
|
2427
|
+
[member.descriptor.kind, member.descriptor.sha256] for member in built.members
|
|
2428
|
+
],
|
|
2429
|
+
"vector_payload_sha256s": list(_vector_payload_sha256s(built)),
|
|
2430
|
+
}
|
|
2431
|
+
|
|
2432
|
+
|
|
2433
|
+
def _completed_descriptor(completed: Mapping[str, Any]) -> PackedRangeDescriptor:
|
|
2434
|
+
return PackedRangeDescriptor(
|
|
2435
|
+
range_index=completed["range_index"],
|
|
2436
|
+
first_key=completed["first_key"],
|
|
2437
|
+
last_key=completed["last_key"],
|
|
2438
|
+
entry_count=completed["entry_count"],
|
|
2439
|
+
facts_sha256=completed["facts_sha256"],
|
|
2440
|
+
range_binding_sha256=completed["range_binding_sha256"],
|
|
2441
|
+
manifest_sha256=completed["range_manifest_sha256"],
|
|
2442
|
+
manifest_bytes=completed["range_manifest_bytes"],
|
|
2443
|
+
member_count=len(completed["member_sha256s"]),
|
|
2444
|
+
)
|
|
2445
|
+
|
|
2446
|
+
|
|
2447
|
+
def _verify_completed_range(
|
|
2448
|
+
directory: PackedDirectory, completed: Mapping[str, Any], range_input: PackedRangeInput
|
|
2449
|
+
) -> None:
|
|
2450
|
+
"""Prove one already-emitted range is the range this stream produces, encoding nothing.
|
|
2451
|
+
|
|
2452
|
+
An ordinary pre-head resume must not re-encode a complete range, so this first proof is drawn
|
|
2453
|
+
entirely from bytes that already exist: the journal states the interval, size and facts order;
|
|
2454
|
+
the range binding has to reproduce from exactly those; and the installed facts member has to
|
|
2455
|
+
list exactly these entries carrying exactly these canonical documents. Terminal recovery adds
|
|
2456
|
+
a deterministic full-range replay after this inexpensive facts check.
|
|
2457
|
+
"""
|
|
2458
|
+
|
|
2459
|
+
path = f"packed.journal.completed_ranges[{range_input.range_index}]"
|
|
2460
|
+
if (
|
|
2461
|
+
completed["range_index"] != range_input.range_index
|
|
2462
|
+
or completed["first_key"] != range_input.first_key
|
|
2463
|
+
or completed["last_key"] != range_input.last_key
|
|
2464
|
+
or completed["entry_count"] != len(range_input.entries)
|
|
2465
|
+
):
|
|
2466
|
+
raise _refuse(
|
|
2467
|
+
"PACKED_JOURNAL_RANGE",
|
|
2468
|
+
path,
|
|
2469
|
+
"a completed range does not describe the range this stream produces",
|
|
2470
|
+
)
|
|
2471
|
+
if completed["range_binding_sha256"] != range_binding_sha256(
|
|
2472
|
+
range_index=range_input.range_index,
|
|
2473
|
+
first_key=range_input.first_key,
|
|
2474
|
+
last_key=range_input.last_key,
|
|
2475
|
+
entry_count=len(range_input.entries),
|
|
2476
|
+
facts_sha256=completed["facts_sha256"],
|
|
2477
|
+
):
|
|
2478
|
+
raise _refuse(
|
|
2479
|
+
"PACKED_JOURNAL_RANGE",
|
|
2480
|
+
f"{path}.range_binding_sha256",
|
|
2481
|
+
"a completed range's binding does not reproduce from its own interval and facts order",
|
|
2482
|
+
)
|
|
2483
|
+
raw = directory.read_member(
|
|
2484
|
+
completed["facts_sha256"], kind="facts", maximum=MAX_FACTS_MEMBER_BYTES
|
|
2485
|
+
)
|
|
2486
|
+
if raw is None or sha256_bytes(raw) != completed["facts_sha256"]:
|
|
2487
|
+
raise _refuse(
|
|
2488
|
+
"PACKED_JOURNAL_MEMBER",
|
|
2489
|
+
f"{path}.facts_sha256",
|
|
2490
|
+
"an already-emitted immutable member does not read back exactly",
|
|
2491
|
+
)
|
|
2492
|
+
try:
|
|
2493
|
+
payload = parse_canonical_json(raw)
|
|
2494
|
+
except CanonicalJSONError as error:
|
|
2495
|
+
raise _refuse(
|
|
2496
|
+
"PACKED_JOURNAL_RANGE", f"{path}.facts_sha256", "a facts member is not canonical"
|
|
2497
|
+
) from error
|
|
2498
|
+
items = payload.get("entries") if isinstance(payload, dict) else None
|
|
2499
|
+
if not isinstance(items, list) or len(items) != len(range_input.entries):
|
|
2500
|
+
raise _refuse(
|
|
2501
|
+
"PACKED_JOURNAL_RANGE",
|
|
2502
|
+
f"{path}.facts_sha256",
|
|
2503
|
+
"a completed range's facts member does not cover this stream's entries",
|
|
2504
|
+
)
|
|
2505
|
+
for position, (item, entry) in enumerate(zip(items, range_input.entries, strict=True)):
|
|
2506
|
+
try:
|
|
2507
|
+
entry_sha256 = sha256_bytes(canonical_json_bytes(entry["entry"]))
|
|
2508
|
+
except CanonicalJSONError as error:
|
|
2509
|
+
raise _refuse(
|
|
2510
|
+
"PACKED_FACTS_LIMIT",
|
|
2511
|
+
f"packed.range.entries[{position}].entry",
|
|
2512
|
+
"a range document is not canonically representable",
|
|
2513
|
+
) from error
|
|
2514
|
+
if (
|
|
2515
|
+
not isinstance(item, dict)
|
|
2516
|
+
or item.get("key") != entry["key"]
|
|
2517
|
+
or item.get("entry_id") != entry["entry_id"]
|
|
2518
|
+
or item.get("classification") != entry["classification"]
|
|
2519
|
+
or item.get("semantic_facts_digest") != entry["semantic_facts_digest"]
|
|
2520
|
+
or item.get("entry_sha256") != entry_sha256
|
|
2521
|
+
):
|
|
2522
|
+
raise _refuse(
|
|
2523
|
+
"PACKED_JOURNAL_RANGE",
|
|
2524
|
+
f"{path}.entries[{position}]",
|
|
2525
|
+
"a completed range does not carry the entry this stream produces",
|
|
2526
|
+
)
|
|
2527
|
+
|
|
2528
|
+
|
|
2529
|
+
def _invocation_record(counters: _Counters, start: Mapping[str, int]) -> dict[str, int]:
|
|
2530
|
+
return {name: getattr(counters, name) - start[name] for name in COUNTER_MEMBERS}
|
|
2531
|
+
|
|
2532
|
+
|
|
2533
|
+
def _write_journal(
|
|
2534
|
+
directory: PackedDirectory,
|
|
2535
|
+
journal: _Journal,
|
|
2536
|
+
counters: _Counters,
|
|
2537
|
+
start: Mapping[str, int],
|
|
2538
|
+
*,
|
|
2539
|
+
status: str,
|
|
2540
|
+
head_sha256: str | None = None,
|
|
2541
|
+
) -> str:
|
|
2542
|
+
document = _journal_document(journal, counters, start, status=status, head_sha256=head_sha256)
|
|
2543
|
+
return directory.publish(PACKED_JOURNAL_FILENAME, document, maximum=MAX_PACKED_JOURNAL_BYTES)
|
|
2544
|
+
|
|
2545
|
+
|
|
2546
|
+
def _journal_document(
|
|
2547
|
+
journal: _Journal,
|
|
2548
|
+
counters: _Counters,
|
|
2549
|
+
start: Mapping[str, int],
|
|
2550
|
+
*,
|
|
2551
|
+
status: str,
|
|
2552
|
+
head_sha256: str | None = None,
|
|
2553
|
+
) -> dict[str, Any]:
|
|
2554
|
+
"""Build the exact journal document before choosing when to install it."""
|
|
2555
|
+
|
|
2556
|
+
invocations = list(journal.invocations)
|
|
2557
|
+
if status == "incomplete":
|
|
2558
|
+
invocations.append(_invocation_record(counters, start))
|
|
2559
|
+
return _Journal(
|
|
2560
|
+
output_root=journal.output_root,
|
|
2561
|
+
generation=journal.generation,
|
|
2562
|
+
completed_ranges=journal.completed_ranges,
|
|
2563
|
+
invocations=invocations,
|
|
2564
|
+
).document(status=status, head_sha256=head_sha256)
|
|
2565
|
+
|
|
2566
|
+
|
|
2567
|
+
def _verify_counter_equations(
|
|
2568
|
+
counters: _Counters, counts: Mapping[str, int], *, backend_count: int
|
|
2569
|
+
) -> None:
|
|
2570
|
+
fresh_entries = counts["new"] + counts["changed"]
|
|
2571
|
+
layers = len(EMBEDDING_LAYERS)
|
|
2572
|
+
expected = {
|
|
2573
|
+
"batch_invocations": counts["fresh_ranges"] * layers * backend_count,
|
|
2574
|
+
"encoded_layer_items": fresh_entries * layers * backend_count,
|
|
2575
|
+
"backend_scalar_invocations": fresh_entries * layers * backend_count,
|
|
2576
|
+
"reused_layer_items": counts["unchanged"] * layers * backend_count,
|
|
2577
|
+
}
|
|
2578
|
+
for name, value in expected.items():
|
|
2579
|
+
if getattr(counters, name) != value:
|
|
2580
|
+
raise _refuse(
|
|
2581
|
+
"PACKED_COUNTER_MISMATCH",
|
|
2582
|
+
f"packed.counters.{name}",
|
|
2583
|
+
f"the publisher's own work does not satisfy its equation: "
|
|
2584
|
+
f"{getattr(counters, name)} != {value}",
|
|
2585
|
+
)
|
|
2586
|
+
|
|
2587
|
+
|
|
2588
|
+
def _terminal_document(
|
|
2589
|
+
*,
|
|
2590
|
+
provider_id: str,
|
|
2591
|
+
generation: Mapping[str, Any],
|
|
2592
|
+
coverage: Mapping[str, Any],
|
|
2593
|
+
delta_manifest: Mapping[str, Any],
|
|
2594
|
+
packed_backends: Sequence[PackedBackend],
|
|
2595
|
+
posting_backend: PackedBackend,
|
|
2596
|
+
counters: _Counters,
|
|
2597
|
+
counts: Mapping[str, int],
|
|
2598
|
+
invocations: Sequence[Mapping[str, int]],
|
|
2599
|
+
descriptors: Sequence[PackedRangeDescriptor],
|
|
2600
|
+
posting_term_count: int,
|
|
2601
|
+
vector_payload_sha256s: Sequence[str],
|
|
2602
|
+
posting_member_sha256s: Sequence[str],
|
|
2603
|
+
bound_member_sha256s: Sequence[str],
|
|
2604
|
+
limits: PackedLimits,
|
|
2605
|
+
residency: Mapping[str, int],
|
|
2606
|
+
journal: _Journal,
|
|
2607
|
+
) -> dict[str, Any]:
|
|
2608
|
+
"""Persist everything needed to finish evidence after the packed head is durable.
|
|
2609
|
+
|
|
2610
|
+
``outputs.vector_member_sha256s`` is the ordered digest of every layer member's vector payload
|
|
2611
|
+
-- see :func:`_vector_payload_sha256s` -- so a successor that reused a predecessor's vectors
|
|
2612
|
+
states the same list the predecessor did, and one that silently re-encoded cannot. The posting
|
|
2613
|
+
and bound digests beside it are the independent final reader's, not the builder's: they name
|
|
2614
|
+
the accelerator members that reader recomputed from the pack bytes it read back.
|
|
2615
|
+
|
|
2616
|
+
``adapter_mode`` is stated rather than inferred. Every vector in this generation came back on a
|
|
2617
|
+
:class:`BatchEncodeResult` that could only have been minted by the bounded scalar adapter, so
|
|
2618
|
+
the receipt says which implementation produced them instead of leaving a reader to assume.
|
|
2619
|
+
"""
|
|
2620
|
+
|
|
2621
|
+
receipt = {
|
|
2622
|
+
"schema_version": PACKED_RECEIPT_SCHEMA,
|
|
2623
|
+
"status": "complete",
|
|
2624
|
+
"provider_id": provider_id,
|
|
2625
|
+
"adapter_mode": BOUNDED_SCALAR_ADAPTER,
|
|
2626
|
+
# The head and this receipt both state the coverage, and the installed-catalogue reader
|
|
2627
|
+
# requires them to agree. A duplicated summary nothing reconciles is the weak kind; this
|
|
2628
|
+
# one is compared member-for-member in `packed_retrieval._bound_receipt`.
|
|
2629
|
+
"coverage": dict(coverage),
|
|
2630
|
+
"inputs": {
|
|
2631
|
+
"delta_manifest_sha256": generation["delta_manifest_sha256"],
|
|
2632
|
+
"delta_root_sha256": delta_manifest["root_sha256"],
|
|
2633
|
+
"authoring_manifest_sha256": generation["authoring_manifest_sha256"],
|
|
2634
|
+
"identity_manifest_sha256": generation["identity_manifest_sha256"],
|
|
2635
|
+
"predecessor_head_sha256": generation["predecessor_head_sha256"],
|
|
2636
|
+
"publication_binding_sha256": generation.get("publication_binding_sha256"),
|
|
2637
|
+
},
|
|
2638
|
+
"backends": [backend.to_dict() for backend in packed_backends],
|
|
2639
|
+
"posting_backend_coordinate": posting_backend.coordinate,
|
|
2640
|
+
"counts": dict(counts),
|
|
2641
|
+
"counters": counters.to_dict(),
|
|
2642
|
+
"invocations": [dict(record) for record in invocations],
|
|
2643
|
+
"outputs": {
|
|
2644
|
+
"range_count": counts["ranges"],
|
|
2645
|
+
"entry_count": counts["entries"],
|
|
2646
|
+
"member_count": counts["members"],
|
|
2647
|
+
"posting_term_count": posting_term_count,
|
|
2648
|
+
"ranges": [descriptor.to_dict() for descriptor in descriptors],
|
|
2649
|
+
"ranges_chain_sha256": ranges_chain_sha256(descriptors),
|
|
2650
|
+
"vector_member_sha256s": list(vector_payload_sha256s),
|
|
2651
|
+
"posting_member_sha256s": list(posting_member_sha256s),
|
|
2652
|
+
"bound_member_sha256s": list(bound_member_sha256s),
|
|
2653
|
+
},
|
|
2654
|
+
"residency": dict(residency),
|
|
2655
|
+
"limits": limits.to_dict(),
|
|
2656
|
+
}
|
|
2657
|
+
body = {
|
|
2658
|
+
"schema_version": PACKED_TERMINAL_SCHEMA,
|
|
2659
|
+
"generation": dict(generation),
|
|
2660
|
+
"journal": {
|
|
2661
|
+
"output_root": journal.output_root,
|
|
2662
|
+
"generation": dict(journal.generation),
|
|
2663
|
+
"completed_ranges": list(journal.completed_ranges),
|
|
2664
|
+
"invocations": [dict(record) for record in invocations],
|
|
2665
|
+
},
|
|
2666
|
+
"receipt": receipt,
|
|
2667
|
+
}
|
|
2668
|
+
return {**body, "root_sha256": canonical_sha256(body)}
|
|
2669
|
+
|
|
2670
|
+
|
|
2671
|
+
def _receipt_from_terminal(
|
|
2672
|
+
terminal: Mapping[str, Any], *, head_sha256: str, journal_sha256: str
|
|
2673
|
+
) -> dict[str, Any]:
|
|
2674
|
+
"""Complete the packed receipt from the head-bound terminal record."""
|
|
2675
|
+
|
|
2676
|
+
stated = terminal.get("receipt")
|
|
2677
|
+
if not isinstance(stated, Mapping):
|
|
2678
|
+
raise _refuse(
|
|
2679
|
+
"PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt", "terminal receipt differs"
|
|
2680
|
+
)
|
|
2681
|
+
outputs = stated.get("outputs")
|
|
2682
|
+
if not isinstance(outputs, Mapping):
|
|
2683
|
+
raise _refuse("PACKED_TERMINAL_SCHEMA", "packed.terminal.receipt.outputs", "outputs differ")
|
|
2684
|
+
body = {
|
|
2685
|
+
**dict(stated),
|
|
2686
|
+
"outputs": {
|
|
2687
|
+
"head_sha256": head_sha256,
|
|
2688
|
+
"journal_sha256": journal_sha256,
|
|
2689
|
+
**dict(outputs),
|
|
2690
|
+
},
|
|
2691
|
+
}
|
|
2692
|
+
return {**body, "coordinate_sha256": canonical_sha256(body)}
|
|
2693
|
+
|
|
2694
|
+
|
|
2695
|
+
def _parse_terminal(raw: bytes) -> dict[str, Any]:
|
|
2696
|
+
"""Read one head-bound terminal record without accepting an alternate spelling."""
|
|
2697
|
+
|
|
2698
|
+
if len(raw) > MAX_PACKED_TERMINAL_BYTES:
|
|
2699
|
+
raise _refuse("PACKED_TERMINAL_LIMIT", "packed.terminal", "terminal record exceeds its cap")
|
|
2700
|
+
try:
|
|
2701
|
+
terminal = parse_canonical_json(raw)
|
|
2702
|
+
except CanonicalJSONError as error:
|
|
2703
|
+
raise _refuse(
|
|
2704
|
+
"PACKED_TERMINAL_SCHEMA", "packed.terminal", "terminal record is not canonical"
|
|
2705
|
+
) from error
|
|
2706
|
+
if (
|
|
2707
|
+
not isinstance(terminal, dict)
|
|
2708
|
+
or set(terminal) != {"schema_version", "generation", "journal", "receipt", "root_sha256"}
|
|
2709
|
+
or terminal.get("schema_version") != PACKED_TERMINAL_SCHEMA
|
|
2710
|
+
or not isinstance(terminal.get("generation"), dict)
|
|
2711
|
+
or not isinstance(terminal.get("journal"), dict)
|
|
2712
|
+
or not isinstance(terminal.get("receipt"), dict)
|
|
2713
|
+
or not _is_digest(terminal.get("root_sha256"))
|
|
2714
|
+
):
|
|
2715
|
+
raise _refuse(
|
|
2716
|
+
"PACKED_TERMINAL_SCHEMA", "packed.terminal", "terminal record contract differs"
|
|
2717
|
+
)
|
|
2718
|
+
body = {key: value for key, value in terminal.items() if key != "root_sha256"}
|
|
2719
|
+
if canonical_sha256(body) != terminal["root_sha256"]:
|
|
2720
|
+
raise _refuse(
|
|
2721
|
+
"PACKED_TERMINAL_DIGEST",
|
|
2722
|
+
"packed.terminal.root_sha256",
|
|
2723
|
+
"terminal record does not reproduce its own digest",
|
|
2724
|
+
)
|
|
2725
|
+
return terminal
|
|
2726
|
+
|
|
2727
|
+
|
|
2728
|
+
def verify_packed_receipt(
|
|
2729
|
+
receipt: Mapping[str, Any], *, root: Path, root_descriptor: int | None = None
|
|
2730
|
+
) -> None:
|
|
2731
|
+
"""Refuse a receipt that is not the exact coordinate of the generation installed at ``root``."""
|
|
2732
|
+
|
|
2733
|
+
if not isinstance(receipt, Mapping) or receipt.get("schema_version") != PACKED_RECEIPT_SCHEMA:
|
|
2734
|
+
raise _refuse("PACKED_RECEIPT_SCHEMA", "packed.receipt", "receipt contract differs")
|
|
2735
|
+
if receipt.get("adapter_mode") != BOUNDED_SCALAR_ADAPTER:
|
|
2736
|
+
raise _refuse(
|
|
2737
|
+
"PACKED_RECEIPT_ADAPTER",
|
|
2738
|
+
"packed.receipt.adapter_mode",
|
|
2739
|
+
f"a packed generation may claim only the {BOUNDED_SCALAR_ADAPTER!r} adapter",
|
|
2740
|
+
)
|
|
2741
|
+
body = {key: value for key, value in receipt.items() if key != "coordinate_sha256"}
|
|
2742
|
+
if canonical_sha256(body) != receipt.get("coordinate_sha256"):
|
|
2743
|
+
raise _refuse(
|
|
2744
|
+
"PACKED_RECEIPT_COORDINATE",
|
|
2745
|
+
"packed.receipt.coordinate_sha256",
|
|
2746
|
+
"the receipt does not reproduce its own coordinate",
|
|
2747
|
+
)
|
|
2748
|
+
installed = verify_packed_generation(
|
|
2749
|
+
Path(root),
|
|
2750
|
+
expected_head_sha256=receipt["outputs"]["head_sha256"],
|
|
2751
|
+
root_descriptor=root_descriptor,
|
|
2752
|
+
)
|
|
2753
|
+
directory = PackedDirectory(Path(root), root_descriptor=root_descriptor)
|
|
2754
|
+
with directory.opened():
|
|
2755
|
+
terminal_raw = directory.read(PACKED_TERMINAL_FILENAME, maximum=MAX_PACKED_TERMINAL_BYTES)
|
|
2756
|
+
journal_raw = directory.read(PACKED_JOURNAL_FILENAME, maximum=MAX_PACKED_JOURNAL_BYTES)
|
|
2757
|
+
if (
|
|
2758
|
+
terminal_raw is None
|
|
2759
|
+
or journal_raw is None
|
|
2760
|
+
or sha256_bytes(terminal_raw) != installed.head["terminal_sha256"]
|
|
2761
|
+
or sha256_bytes(journal_raw) != receipt["outputs"]["journal_sha256"]
|
|
2762
|
+
):
|
|
2763
|
+
raise _refuse(
|
|
2764
|
+
"PACKED_RECEIPT_MISMATCH",
|
|
2765
|
+
"packed.receipt.outputs",
|
|
2766
|
+
"the receipt does not bind the installed terminal record and journal",
|
|
2767
|
+
)
|
|
2768
|
+
terminal = _parse_terminal(terminal_raw)
|
|
2769
|
+
journal = _read_journal(directory)
|
|
2770
|
+
if (
|
|
2771
|
+
journal is None
|
|
2772
|
+
or journal.get("status") != "complete"
|
|
2773
|
+
or journal.get("head_sha256") != installed.head_sha256
|
|
2774
|
+
or _receipt_from_terminal(
|
|
2775
|
+
terminal,
|
|
2776
|
+
head_sha256=installed.head_sha256,
|
|
2777
|
+
journal_sha256=sha256_bytes(journal_raw),
|
|
2778
|
+
)
|
|
2779
|
+
!= dict(receipt)
|
|
2780
|
+
):
|
|
2781
|
+
raise _refuse(
|
|
2782
|
+
"PACKED_RECEIPT_MISMATCH",
|
|
2783
|
+
"packed.receipt",
|
|
2784
|
+
"the receipt is not the head-bound terminal publication",
|
|
2785
|
+
)
|
|
2786
|
+
if (
|
|
2787
|
+
receipt.get("outputs")
|
|
2788
|
+
!= {
|
|
2789
|
+
"head_sha256": installed.head_sha256,
|
|
2790
|
+
"journal_sha256": receipt["outputs"]["journal_sha256"],
|
|
2791
|
+
**_verified_receipt_outputs(installed),
|
|
2792
|
+
}
|
|
2793
|
+
or installed.head["counters"] != receipt["counters"]
|
|
2794
|
+
or installed.head["backends"] != receipt.get("backends")
|
|
2795
|
+
or installed.head["posting_backend_coordinate"] != receipt.get("posting_backend_coordinate")
|
|
2796
|
+
or installed.head["provider_id"] != receipt.get("provider_id")
|
|
2797
|
+
):
|
|
2798
|
+
raise _refuse(
|
|
2799
|
+
"PACKED_RECEIPT_MISMATCH",
|
|
2800
|
+
"packed.receipt.outputs",
|
|
2801
|
+
"the receipt does not describe the generation installed at this root",
|
|
2802
|
+
)
|