mostlyright-data 0.25.4__tar.gz → 0.25.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/PKG-INFO +15 -4
  2. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/README.md +14 -3
  3. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/pyproject.toml +1 -1
  4. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/SKILL.md +7 -4
  5. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/3-probe-read-a-source-before-committing-to-it.md +8 -8
  6. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/6-build-one-run-sized-to-acquire-every-measured-source-whole.md +27 -27
  7. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/commands.md +6 -6
  8. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/live-run.md +12 -4
  9. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4.py +14 -2
  10. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_query.py +94 -11
  11. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_runs.py +76 -14
  12. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_tables.py +2 -1
  13. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/vocabulary.py +5 -2
  14. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/.gitignore +0 -0
  15. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/scripts/hatch_build.py +0 -0
  16. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/agents/openai.yaml +0 -0
  17. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/1-open-the-page-and-the-link-to-it-in-the-first-message.md +0 -0
  18. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/2-brief-two-to-four-questions-each-with-a-recommended-answer.md +0 -0
  19. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/4-decide-say-what-you-chose-what-you-refused-and-ask-one-question.md +0 -0
  20. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/5-draft-one-recipe-document-one-call.md +0 -0
  21. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/7-interrogate-ask-the-run-what-it-actually-delivered.md +0 -0
  22. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/8-fix-revise-the-document-and-register-it-again.md +0 -0
  23. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/9-present-only-what-survived-inspection-with-caveats.md +0 -0
  24. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/agent-protocol.md +0 -0
  25. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/autonomous-delivery.md +0 -0
  26. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/before-the-first-tool-call.md +0 -0
  27. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/boundaries.md +0 -0
  28. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/cloud-authentication-preflight.md +0 -0
  29. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/cross-repository-protocol-reference.md +0 -0
  30. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/installation-parity.md +0 -0
  31. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/narrating-the-run.md +0 -0
  32. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/not-hosted-yet.md +0 -0
  33. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/one-install.md +0 -0
  34. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/prediction-labels.md +0 -0
  35. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/promote.md +0 -0
  36. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/readers.md +0 -0
  37. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/receipts.md +0 -0
  38. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/recording-a-stream-venue.md +0 -0
  39. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/recovering-an-import-failure.md +0 -0
  40. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/reference-pages.md +0 -0
  41. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/required-protocol.md +0 -0
  42. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/source-credentials.md +0 -0
  43. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/sources.md +0 -0
  44. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/the-one-thing-to-say-about-the-skill-itself.md +0 -0
  45. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/transforms.md +0 -0
  46. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/user-communication-contract.md +0 -0
  47. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/references/writing-a-decision-record.md +0 -0
  48. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/skills/mr-data-build/scripts/write_research_notebook.py +0 -0
  49. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/__init__.py +0 -0
  50. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/agent_protocol.py +0 -0
  51. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/canonical.py +0 -0
  52. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/formats.py +0 -0
  53. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/hosted_crawler_protocol.py +0 -0
  54. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/key_seam.py +0 -0
  55. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/page_coverage.py +0 -0
  56. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/part_check_evidence.py +0 -0
  57. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/session_probes.py +0 -0
  58. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/skill_assets.py +0 -0
  59. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/table_manifest.py +0 -0
  60. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/__init__.py +0 -0
  61. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/acquire.py +0 -0
  62. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/acquire_cancel.py +0 -0
  63. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/activity.py +0 -0
  64. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/approvals.py +0 -0
  65. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/categories.py +0 -0
  66. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/commands.py +0 -0
  67. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/dataset-categories-v1.json +0 -0
  68. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/download.py +0 -0
  69. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/narrative.py +0 -0
  70. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/parity.py +0 -0
  71. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/probe.py +0 -0
  72. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/progress_vocabulary.py +0 -0
  73. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/propose.py +0 -0
  74. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/recipe.py +0 -0
  75. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/recipe_brief.py +0 -0
  76. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/recipe_lint.py +0 -0
  77. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/research.py +0 -0
  78. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/router.py +0 -0
  79. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/runs.py +0 -0
  80. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/session.py +0 -0
  81. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/stream.py +0 -0
  82. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/stream_venue.py +0 -0
  83. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/transport.py +0 -0
  84. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/user_agent.py +0 -0
  85. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_artifacts.py +0 -0
  86. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_catalog.py +0 -0
  87. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_connections.py +0 -0
  88. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_dataset_covers.py +0 -0
  89. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_datasets.py +0 -0
  90. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_handoff.py +0 -0
  91. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_narrative.py +0 -0
  92. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_reader.py +0 -0
  93. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_secrets.py +0 -0
  94. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/thin/v4_stream.py +0 -0
  95. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/__init__.py +0 -0
  96. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/attendance.py +0 -0
  97. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/clarification.py +0 -0
  98. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/cloud_auth.py +0 -0
  99. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/commands/__init__.py +0 -0
  100. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/commands/auth.py +0 -0
  101. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/commands/clarify.py +0 -0
  102. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/commands/login.py +0 -0
  103. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/commands/whoami.py +0 -0
  104. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/credential_native.py +0 -0
  105. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/credential_store.py +0 -0
  106. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/credentials.py +0 -0
  107. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/login.py +0 -0
  108. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/path_kind.py +0 -0
  109. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/plain_file.py +0 -0
  110. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/remediation.py +0 -0
  111. {mostlyright_data-0.25.4 → mostlyright_data-0.25.6}/src/mostlyright/data_harness/ux/render.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyright-data
3
- Version: 0.25.4
3
+ Version: 0.25.6
4
4
  Summary: Mostly Right hosted CLI for reviewed datasets
5
5
  Project-URL: Homepage, https://mostlyright.md/
6
6
  Project-URL: Documentation, https://mostlyright.md/docs/guides/cli/
@@ -82,6 +82,17 @@ mr-data checks RUN_ID
82
82
  mr-data download RUN_ID --output ./out
83
83
  ```
84
84
 
85
+ Full builds are progressive by default: one acquisition exposes an inspection checkpoint after
86
+ about five minutes and continues toward the finished table. Inspect it with
87
+ `mr-data status RUN_ID --inspection --json`; the checkpoint does not pause for approval or start
88
+ another full acquisition. Any required spend confirmation happens before acquisition.
89
+ When a run has sealed queryable table checkpoints, `mr-data status RUN_ID --checkpoints --json`
90
+ lists their exact immutable identifiers and `mr-data query RUN_ID "SELECT …" --checkpoint
91
+ CHECKPOINT_ID` reads one without advancing underneath the query. These workspace-private snapshots
92
+ remain incomplete, keep checks pending, and never become the serving table.
93
+ `--progressive` remains a compatibility alias. Use `--sample` only for an explicit standalone
94
+ bounded inspection; row ceilings do not impose a five-minute acquisition limit.
95
+
85
96
  Run submission is idempotent: sending identical recipe, mode and bounds returns the same run,
86
97
  including a previous failure. The CLI reports a non-queued result as `run_returned` and exits
87
98
  with code 2 when that run failed. Read its failure before retrying; `mr-data run --retry RUN_ID`
@@ -94,9 +105,9 @@ falls back to a mutable whole-source fetch. `mr-data table resync TABLE_ID --req
94
105
  explicitly requests a full source reread when that is needed. Choose and retain the UUID before
95
106
  submitting: if the response is lost, repeat that exact command with the same UUID rather than
96
107
  starting another reread. A persisted spend hold is still an accepted resync: its receipt names
97
- the run, projection and exact `mr-data run --confirm-held RUN_ID` action. If it is a sample-first
98
- preview, confirm that preview first, then release its linked full
99
- only after the preview succeeds. `mr-data recipe readiness` reads Studio's paginated,
108
+ the run, projection and exact `mr-data run --confirm-held RUN_ID` action. New full resyncs use
109
+ progressive acquisition too. Existing legacy sample-first pairs can still finish through their
110
+ original confirmation and approval actions. `mr-data recipe readiness` reads Studio's paginated,
100
111
  immutable-revision inventory before a production-wide schedule sweep: predecessor,
101
112
  incremental-materialization and differential-proof readiness, together with each source's next
102
113
  action. It reports Studio's persisted facts; it does not authorize a schedule.
@@ -70,6 +70,17 @@ mr-data checks RUN_ID
70
70
  mr-data download RUN_ID --output ./out
71
71
  ```
72
72
 
73
+ Full builds are progressive by default: one acquisition exposes an inspection checkpoint after
74
+ about five minutes and continues toward the finished table. Inspect it with
75
+ `mr-data status RUN_ID --inspection --json`; the checkpoint does not pause for approval or start
76
+ another full acquisition. Any required spend confirmation happens before acquisition.
77
+ When a run has sealed queryable table checkpoints, `mr-data status RUN_ID --checkpoints --json`
78
+ lists their exact immutable identifiers and `mr-data query RUN_ID "SELECT …" --checkpoint
79
+ CHECKPOINT_ID` reads one without advancing underneath the query. These workspace-private snapshots
80
+ remain incomplete, keep checks pending, and never become the serving table.
81
+ `--progressive` remains a compatibility alias. Use `--sample` only for an explicit standalone
82
+ bounded inspection; row ceilings do not impose a five-minute acquisition limit.
83
+
73
84
  Run submission is idempotent: sending identical recipe, mode and bounds returns the same run,
74
85
  including a previous failure. The CLI reports a non-queued result as `run_returned` and exits
75
86
  with code 2 when that run failed. Read its failure before retrying; `mr-data run --retry RUN_ID`
@@ -82,9 +93,9 @@ falls back to a mutable whole-source fetch. `mr-data table resync TABLE_ID --req
82
93
  explicitly requests a full source reread when that is needed. Choose and retain the UUID before
83
94
  submitting: if the response is lost, repeat that exact command with the same UUID rather than
84
95
  starting another reread. A persisted spend hold is still an accepted resync: its receipt names
85
- the run, projection and exact `mr-data run --confirm-held RUN_ID` action. If it is a sample-first
86
- preview, confirm that preview first, then release its linked full
87
- only after the preview succeeds. `mr-data recipe readiness` reads Studio's paginated,
96
+ the run, projection and exact `mr-data run --confirm-held RUN_ID` action. New full resyncs use
97
+ progressive acquisition too. Existing legacy sample-first pairs can still finish through their
98
+ original confirmation and approval actions. `mr-data recipe readiness` reads Studio's paginated,
88
99
  immutable-revision inventory before a production-wide schedule sweep: predecessor,
89
100
  incremental-materialization and differential-proof readiness, together with each source's next
90
101
  action. It reports Studio's persisted facts; it does not authorize a schedule.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "mostlyright-data"
3
- version = "0.25.4"
3
+ version = "0.25.6"
4
4
  description = "Mostly Right hosted CLI for reviewed datasets"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -62,13 +62,16 @@ references. Load the reference for the current stage only, not the entire librar
62
62
  source's own address proves, and name the `resync_only` sources and the reason in the build
63
63
  message.
64
64
  6. Read [build sizing](references/6-build-one-run-sized-to-acquire-every-measured-source-whole.md).
65
- Choose bounds from measured source size; row and byte clamps apply per source. Use a bounded
66
- preview for unmeasured sources. Preserve the agreed semantics and quality checks.
65
+ Use `--full` for the requested build: it is progressive by default, with a five-minute
66
+ inspection checkpoint while the same acquisition continues. Choose bounds from measured
67
+ source size; row and byte clamps apply per source, not elapsed time. Use explicit `--sample`
68
+ only for a deliberate standalone inspection. Preserve the agreed semantics and quality checks.
67
69
  7. Follow the returned run with `mr-data watch RUN_ID --json`. Read `run.status`: returning an
68
70
  existing run does not mean new work was queued. Resume the known run after an interruption,
69
71
  using the saved cursor where available. Read [run lifecycle](references/live-run.md) for
70
- holds and continuation commands. Spend confirmation and releasing a held full are distinct
71
- actions. Honor the user's authorization; do not request an existing delegation again. Send a
72
+ holds and continuation commands. Confirm projected spend before acquisition when required;
73
+ the progressive checkpoint needs no approval or second full run. Honor the user's
74
+ authorization; do not request an existing delegation again. Send a
72
75
  concise chat update for each newly observed meaningful lifecycle transition: accepted or
73
76
  queued, released after confirmation, started, failed, retried or replaced, completed, and
74
77
  verified. Include the run ID so the user can tell attempts apart; state the supported failure
@@ -86,22 +86,22 @@ happens AFTER stage 2's answers arrive — or at once under a delegation. What c
86
86
  wait is research that submits nothing: documentation, publisher comparison, catalog.
87
87
 
88
88
  ```sh
89
- mr-data run --recipe RECIPE_ID --digest RECIPE_DIGEST --sample --max-rows 400000 --json
89
+ mr-data run --recipe RECIPE_ID --digest RECIPE_DIGEST --full --json
90
+ mr-data status RUN --inspection --json # inspect while the same full acquisition continues
90
91
  mr-data peek RUN --json # the columns and types the run sealed
91
92
  mr-data query RUN "select count(*) from run_table" --json
92
93
  mr-data receipt RUN --json # what was fetched and what those bytes hashed to
93
94
  ```
94
95
 
95
- **Size it to measure rather than to sip.** A ceiling under the real size reports the ceiling back
96
- rather than the source: every larger feed comes back at exactly that number with `truncated` true,
97
- and a stage-6 ceiling set above THAT is a number this run invented. Set it high enough that the
98
- feeds come back whole, then keep the row counts off `coverage` — a source still truncated here is
99
- a source nobody measured, the one reason stage 6 falls back to a preview.
96
+ **Inspect the full build while it runs.** Its progressive checkpoint appears after about five
97
+ minutes without stopping acquisition. Use an explicit `--sample` only for a deliberate standalone
98
+ experiment: a row ceiling can truncate output after every collection page was fetched and does
99
+ not promise a short acquisition. Never derive source size from a truncated sample's ceiling.
100
100
 
101
101
  **A reading run that came back whole IS the build, so stage 6 adopts it rather than repeating
102
102
  it.** Untruncated everywhere is the build, whichever stage started the run: keep this `run_id`.
103
- Nor is the ceiling above sneaked past anybody — this run meets the same spend confirmation and
104
- sample-first threshold as every other, so a large one is held rather than quietly spent.
103
+ The run meets the ordinary spend confirmation before acquisition; its checkpoint requires no
104
+ second approval or second full build.
105
105
 
106
106
  That is stages 5 through 7 run once to settle the shape. Where the deployment does answer a probe,
107
107
  keep its status, reason code and exit value internal and describe only what it settled.
@@ -5,28 +5,25 @@
5
5
  nothing here. Re-running the coordinate replays it; a different ceiling buys only another
6
6
  acquisition.
7
7
 
8
- **Otherwise the first run is the build, and is sized to be one.** Stage 3 opened these feeds and
9
- knows roughly how many rows each carries. Set the row ceiling comfortably above the LARGEST and
10
- every source is acquired whole: no artificial seams, every declared check meaningful, and
11
- `coverage.truncated` false on every entry — which makes this finished work rather than a shape
12
- test.
8
+ **Otherwise start one progressive full build.** Use the registered recipe and measured bounds
9
+ for the requested coverage. Every full run is progressive by default; agents do not choose
10
+ between progressive and sample-first acquisition.
13
11
 
14
12
  ```sh
15
- mr-data run --recipe RECIPE_ID --digest RECIPE_DIGEST --sample --max-rows 400000 --json
13
+ mr-data run --recipe RECIPE_ID --digest RECIPE_DIGEST --full --json
16
14
  ```
17
15
 
18
- `sample` is a machine word for the run MODE and is not what the run is called. When `truncated` is
19
- false on every coverage entry, this is **the build**: the table a full run would have sealed,
20
- already live, presented as the build at stage 9. Do not run it again in another mode to change the
21
- word on it — the record keeps `mode: sample` for ever either way, and what makes the table finished
22
- is its coverage. Where somebody reads that word off the page and asks, say so: the mode is how the
23
- run was bounded, and nothing was cut.
16
+ After about five minutes, inspect `mr-data status RUN_ID --inspection --json` while the same
17
+ acquisition continues. This checkpoint is read-only, not a second approval gate or a hard runtime
18
+ limit. A short build can finish before a checkpoint is needed. Do not submit another full run
19
+ after inspecting it. Required spend confirmation happens before acquisition. The older
20
+ `--progressive` flag is accepted for compatibility but is unnecessary.
24
21
 
25
- **Two thousand rows is only for a source nobody measured.** Where stage 3 could not open a feed and
26
- its size is a guess, `--max-rows 2000` exercises every column, cast and declared check and comes
27
- back while the person who asked is still watching. A run bounded that way is a **preview**, not the
28
- build, and is presented as one: say which source stopped short and at which row — both are on
29
- `coverage` — what a full build would cover, and ask whether to run it.
22
+ **A standalone sample is an explicit inspection choice.** Use `--sample --max-rows 2000` only
23
+ when a separate bounded experiment is needed. Rows are clamped after acquisition, so a collection
24
+ may still fetch every page and take longer than five minutes. For an already authorized full
25
+ build, use the progressive checkpoint. A truncated sample is a preview: report its actual
26
+ coverage and the full build's proposed coverage before requesting any missing authorization.
30
27
 
31
28
  > The preview stopped at 2,000 rows on `asos_observations`, which is the ceiling rather than the
32
29
  > end of the feed; the other three sources came back whole. The full build would cover 2000 to 2026
@@ -39,22 +36,19 @@ ask again: state the projection and run it. A narrower delegation does not autho
39
36
  mr-data run --recipe RECIPE_ID --digest RECIPE_DIGEST --full --json
40
37
  ```
41
38
 
42
- **A held full is released with one command.** Over the sample-first threshold Studio does not start
43
- the full: it starts a bounded preview and holds the linked full at `awaiting_sample_approval`. Read
44
- `sample_run_id` off that status, interrogate that preview as stage 7 says, and present it as one.
45
- Then, on the user's word or under a delegation:
39
+ **Existing legacy pairs can finish.** New full runs never create a linked sample/full pair.
40
+ If adopting a legacy full already at `awaiting_sample_approval`, read `sample_run_id`, verify
41
+ that preview, and release the held full under the user's authorization:
46
42
 
47
43
  ```sh
48
44
  mr-data run --approve-full RUN_ID --json
49
45
  ```
50
46
 
51
47
  `RUN_ID` may be either half of the pair: the held full, or the preview, whose `full_run_id` this
52
- command follows. It reads the run for its version and releases it under that version, so a pair
53
- that moved is refused rather than released against a record you did not read. **A deployment that
54
- still wants a person at a browser answers `THIN_INTERACTIVE_HUMAN_REQUIRED`, and today that is the
55
- usual answer** — Studio's own change is what makes it succeed. That is a typed refusal with a human
56
- fallback, never a crash: say the release needs them on the run page, give them the address the
57
- refusal names, and stop. Do not offer `--confirm`; it cannot settle this state.
48
+ command follows. It releases under the version read. If a deployment returns
49
+ `THIN_INTERACTIVE_HUMAN_REQUIRED`, report its named browser action. Do not offer `--confirm`;
50
+ it cannot settle this legacy state. An existing untruncated sample that already satisfies the
51
+ brief remains the completed build; never acquire it again merely to change its mode label.
58
52
 
59
53
  A stated `--window` may be refused before acquisition; report the typed code it answers with
60
54
  rather than retrying the same run in another mode.
@@ -84,6 +78,12 @@ attempt. Do not repeat either update while polling the same state.
84
78
  mr-data run --retry RUN_ID --json
85
79
  ```
86
80
 
81
+ For a failed progressive full, this also names the failed run as the capture ancestor, letting
82
+ Studio reuse compatible sealed pages. For a cancelled progressive full, or a failure outside
83
+ the supported room-fault retry set, start an authorized replacement with
84
+ `--full --resume-capture-run-id RUN_ID` and the registered recipe and digest. Legacy full runs
85
+ do not provide progressive captures.
86
+
87
87
  It re-states that run's own recipe, digest, mode and ceilings under one byte ceiling above what
88
88
  the recipe declares, because Studio replays a terminal result for an identical command and the
89
89
  coordinate has to move — and the number that ceiling must clear is what the run is projected to
@@ -14,7 +14,7 @@ gap rather than doing anything, and `export-hosted-candidate`, which is a backen
14
14
  | `mr-data dataset` | Bring the dataset page into existence before there is anything on it, then fill it in while somebody watches. `dataset create --name TEXT` mints it and prints the `dataset_id`; `dataset show ID` reads it back; `dataset set ID --name TEXT --topics "a,b,c" --license ID --description-file F` writes the title, the descriptive tags, the SPDX licence and the description under the version it was read at, retrying once if somebody else wrote first, and an empty `--topics` or `--license` takes that value off the page; a saved write whose public sync fails exits 2 and reports `public_projection_synced: false` — use `dataset sync ID` to retry that sync without rewriting Studio; `dataset note ID --heading H --blocks-file B` writes one cell of the decision record that OUTLIVES every run, and `--list` reads it back; `dataset watch ID` follows the page's own event stream; `dataset activity ID --phase P --message TEXT` says what is happening right now, silently, and is never a chat message and never a cell; `dataset publish ID [--mode public|link|private]` says who can read the dataset — `public` lists it in the public directory and serves it at an address anybody can read, `link` serves it at an unlisted address, `private` takes it back to the workspace — and `dataset publish ID --show` reads that back without changing it; `dataset archive ID --confirm-name TITLE` retires the page and frees its title, deleting nothing. |
15
15
  | `mr-data recipe` | Register one recipe document — dataset, question, table plan, sources, transform, checks and units — in one call, and print the identifiers the server derived. The document is read first and every fault comes back in one refusal, each with the JSON pointer that names it; `--no-lint` sends it as written instead. Registration refuses `THIN_BRIEF_MISSING` when the dataset's record carries no answered question and no delegation, and `THIN_RECIPE_DATASET_UNBOUND` when the document names no `dataset.id`; `--delegated "their words"` writes the delegation onto the dataset and registers in the same command, and clears the first of those two and never the second. `mr-data recipe show ID` reads one back. |
16
16
  | `mr-data cover` | Generate and attach one branded 1200×630 dataset cover. Give it the dataset ID and what the image should depict; Studio fixes the model, single-color style, curated random palette, dimensions and storage. |
17
- | `mr-data run` | Start one run against a registered recipe, named as `--recipe RECIPE_ID --digest RECIPE_DIGEST` — both required, neither positional, and the digest is the bare hex the registration receipt printed. The mode is one of six: `sample`, `full`, `refresh`, `backfill`, `compact` and `replay`. Four have a shorthand flag (`--sample`, `--full`, `--refresh`, `--backfill`) and two do not, so write `--mode MODE`, which is accepted for every one of them. A refresh executes only Studio's persisted strict source-action plan. Although the classifier may name closed reuse, a request-window or collection delta, or recorded input, current normal refresh materializes only a plan carrying at least one direct `partition_replace` request window with a compatible predecessor — `closed` siblings may reuse their sealed bytes beside it, a lone window needs a retained source-history record, and several must agree on materialization and partition layout — or an all-recorded-stream plan. Closed-only, any plan carrying a collection, a window beside a recorded stream, snapshot and unwindowed plans return `RESYNC_REQUIRED` before acquisition. Use `mr-data table resync TABLE_ID --request-id UUID` for an explicit full source reread, retaining that UUID if the response is uncertain. `--backfill` states the exact window with `--window START END`. `--mode replay --sources-from RUN_ID` asks Studio to run the registered revision against the RETAINED RAW INPUTS of one named successful run of the same table: nothing is fetched from this computer, nothing is acquired again, the result is compared against that run and never becomes the live version, and it neither asks for nor records an approval. Where Studio has replay switched off, or the named run is not successful, not the same table, or no longer retains its inputs, it answers a typed refusal — report the code rather than retrying in another mode. `--sources-from` on any other mode is refused `THIN_ARGUMENT_INVALID` before anything is sent. `--max-rows` and `--max-source-bytes` bound ONE SOURCE rather than the finished table. A large run is held for a spend confirmation, which `--confirm` settles and whose printed `confirm_command` re-states every argument the request carried. `--approve-full RUN_ID` releases the full a sample-first pair is holding once its preview has succeeded, naming either half of the pair; where the deployment still wants a person at a browser it answers `THIN_INTERACTIVE_HUMAN_REQUIRED` and names the run page. `--retry RUN_ID` tries one failed run again when its triple carries `room_fault: true`, re-stating that run's own coordinate under a byte ceiling that cannot narrow it, and refuses with the triple when the recipe was at fault. `--cancel RUN` stops one. |
17
+ | `mr-data run` | Start one run against a registered recipe, named as `--recipe RECIPE_ID --digest RECIPE_DIGEST` — both required, neither positional, and the digest is the bare hex the registration receipt printed. The mode is one of six: `sample`, `full`, `refresh`, `backfill`, `compact` and `replay`. Four have a shorthand flag (`--sample`, `--full`, `--refresh`, `--backfill`) and two do not, so write `--mode MODE`, which is accepted for every one of them. A refresh executes only Studio's persisted strict source-action plan. Although the classifier may name closed reuse, a request-window or collection delta, or recorded input, current normal refresh materializes only a plan carrying at least one direct `partition_replace` request window with a compatible predecessor — `closed` siblings may reuse their sealed bytes beside it, a lone window needs a retained source-history record, and several must agree on materialization and partition layout — or an all-recorded-stream plan. Closed-only, any plan carrying a collection, a window beside a recorded stream, snapshot and unwindowed plans return `RESYNC_REQUIRED` before acquisition. Use `mr-data table resync TABLE_ID --request-id UUID` for an explicit full source reread, retaining that UUID if the response is uncertain. `--backfill` states the exact window with `--window START END`. `--mode replay --sources-from RUN_ID` asks Studio to run the registered revision against the RETAINED RAW INPUTS of one named successful run of the same table: nothing is fetched from this computer, nothing is acquired again, the result is compared against that run and never becomes the live version, and it neither asks for nor records an approval. Where Studio has replay switched off, or the named run is not successful, not the same table, or no longer retains its inputs, it answers a typed refusal — report the code rather than retrying in another mode. `--sources-from` on any other mode is refused `THIN_ARGUMENT_INVALID` before anything is sent. `--max-rows` and `--max-source-bytes` bound ONE SOURCE rather than the finished table. A large run is held for a spend confirmation, which `--confirm` settles and whose printed `confirm_command` re-states every argument the request carried. `--approve-full RUN_ID` releases the full an existing legacy sample-first pair is holding once its preview has succeeded, naming either half of the pair; where the deployment still wants a person at a browser it answers `THIN_INTERACTIVE_HUMAN_REQUIRED` and names the run page. `--retry RUN_ID` tries one failed run again when its triple carries `room_fault: true`, re-stating that run's own coordinate under a byte ceiling that cannot narrow it, and refuses with the triple when the recipe was at fault. `--cancel RUN` stops one. |
18
18
  | `mr-data status` | Report which of the seven states one run is in, what it delivered and whether a ceiling cut it short, and — on a failure — what failed, where, and whether the fault was the execution room's rather than the recipe's. Exits non-zero on a failed run. |
19
19
  | `mr-data runs` | Report this workspace's own runs: identifier, mode, state and creation time, plus the failure triple of any that failed and whether that failure was the execution room's. `--status` and `--mode` narrow it and are refused `THIN_ARGUMENT_INVALID` for a word outside the seven states or the six modes; `--limit` says how many, and pages are followed to reach it. A narrow filter over a workspace of thousands of runs reads a long way to find its matches, and a listing that does not end says how many were read and suggests a smaller `--limit`. |
20
20
  | `mr-data watch` | Stream one run's live progress under the durable event type of each stage, resuming across stream cuts. Exits non-zero when the run failed. |
@@ -38,14 +38,14 @@ gap rather than doing anything, and `export-hosted-candidate`, which is a backen
38
38
  | `mr-data unpin` | Ask Studio to resume pointer tracking; read the returned state before claiming it resumed. |
39
39
  | `mr-data demote` | Ask Studio to withdraw the pointer and any schedule it owns, for one table or for as many as you name: `demote TABLE [TABLE ...]` withdraws them one after another, attempts every one of them whatever the one before it answered, prints a line for each, and exits non-zero if any is still live. Pair it with `table archive` when you are retiring a set: a live table cannot be archived, so it is withdraw-then-archive, two commands rather than a loop. |
40
40
 
41
- For a large full build on a deployment that serves progressive runs, add `--progressive` to
42
- `mr-data run --mode full`. Studio admits one full run and keeps acquisition moving while the
41
+ Every `mr-data run --full` or `--mode full` build is progressive by default. The old
42
+ `--progressive` flag remains a compatibility alias. Studio admits one full run and keeps acquisition moving while the
43
43
  owner inspects its five-minute checkpoint. A spend confirmation, when required, still comes
44
44
  before acquisition. Read `mr-data status RUN_ID --inspection --json` for the declared columns,
45
45
  missing-data policy, every source's bounded progress, and any observed table evidence. Source
46
46
  relation rows are not finished table rows; unknown transformed values remain null. A failed or
47
47
  cancelled progressive run may be followed by a new
48
- `--progressive --mode full --resume-capture-run-id RUN_ID` request under a registered compatible
48
+ `--mode full --resume-capture-run-id RUN_ID` request under a registered compatible
49
49
  recipe. Studio decides which sealed captures can be reused. Keep the returned new run ID and
50
50
  verify its final receipt.
51
51
 
@@ -64,8 +64,8 @@ must agree on materialization and partition layout. Closed-only, any plan carryi
64
64
  window beside a recorded stream, snapshot and unwindowed plans return `RESYNC_REQUIRED` before
65
65
  acquisition. `mr-data table resync TABLE --request-id UUID` is the explicit
66
66
  full reread; retain and reuse the caller-generated UUID after a lost response. A held resync receipt
67
- prints `mr-data run --confirm-held RUN_ID` for that exact run. A sample-first resync instead uses
68
- `--approve-full` only after its preview succeeds.
67
+ prints `mr-data run --confirm-held RUN_ID` for that exact run. New full resyncs are progressive;
68
+ an existing legacy sample-first resync still uses `--approve-full` after its preview succeeds.
69
69
 
70
70
  **`--refresh` states its own range, and a run that starts one by hand must pass it.** `mr-data run
71
71
  --refresh` sends what it is given. A refresh of a recipe carrying a non-snapshot request window
@@ -50,14 +50,22 @@ what it proved, and a run that emitted none produces the same build as one that
50
50
  them. Never turn a progress count into a percentage, never infer activity from elapsed time, and
51
51
  never cite a progress event where a receipt is what was asked for.
52
52
 
53
- **Seven states, one machine.** A run is `queued`, `running`, `awaiting_confirmation`,
53
+ **Full builds are progressive.** `--full` and `--mode full` acquire once and expose a checkpoint
54
+ after about five minutes while that same run continues. Read it with
55
+ `mr-data status RUN_ID --inspection --json`. It is an inspection point, not a five-minute hard
56
+ timeout or an approval gate. Do not submit another full run after inspecting it. A short build
57
+ can finish before a checkpoint is needed. `--sample` remains an explicit standalone bounded run;
58
+ its row ceiling does not bound acquisition time.
59
+
60
+ **Seven states, including legacy holds.** A run is `queued`, `running`, `awaiting_confirmation`,
54
61
  `awaiting_sample_approval`, `succeeded`, `failed` or `cancelled`; three are terminal and end a
55
62
  watch. `awaiting_confirmation` is the ordinary spend gate: nothing has been executed, and the same
56
63
  run command with `--confirm` on the end of it releases it. `awaiting_sample_approval` is the held
57
- full of a sample-first pair: its bounded preview must seal a downloadable table first, and then
64
+ full of an existing legacy sample-first pair: its bounded preview must seal a downloadable table
65
+ first, and then
58
66
  `mr-data run --approve-full RUN_ID` releases it — `--confirm` cannot, and must never be offered as
59
- though it could. Keep the raw state internal: say what product decision is needed and give the
60
- person the single action that lets the build continue.
67
+ though it could. New full builds never create these pairs. Keep the raw state internal: say what
68
+ product decision is needed and give the person the single action that lets the build continue.
61
69
 
62
70
  **`build_scope` says what the build covers, not what it is doing.** A `failed` or `cancelled` run
63
71
  covers nothing and says so. A `succeeded` one reports what `coverage.truncated` recorded. Anything
@@ -119,8 +119,8 @@ from mostlyright.data_harness.thin.transport import ThinLaneError
119
119
  #: ``JOB_INVALID`` outright, which in production was every refresh a collection epoch was offered
120
120
  #: for rather than only the collection runs. This package must not reach a Studio older than the
121
121
  #: commit it pins; ``docs/V4-WORKER-PROTOCOL.md`` states it beside the layout's own ordering rule.
122
- PINNED_V4_OPENAPI_SOURCE_SHA256 = "0cfa1dda766dfc8861286a3cf8bfa5a86cd71d396f49b134882a6e8c95e9ca50"
123
- PINNED_V4_CONTRACT_VERSION = "4.10.1"
122
+ PINNED_V4_OPENAPI_SOURCE_SHA256 = "2f4e1f566bfd788ba1f9981d8b214c62e43294320da87167ced462a527abe848"
123
+ PINNED_V4_CONTRACT_VERSION = "4.12.0"
124
124
 
125
125
  # --------------------------------------------------------------------------------------------
126
126
  # The routes
@@ -215,6 +215,14 @@ RUN_PROGRESS_PATH = "/v4/runs/{run_id}/progress"
215
215
  #: ``GET`` -- declared shape and bounded observed evidence for a progressive full run.
216
216
  RUN_INSPECTION_PATH = "/v4/runs/{run_id}/inspection"
217
217
 
218
+ #: ``GET`` -- the bounded list of immutable, workspace-private snapshots this run has sealed.
219
+ RUN_CHECKPOINTS_PATH = "/v4/runs/{run_id}/checkpoints"
220
+
221
+ #: ``GET`` -- one exact immutable build checkpoint. ``POST :resolve`` on the same path belongs to
222
+ #: Cloud's query executor; the thin client queries through the ordinary run query route instead
223
+ #: of receiving artifact storage coordinates.
224
+ RUN_CHECKPOINT_PATH = "/v4/runs/{run_id}/checkpoints/{checkpoint_id}"
225
+
218
226
  # Reader recovery uses Studio's versioned authority. These routes deliberately live beside the
219
227
  # other V4 wire constants so a client cannot silently turn transport canonicalisation into reader
220
228
  # validation.
@@ -603,6 +611,8 @@ DECLARED_V4_PATHS: frozenset[str] = frozenset(
603
611
  RUN_EVENTS_PATH,
604
612
  RUN_PROGRESS_PATH,
605
613
  RUN_INSPECTION_PATH,
614
+ RUN_CHECKPOINTS_PATH,
615
+ RUN_CHECKPOINT_PATH,
606
616
  # ⚠ PROMOTED OUT OF `PENDING_V4_PATHS` IN THE COMMIT THAT STARTED CALLING THEM. Studio
607
617
  # shipped the whole promotion surface while this client was being written against the
608
618
  # contract for it, and the ledger in `tests/test_thin_v4_contract.py` went red naming all
@@ -944,6 +954,8 @@ __all__ = [
944
954
  "RESCHEDULE_TABLE_PATH",
945
955
  "RESYNC_TABLE_PATH",
946
956
  "RUN_ARTIFACTS_PATH",
957
+ "RUN_CHECKPOINTS_PATH",
958
+ "RUN_CHECKPOINT_PATH",
947
959
  "RUN_EVENTS_PATH",
948
960
  "RUN_INSPECTION_PATH",
949
961
  "RUN_PROGRESS_PATH",
@@ -38,6 +38,7 @@ hop later.
38
38
  from __future__ import annotations
39
39
 
40
40
  import argparse
41
+ import re
41
42
  import time
42
43
  from collections.abc import Callable, Mapping
43
44
  from typing import Any
@@ -114,6 +115,8 @@ MASKED_SCAN_WARNING = (
114
115
  'export is refused; writing that identifier in "double quotes" puts it outside the scan'
115
116
  )
116
117
 
118
+ _CONTENT_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
119
+
117
120
  #: The four refusals that happen BEFORE anything is queued, and the fifth that bounds the body,
118
121
  #: each with the sentence this command adds to Studio's own message. Keyed by Studio's code, which
119
122
  #: is what ``thin.runs._studio_refusal`` carries through verbatim as ``THIN_STUDIO_{code}``.
@@ -199,7 +202,8 @@ class StudioV4QueryClient(StudioV4Client):
199
202
  """Queue one bounded read. ``202`` on a new query AND on a repeat of an identical one.
200
203
 
201
204
  ⚠ ``202`` IS THE ONLY ACCEPTED STATUS, and a repeat is not a second status. Studio derives
202
- the identifier from the digest of the canonical ``{sql, max_rows}`` command, so two
205
+ the identifier from the digest of the canonical command, including its immutable
206
+ checkpoint coordinate when present, so two
203
207
  identical submits converge on one query and the second answers ``202`` again with
204
208
  ``Idempotent-Replay: true``. Reading the header is what lets this command say which of the
205
209
  two happened; treating the replay as a fresh execution would tell an agent it had spent
@@ -295,7 +299,9 @@ def declare_arguments(parser: argparse.ArgumentParser) -> None:
295
299
  f"for at most {int(MAX_QUERY_WAIT_SECONDS)} seconds, whichever comes first -- it is never "
296
300
  "a long poll. A wait that ends first leaves the question running and prints its "
297
301
  "identifier; running the same command again reads that same one back, because the "
298
- "identifier is derived from the statement, the row limit and the prune block. "
302
+ "identifier is derived from the statement, the row limit, the prune block and the "
303
+ "optional checkpoint. A checkpoint is already one exact immutable snapshot, so it "
304
+ "cannot be combined with --field, --from, --to or --partition. "
299
305
  "A table version is many Parquet parts and a bounded read may open only so many of "
300
306
  "them, so a question about a wide table is narrowed with --field plus --from/--to, or "
301
307
  "with --partition; Studio evaluates that against the part list before a worker is woken, "
@@ -317,6 +323,13 @@ def declare_arguments(parser: argparse.ArgumentParser) -> None:
317
323
  help="stop at this many rows; left out, the server's own ceiling applies. A number above "
318
324
  "that ceiling is refused rather than lowered",
319
325
  )
326
+ parser.add_argument(
327
+ "--checkpoint",
328
+ metavar="CHECKPOINT_ID",
329
+ help="ask one immutable build checkpoint instead of the run's final table. The result "
330
+ "stays bound to this exact checkpoint while newer build data arrives; it cannot be "
331
+ "combined with --field, --from, --to or --partition",
332
+ )
320
333
  parser.add_argument(
321
334
  "--field",
322
335
  help="which field --from and --to are about, by column name. Without --from or --to it "
@@ -369,10 +382,10 @@ def _client(args: argparse.Namespace) -> StudioV4QueryClient:
369
382
  def submit_body(args: argparse.Namespace) -> dict[str, Any]:
370
383
  """The ``submit_query_command`` this invocation means, checked before anything is sent.
371
384
 
372
- Two members, because Studio decides everything else: the run names the table, the bearer names
373
- the workspace, and the deadline is deployment policy. ``max_rows`` is left OUT rather than sent
374
- as ``null`` when the caller states none -- "as many as the server allows" is a member absent,
375
- and a null would be refused by a command schema that admits two properties and no nulls.
385
+ The run names the table, the bearer names the workspace, and the deadline is deployment
386
+ policy. ``max_rows`` and ``checkpoint_id`` are left OUT rather than sent as ``null`` when the
387
+ caller states none. An absent checkpoint asks the final run table; an exact checkpoint keeps
388
+ the answer stable while later build data arrives.
376
389
  """
377
390
 
378
391
  statement = str(getattr(args, "sql", "") or "")
@@ -382,7 +395,16 @@ def submit_body(args: argparse.Namespace) -> dict[str, Any]:
382
395
  'mr-data query needs the question to ask: mr-data query RUN "SELECT ..."',
383
396
  )
384
397
  body: dict[str, Any] = {"sql": statement}
398
+ checkpoint = getattr(args, "checkpoint", None)
399
+ if checkpoint is not None:
400
+ body["checkpoint_id"] = identifier(str(checkpoint), "checkpoint identifier")
385
401
  prune = _prune(args)
402
+ if checkpoint is not None and prune:
403
+ raise ThinLaneError(
404
+ "THIN_REQUEST_INVALID",
405
+ "--checkpoint already names one exact immutable snapshot and cannot be combined "
406
+ "with --field, --from, --to or --partition",
407
+ )
386
408
  if prune:
387
409
  body["prune"] = prune
388
410
  max_rows = getattr(args, "max_rows", None)
@@ -464,6 +486,40 @@ def _record_identifier(record: Mapping[str, Any], run_id: str) -> str:
464
486
  return query_id
465
487
 
466
488
 
489
+ def _checkpoint_binding(
490
+ record: Mapping[str, Any],
491
+ checkpoint_id: str | None,
492
+ checkpoint_content_digest: str | None = None,
493
+ ) -> str | None:
494
+ """Validate the immutable coordinate on every submit and poll response."""
495
+
496
+ actual_id = record.get("checkpoint_id")
497
+ actual_digest = record.get("checkpoint_content_digest")
498
+ if checkpoint_id is None:
499
+ if actual_id is not None or actual_digest is not None:
500
+ raise ThinLaneError(
501
+ "THIN_RESPONSE_INVALID",
502
+ "Studio bound a final-table query to an unexpected build checkpoint",
503
+ )
504
+ return None
505
+ if actual_id != checkpoint_id:
506
+ raise ThinLaneError(
507
+ "THIN_RESPONSE_INVALID",
508
+ "Studio answered with a query bound to another or no build checkpoint",
509
+ )
510
+ if not isinstance(actual_digest, str) or _CONTENT_DIGEST.fullmatch(actual_digest) is None:
511
+ raise ThinLaneError(
512
+ "THIN_RESPONSE_INVALID",
513
+ "Studio answered a checkpoint query without its sealed content digest",
514
+ )
515
+ if checkpoint_content_digest is not None and actual_digest != checkpoint_content_digest:
516
+ raise ThinLaneError(
517
+ "THIN_RESPONSE_INVALID",
518
+ "Studio changed the checkpoint content digest while the query was running",
519
+ )
520
+ return actual_digest
521
+
522
+
467
523
  def _named_refusal(refusal: ThinLaneError) -> ThinLaneError:
468
524
  """One pre-queue refusal, with the sentence this command adds and the ``details`` it carries.
469
525
 
@@ -547,6 +603,8 @@ def _answer(record: Mapping[str, Any]) -> dict[str, Any]:
547
603
  "state": record.get("state"),
548
604
  "sql": record.get("sql"),
549
605
  "max_rows": record.get("max_rows"),
606
+ "checkpoint_id": record.get("checkpoint_id"),
607
+ "checkpoint_content_digest": record.get("checkpoint_content_digest"),
550
608
  # How much of the table this answer is an answer about. Studio records it because a
551
609
  # caller who pruned should be able to see that a one-day window read two parts rather
552
610
  # than be told to trust that it did.
@@ -606,7 +664,16 @@ def query(
606
664
  again.retry_after_seconds = window # type: ignore[attr-defined]
607
665
  raise _named_refusal(again) from None
608
666
  try:
609
- payload = _settled(selected, args, run_id, record, headers, sleep=sleep, clock=clock)
667
+ payload = _settled(
668
+ selected,
669
+ args,
670
+ run_id,
671
+ record,
672
+ headers,
673
+ checkpoint_id=body.get("checkpoint_id"),
674
+ sleep=sleep,
675
+ clock=clock,
676
+ )
610
677
  finally:
611
678
  closed = warm.close() if warm is not None else None
612
679
  if warm is not None:
@@ -625,6 +692,7 @@ def _settled(
625
692
  record: Mapping[str, Any],
626
693
  headers: Mapping[str, str],
627
694
  *,
695
+ checkpoint_id: str | None,
628
696
  sleep: Callable[[float], None],
629
697
  clock: Callable[[], float],
630
698
  ) -> dict[str, Any]:
@@ -636,7 +704,16 @@ def _settled(
636
704
  or str(headers.get(IDEMPOTENT_REPLAY_HEADER.lower(), "")).lower() == "true"
637
705
  )
638
706
  query_id = _record_identifier(record, run_id)
639
- record, read = _poll(selected, run_id, query_id, sleep=sleep, clock=clock)
707
+ checkpoint_content_digest = _checkpoint_binding(record, checkpoint_id)
708
+ record, read = _poll(
709
+ selected,
710
+ run_id,
711
+ query_id,
712
+ checkpoint_id=checkpoint_id,
713
+ checkpoint_content_digest=checkpoint_content_digest,
714
+ sleep=sleep,
715
+ clock=clock,
716
+ )
640
717
  head: dict[str, Any] = {
641
718
  "schema_version": QUERY_SCHEMA,
642
719
  "lane": "hosted",
@@ -656,12 +733,15 @@ def _settled(
656
733
  "state": state,
657
734
  "sql": record.get("sql"),
658
735
  "max_rows": record.get("max_rows"),
736
+ "checkpoint_id": record.get("checkpoint_id"),
737
+ "checkpoint_content_digest": record.get("checkpoint_content_digest"),
659
738
  "deadline_at": record.get("deadline_at"),
660
739
  "note": (
661
740
  f"This command read the answer back {read} times and it is still {state}. The "
662
741
  "query is unaffected: Studio terminalizes it at its own deadline above, and "
663
742
  "running this same command again reads it back rather than executing a second, "
664
- "because the identifier is derived from the statement and the row limit."
743
+ "because the identifier is derived from the statement, row limit, pruning and "
744
+ "checkpoint."
665
745
  ),
666
746
  }
667
747
  payload: dict[str, Any] = {**head, "status": "query_answered", **_answer(record)}
@@ -673,8 +753,8 @@ def _settled(
673
753
  )
674
754
  if replayed:
675
755
  notes.append(
676
- "This statement and row limit had already been asked of this run, so Studio answered "
677
- "with that one query rather than executing a second."
756
+ "This statement, row limit, pruning and checkpoint had already been asked of this "
757
+ "run, so Studio answered with that one query rather than executing a second."
678
758
  )
679
759
  if notes:
680
760
  payload["note"] = " ".join(notes)
@@ -763,6 +843,8 @@ def _poll(
763
843
  run_id: str,
764
844
  query_id: str,
765
845
  *,
846
+ checkpoint_id: str | None,
847
+ checkpoint_content_digest: str | None,
766
848
  sleep: Callable[[float], None],
767
849
  clock: Callable[[], float],
768
850
  ) -> tuple[dict[str, Any], int]:
@@ -791,6 +873,7 @@ def _poll(
791
873
  raise ThinLaneError(
792
874
  "THIN_RESPONSE_INVALID", "Studio answered with a query belonging to another run"
793
875
  )
876
+ _checkpoint_binding(record, checkpoint_id, checkpoint_content_digest)
794
877
  if record.get("state") in TERMINAL_QUERY_STATES:
795
878
  break
796
879
  return record, read
@@ -47,6 +47,8 @@ from mostlyright.data_harness.thin.v4 import (
47
47
  CREATE_RUN_PATH,
48
48
  GET_RUN_PATH,
49
49
  LIST_RUNS_PATH,
50
+ RUN_CHECKPOINT_PATH,
51
+ RUN_CHECKPOINTS_PATH,
50
52
  RUN_INSPECTION_PATH,
51
53
  bare_digest,
52
54
  identifier,
@@ -277,6 +279,45 @@ class StudioV4RunClient(StudioV4NarrativeClient):
277
279
  raise ThinLaneError("THIN_RESPONSE_INVALID", "Studio returned another run's inspection")
278
280
  return answer
279
281
 
282
+ def checkpoints(self, run_id: str) -> dict[str, Any]:
283
+ """Read the bounded discoverable checkpoint list for one run."""
284
+
285
+ answer = self._call("GET", RUN_CHECKPOINTS_PATH.format(run_id=run_id), expected=(200,))
286
+ if answer.get("schema_version") != "mostlyright-run-checkpoint.v1":
287
+ raise ThinLaneError(
288
+ "THIN_RESPONSE_INVALID", "Studio returned an invalid run checkpoint list"
289
+ )
290
+ rows = answer.get("checkpoints")
291
+ if answer.get("run_id") != run_id or not isinstance(rows, list):
292
+ raise ThinLaneError(
293
+ "THIN_RESPONSE_INVALID", "Studio returned another run's checkpoint list"
294
+ )
295
+ if any(
296
+ not isinstance(row, Mapping)
297
+ or row.get("schema_version") != "mostlyright-run-checkpoint.v1"
298
+ or row.get("run_id") != run_id
299
+ or not isinstance(row.get("checkpoint_id"), str)
300
+ for row in rows
301
+ ):
302
+ raise ThinLaneError(
303
+ "THIN_RESPONSE_INVALID", "Studio returned an invalid checkpoint in the run list"
304
+ )
305
+ return answer
306
+
307
+ def checkpoint(self, run_id: str, checkpoint_id: str) -> dict[str, Any]:
308
+ """Read one exact immutable checkpoint without resolving its artifact bytes."""
309
+
310
+ answer = self._call(
311
+ "GET",
312
+ RUN_CHECKPOINT_PATH.format(run_id=run_id, checkpoint_id=checkpoint_id),
313
+ expected=(200,),
314
+ )
315
+ if answer.get("schema_version") != "mostlyright-run-checkpoint.v1":
316
+ raise ThinLaneError("THIN_RESPONSE_INVALID", "Studio returned an invalid checkpoint")
317
+ if answer.get("run_id") != run_id or answer.get("checkpoint_id") != checkpoint_id:
318
+ raise ThinLaneError("THIN_RESPONSE_INVALID", "Studio returned another checkpoint")
319
+ return answer
320
+
280
321
  def runs(
281
322
  self,
282
323
  *,
@@ -400,8 +441,8 @@ def declare_run_arguments(parser: argparse.ArgumentParser) -> None:
400
441
  dest="mode",
401
442
  action="store_const",
402
443
  const="full",
403
- help="build the whole dataset; over the projection threshold Studio holds it for your "
404
- "confirmation, and above the sample-first threshold it holds it for a bounded preview",
444
+ help="build the whole dataset progressively; inspect its five-minute checkpoint while "
445
+ "acquisition continues, with spend confirmation before acquisition when required",
405
446
  )
406
447
  mode.add_argument(
407
448
  "--refresh",
@@ -445,7 +486,7 @@ def declare_run_arguments(parser: argparse.ArgumentParser) -> None:
445
486
  parser.add_argument(
446
487
  "--progressive",
447
488
  action="store_true",
448
- help="start a full acquisition with a five-minute inspection checkpoint while it continues",
489
+ help="compatibility alias: every full run already uses progressive acquisition",
449
490
  )
450
491
  parser.add_argument(
451
492
  "--resume-capture-run-id",
@@ -481,7 +522,7 @@ def declare_run_arguments(parser: argparse.ArgumentParser) -> None:
481
522
  "--approve-full",
482
523
  dest="approve_full",
483
524
  metavar="RUN_ID",
484
- help="release the full run a sample-first pair is holding, once its preview has "
525
+ help="release an existing legacy sample-first full run, once its preview has "
485
526
  "succeeded; names the held full or the preview it is linked to",
486
527
  )
487
528
 
@@ -495,6 +536,17 @@ def declare_status_arguments(parser: argparse.ArgumentParser) -> None:
495
536
  action="store_true",
496
537
  help="include the run's declared shape and observed checkpoint",
497
538
  )
539
+ selected = parser.add_mutually_exclusive_group()
540
+ selected.add_argument(
541
+ "--checkpoints",
542
+ action="store_true",
543
+ help="include the bounded list of immutable workspace build checkpoints",
544
+ )
545
+ selected.add_argument(
546
+ "--checkpoint",
547
+ metavar="CHECKPOINT_ID",
548
+ help="include one exact immutable build checkpoint",
549
+ )
498
550
  parser.add_argument("--receipts", action="store_true", help=_NO_EFFECT_RECEIPTS)
499
551
 
500
552
 
@@ -683,14 +735,12 @@ def create_run_body(args: argparse.Namespace) -> dict[str, Any]:
683
735
  "state --mode replay or drop the flag",
684
736
  )
685
737
  resource_class = getattr(args, "resource_class", None)
686
- progressive = bool(getattr(args, "progressive", False))
738
+ progressive_requested = bool(getattr(args, "progressive", False))
687
739
  resume_capture_run_id = getattr(args, "resume_capture_run_id", None)
688
- if progressive and mode != "full":
740
+ if progressive_requested and mode != "full":
689
741
  raise ThinLaneError(ARGUMENT_INVALID_CODE, "--progressive requires --full")
690
- if resume_capture_run_id and not progressive:
691
- raise ThinLaneError(
692
- ARGUMENT_INVALID_CODE, "--resume-capture-run-id requires --progressive --full"
693
- )
742
+ if resume_capture_run_id and mode != "full":
743
+ raise ThinLaneError(ARGUMENT_INVALID_CODE, "--resume-capture-run-id requires --full")
694
744
  clamps = _clamps(args)
695
745
  if mode == "sample" and not clamps:
696
746
  raise ThinLaneError(
@@ -721,7 +771,7 @@ def create_run_body(args: argparse.Namespace) -> dict[str, Any]:
721
771
  body["clamps"] = clamps
722
772
  if resource_class:
723
773
  body["resource_class"] = resource_class
724
- if progressive:
774
+ if mode == "full":
725
775
  body["progressive"] = True
726
776
  if resume_capture_run_id:
727
777
  body["resume_capture_run_id"] = identifier(
@@ -1067,6 +1117,10 @@ def retry_body(run: Mapping[str, Any], run_id: str, *, declared_bytes: int = 0)
1067
1117
  )
1068
1118
  body[member] = stated
1069
1119
  body["clamps"] = retry_clamps(run, declared_bytes=declared_bytes)
1120
+ if body["mode"] == "full":
1121
+ body["progressive"] = True
1122
+ if run.get("progressive") is True:
1123
+ body["resume_capture_run_id"] = identifier(run_id, "capture ancestor run identifier")
1070
1124
  resource_class = run.get("resource_class")
1071
1125
  if isinstance(resource_class, str) and resource_class:
1072
1126
  body["resource_class"] = resource_class
@@ -1261,9 +1315,11 @@ def _retry_cell(
1261
1315
  f"Run {run_id} failed at the {triple['failed_stage']} stage with "
1262
1316
  f"{triple['failure_code']}: {triple['failure_detail']}\n\n"
1263
1317
  "That code is about the room the run landed in rather than about this recipe, so the "
1264
- "document is unchanged and this is the same command again. Its only difference is a "
1265
- "byte ceiling above anything the sources can deliver, which is what gives the attempt "
1266
- "a coordinate of its own; an identical command would have replayed the stored failure."
1318
+ "document is unchanged. A byte ceiling above anything the sources can deliver gives "
1319
+ "the attempt a coordinate of its own; an identical command would have replayed the "
1320
+ "stored failure. Full retries use progressive acquisition. When the failed full was "
1321
+ "already progressive, the retry also names it as the capture ancestor so Studio can "
1322
+ "reuse compatible sealed pages."
1267
1323
  ),
1268
1324
  blocks=[
1269
1325
  {
@@ -1523,6 +1579,12 @@ def status(args: argparse.Namespace, *, client: StudioV4RunClient | None = None)
1523
1579
  payload["flags_without_effect"] = {"--receipts": _NO_EFFECT_RECEIPTS}
1524
1580
  if getattr(args, "inspection", False):
1525
1581
  payload["inspection"] = selected.inspection(run_id)
1582
+ if getattr(args, "checkpoints", False):
1583
+ payload["checkpoints"] = selected.checkpoints(run_id)
1584
+ checkpoint = getattr(args, "checkpoint", None)
1585
+ if checkpoint is not None:
1586
+ checkpoint_id = identifier(str(checkpoint), "checkpoint identifier")
1587
+ payload["checkpoint"] = selected.checkpoint(run_id, checkpoint_id)
1526
1588
  if state == "awaiting_confirmation":
1527
1589
  held = {member: run[member] for member in _PROJECTION_MEMBERS[1:] if member in run}
1528
1590
  if held:
@@ -687,7 +687,8 @@ def declare_arguments(command: str, parser: argparse.ArgumentParser) -> None:
687
687
  "it before submitting, then reuse it after a lost response to ask for the "
688
688
  "same resync, not another one. If Studio persists it at the spend gate, the "
689
689
  "receipt prints mr-data run --confirm-held RUN_ID for that exact run; a "
690
- "sample-first receipt confirms its preview before its linked full can be released"
690
+ "legacy sample-first receipt confirms its preview before its linked full can "
691
+ "be released"
691
692
  ),
692
693
  )
693
694
  parser.add_argument(
@@ -229,11 +229,14 @@ CLASSIFICATION: dict[str, Lane] = {
229
229
  HOSTED,
230
230
  "it starts one run of a registered recipe in one of the six protocol modes -- sample, "
231
231
  "full, refresh, backfill, compact or replay -- under the row and byte ceilings you give "
232
- "it, and settles the spend confirmation a large one is held for. A replay names the "
232
+ "it, and settles the spend confirmation a large one is held for. Full builds are "
233
+ "progressive by default, with an inspection checkpoint during the same acquisition. "
234
+ "A replay names the "
233
235
  "retained raw inputs of one successful run of the same table with --sources-from, "
234
236
  "fetches nothing here and never becomes the live version. --retry names one failed run "
235
237
  "and tries it again when what failed was the room rather than the recipe, and "
236
- "--approve-full releases the full a sample-first pair is holding once its preview has "
238
+ "--approve-full releases the full an existing legacy sample-first pair is holding "
239
+ "once its preview has "
237
240
  "succeeded; a deployment that still wants a person at a browser refuses that release by "
238
241
  "name and this command says where they settle it",
239
242
  ),