mostlyright-data 0.21.0__tar.gz → 0.21.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/PKG-INFO +2 -1
  2. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/README.md +1 -0
  3. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/pyproject.toml +1 -1
  4. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/skills/mr-data-build/SKILL.md +79 -43
  5. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/stream_venue.py +26 -1
  6. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4.py +13 -7
  7. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/.gitignore +0 -0
  8. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/scripts/hatch_build.py +0 -0
  9. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/skills/mr-data-build/agents/openai.yaml +0 -0
  10. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/skills/mr-data-build/scripts/write_research_notebook.py +0 -0
  11. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/__init__.py +0 -0
  12. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/canonical.py +0 -0
  13. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/formats.py +0 -0
  14. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/hosted_crawler_protocol.py +0 -0
  15. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/key_seam.py +0 -0
  16. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/page_coverage.py +0 -0
  17. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/part_check_evidence.py +0 -0
  18. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/session_probes.py +0 -0
  19. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/skill_assets.py +0 -0
  20. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/table_manifest.py +0 -0
  21. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/__init__.py +0 -0
  22. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/acquire.py +0 -0
  23. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/acquire_cancel.py +0 -0
  24. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/activity.py +0 -0
  25. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/approvals.py +0 -0
  26. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/categories.py +0 -0
  27. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/commands.py +0 -0
  28. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/dataset-categories-v1.json +0 -0
  29. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/download.py +0 -0
  30. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/narrative.py +0 -0
  31. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/parity.py +0 -0
  32. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/probe.py +0 -0
  33. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/progress_vocabulary.py +0 -0
  34. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/propose.py +0 -0
  35. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/recipe.py +0 -0
  36. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/recipe_brief.py +0 -0
  37. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/recipe_lint.py +0 -0
  38. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/research.py +0 -0
  39. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/router.py +0 -0
  40. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/runs.py +0 -0
  41. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/session.py +0 -0
  42. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/stream.py +0 -0
  43. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/transport.py +0 -0
  44. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/user_agent.py +0 -0
  45. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_artifacts.py +0 -0
  46. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_catalog.py +0 -0
  47. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_connections.py +0 -0
  48. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_dataset_covers.py +0 -0
  49. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_datasets.py +0 -0
  50. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_handoff.py +0 -0
  51. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_narrative.py +0 -0
  52. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_query.py +0 -0
  53. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_runs.py +0 -0
  54. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_secrets.py +0 -0
  55. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_stream.py +0 -0
  56. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/v4_tables.py +0 -0
  57. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/thin/vocabulary.py +0 -0
  58. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/__init__.py +0 -0
  59. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/attendance.py +0 -0
  60. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/clarification.py +0 -0
  61. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/cloud_auth.py +0 -0
  62. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/commands/__init__.py +0 -0
  63. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/commands/auth.py +0 -0
  64. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/commands/clarify.py +0 -0
  65. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/commands/login.py +0 -0
  66. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/commands/whoami.py +0 -0
  67. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/credential_native.py +0 -0
  68. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/credential_store.py +0 -0
  69. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/credentials.py +0 -0
  70. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/login.py +0 -0
  71. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/path_kind.py +0 -0
  72. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/plain_file.py +0 -0
  73. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/remediation.py +0 -0
  74. {mostlyright_data-0.21.0 → mostlyright_data-0.21.2}/src/mostlyright/data_harness/ux/render.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyright-data
3
- Version: 0.21.0
3
+ Version: 0.21.2
4
4
  Summary: Mostly Right hosted CLI for reviewed datasets
5
5
  Project-URL: Homepage, https://mostlyright.md/
6
6
  Project-URL: Documentation, https://mostlyright.md/docs/guides/cli/
@@ -135,6 +135,7 @@ worker executables or image publisher.
135
135
  - [Aggregate a stream into bars](https://mostlyright.md/docs/recipes/stream-to-bars/)
136
136
  - [Use a public dataset](https://mostlyright.md/docs/guides/use-public-datasets/)
137
137
  - [Certified document extraction](docs/DOCUMENT-EXTRACTION.md)
138
+ - [Use a table to drive a stream's market roster](docs/TABLE-DRIVEN-DISCOVERY.md)
138
139
 
139
140
  Use `mr-data --help` for the full command list and options. Commands that support
140
141
  `--json` write one JSON object.
@@ -123,6 +123,7 @@ worker executables or image publisher.
123
123
  - [Aggregate a stream into bars](https://mostlyright.md/docs/recipes/stream-to-bars/)
124
124
  - [Use a public dataset](https://mostlyright.md/docs/guides/use-public-datasets/)
125
125
  - [Certified document extraction](docs/DOCUMENT-EXTRACTION.md)
126
+ - [Use a table to drive a stream's market roster](docs/TABLE-DRIVEN-DISCOVERY.md)
126
127
 
127
128
  Use `mr-data --help` for the full command list and options. Commands that support
128
129
  `--json` write one JSON object.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "mostlyright-data"
3
- version = "0.21.0"
3
+ version = "0.21.2"
4
4
  description = "Mostly Right hosted CLI for reviewed datasets"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -586,7 +586,7 @@ dataset from the Kathmandu one the earlier stages follow, chosen because its num
586
586
  count.)
587
587
 
588
588
  ```sh
589
- mr-data dataset set DATASET_ID --name "Hourly weather at Prague Airport (LKPR), 2000-2026" \
589
+ mr-data dataset set DATASET_ID --name "Prague airport weather history (LKPR), 2000-2026" \
590
590
  --category climate-environment --topics "weather,prague,czech republic,metar,iowa environmental mesonet" \
591
591
  --license "CC-BY-4.0" --json
592
592
  ```
@@ -615,13 +615,26 @@ interpret them.
615
615
  | `--topics` | JSON-LD `keywords`, the `Topics:` line answer engines fetch, the `topic` filter in the public API and MCP | 3 to 8 topics, each at most 40 characters |
616
616
 
617
617
  **Title rules**
618
- 1. Shape: subject, place, period. `Hourly weather at Prague Airport (LKPR), 2000-2026`.
619
- 2. At most 60 characters. Count them.
620
- 3. Use the words a person would type: the common place name, and the official identifier in
621
- parentheses once if one exists (ICAO code, station id, ticker, constituency name).
622
- 4. Never: a brand, the word "dataset" or "data", a colon, a pipe, an exclamation mark, a version
623
- number, a run date.
624
- 5. Sentence case. No trailing full stop.
618
+ 1. Lead with what a person would type to find it: the subject and its place in plain words, in
619
+ the order a searcher says them. `Denver weather history since 2020`, not `weather history for
620
+ Denver`. `Kalshi Dogecoin hourly price markets, live order book`. `Kalshi market settlement
621
+ rules, 2026`.
622
+ 2. Never open with a grain word (hourly, daily, 15-minute), a mechanism (order book,
623
+ observations, reports, METAR), a station code, or a publisher that is not itself the subject.
624
+ Those come later in the title, or in the opening paragraph. A venue whose own markets, books
625
+ or rules are what the dataset holds IS the subject and leads: `Kalshi Dogecoin hourly price
626
+ markets`. A publisher that only relays somebody else's measurements is not the subject and
627
+ never leads: write `Prague airport weather history`, not `Iowa Environmental Mesonet hourly
628
+ weather`.
629
+ 3. At most 60 characters. Count them. Sentence case, no trailing full stop. The official
630
+ identifier once, in parentheses, only where people search by it and it fits (ICAO code,
631
+ station id, ticker, constituency name).
632
+ 4. Period last, after the subject and the place: `since 2020`, `2000-2026`, `live`. Where the
633
+ title ends on the mechanism, the period sits in front of it: `live order book`.
634
+ 5. Never: a brand that is not the subject, the word "dataset" or "data", a colon, a pipe, an
635
+ exclamation mark, a version number, a run date. A run date is the day the build ran; the
636
+ period the data covers is not one, even where it is the current year.
637
+ 6. The place or the subject word is what tells two titles in the set apart, never a code alone.
625
638
 
626
639
  **Primary category — required before recipe registration**
627
640
  Run `mr-data dataset categories --json` to read the fixed vocabulary and its definitions,
@@ -676,8 +689,8 @@ seconds -- not a paragraph, and not a report. Use exactly this shape, in this or
676
689
  section out only when there is genuinely nothing to say (shown indented here; the document
677
690
  itself carries the headings at column one):
678
691
 
679
- A plain opening paragraph: what the data contains, what question or model it supports, and
680
- the facts a reader needs before using it.
692
+ A plain opening paragraph: what this is, what it is for, and then the facts a reader needs
693
+ before using it.
681
694
 
682
695
  ## Coverage
683
696
  - Window: the exact dates and the time zone.
@@ -701,33 +714,52 @@ That is six sections: the opening paragraph, and the five headings under it.
701
714
 
702
715
  **Opening paragraph** (everything before the first `##` heading)
703
716
 
704
- Write as an engineer explaining a useful table to another engineer. The opening may have more than
705
- two sentences. Start with the concrete records the dataset contains. Then say which question the
706
- records help answer or which model they can support. Include geography, time range, row grain,
707
- update cadence and material limitations when those facts are known. Put a join, transformation or
708
- source caveat here only when it changes how somebody should use the data.
709
-
710
- 1. The first two sentences must distinguish this dataset from every other dataset in the set. If
711
- both sentences could describe another dataset unchanged, rewrite them.
712
- 2. Use ordinary verbs such as contains, tracks, joins, updates, records and returns. Vary sentence
713
- length, and use short sentences for the important facts.
714
- 3. Preserve every established fact. Never invent coverage, freshness, quality, licensing or an
715
- intended use.
716
- 4. Write `Mostly Right`, with a space, except inside a literal identifier that uses another form.
717
- 5. Use no Markdown in the opening paragraph. Do not use generic openings such as `This dataset
718
- provides`, marketing claims such as `comprehensive`, `powerful` or `high-quality`, or abstract
719
- phrases such as `enables insights`, `facilitates analysis` or `serves as a valuable resource`.
720
- 6. Do not explain page design or metadata fields. Do not use em dashes, semicolons, bold labels or
721
- fake quotations. Explain the data before implementation details.
722
-
723
- After the build establishes each fact, an airport-weather opening can read:
724
-
725
- Hourly weather observations from 20 airport stations across the United States. Each row records
726
- one station report with temperature, dew point, wind, visibility, and observation time. Use it
727
- to compare observed conditions across stations and time or train a short-term weather model.
728
- The table keeps multiple reports within an hour when conditions change, so aggregate it before
729
- using one row per station-hour. Mostly Right refreshes the recent window and preserves older
730
- observations.
717
+ Its first 155 characters are the snippet a search engine usually prints under the title, its first
718
+ 300 the meta description that search engine read to build it, its first 240 the summary an answer
719
+ engine quotes. The order is fixed: what it is, what it is for, then the facts.
720
+
721
+ 1. Sentence one says what this is, in the words a searcher uses, with the place where there is
722
+ one and the period as a searcher writes it (`since 2020`, `2000-2026`, `live`). Keep it under
723
+ 155 characters, because the snippet cuts there and a sentence that fits is printed whole.
724
+ `Denver weather history since 2020: every airport report from Denver International (KDEN) with
725
+ the official daily high and low.`
726
+ 2. Sentence two says what it is for: the question it answers or the model it feeds, concretely.
727
+ `Built for daily temperature forecasting and for checking the weather at any hour.`
728
+ 3. Then the facts a reader needs before opening a row: grain, cadence, the material limitation,
729
+ and the exact window where sentence one could carry only a year. `One row per report, about 30
730
+ a day, refreshed each morning with the previous day added.`
731
+ 4. Never open with the grain, the row, the mechanism, a station code, a publisher that is not
732
+ itself the subject, a date, or the words "records", "observations", "each row", "rows". Those
733
+ belong in sentence three. This governs first position only; `records` and `returns` are still
734
+ good verbs later in the paragraph.
735
+ 5. Name the subject the way people search it once (weather history, hourly weather, order book,
736
+ settlement rules) and the official term once (METAR, SPECI, YES bid), in that order.
737
+ 6. Everything that already holds still holds. The first two sentences distinguish this dataset
738
+ from every other dataset in the set; if both could describe another dataset unchanged, rewrite
739
+ them. Use ordinary verbs such as contains, tracks, joins, updates, records and returns, and
740
+ vary sentence length. Preserve every established fact, and never invent coverage, freshness,
741
+ quality, licensing or an intended use. Write `Mostly Right`, with a space, except inside a
742
+ literal identifier that uses another form. Use no Markdown. Do not use generic openings such as
743
+ `This dataset provides`, marketing claims such as `comprehensive`, `powerful` or
744
+ `high-quality`, or abstract phrases such as `enables insights`, `facilitates analysis` or
745
+ `serves as a valuable resource`. Do not explain page design or metadata fields, and do not use
746
+ em dashes, semicolons, bold labels or fake quotations.
747
+
748
+ After the build establishes each fact, a daily-weather opening can read:
749
+
750
+ Denver weather history since 2020: every airport report from Denver International (KDEN) with
751
+ the official daily high and low. Built for daily temperature forecasting and for checking the
752
+ weather at any hour. One row per report, about 30 a day, refreshed each morning with the
753
+ previous day added. A METAR arrives more often than hourly when conditions change, so
754
+ aggregate before using one row per hour.
755
+
756
+ And a market opening, on the same three steps:
757
+
758
+ Kalshi Dogecoin hourly price markets, live: the order book behind every price-range contract,
759
+ as traders posted their quotes. Built for studying how traders price hourly moves and for
760
+ backtesting market making against quotes that really rested. One row per market-second while a
761
+ YES bid rests, written as each quote lands. Quiet seconds are empty, and recording has run
762
+ since 10 September 2026, only while a market is open.
731
763
 
732
764
  **Body rules** (the five sections under the headings)
733
765
  1. Every sentence must stay true and complete when quoted alone, away from the page. Never
@@ -746,16 +778,20 @@ code blocks, no links to internal tools, no headings beyond
746
778
  the ones above. Spell out the names a reader would search for -- the place, the identifiers in
747
779
  every common form, the programme and the publisher -- once each, under Coverage and under
748
780
  Source and rights, and never as a keyword list. A one-line description is a defect to fix in the
749
- revision, not a style choice. Give `table.description` the grain and the window.
781
+ revision, not a style choice. Give `table.description` the same order in one sentence: what
782
+ the table is, then its grain and its window.
750
783
  Every statement in the opening and body must preserve an established fact. Never add unsupported
751
784
  coverage, freshness, quality, licensing or intended use.
752
785
 
753
786
  **Check before writing.** Fix and re-check until every line passes:
754
- - title: at most 60 characters, contains the place, contains no colon, pipe or exclamation mark,
755
- no brand, no "dataset" or "data", no version number, no run date
756
- - opening paragraph: at least two distinct sentences, with the concrete records first and the
757
- question or model they support next; include known geography, time range, grain, update cadence
758
- and material limitations; no Markdown, banned style or invented fact
787
+ - title: leads with the search phrase, the subject and its place in the order a searcher says
788
+ them, and does not open with a grain word, a mechanism, a station code or a publisher that is
789
+ not the subject; at most 60 characters, contains the place where the subject has one, contains
790
+ no colon, pipe or exclamation mark, no outside brand, no "dataset" or "data", no version
791
+ number, no run date
792
+ - opening paragraph: sentence one is under 155 characters and says what this is, with the place
793
+ and the period; sentence two says what it is for; grain, cadence and the material limitation
794
+ appear only after those two; no Markdown, banned style or invented fact
759
795
  - category: exactly one fixed ID saved and verified before recipe registration
760
796
  - topics: 3 to 8 descriptive tags, no duplicates, lowercase, at most 40 characters each
761
797
  - licence: an SPDX identifier the sources actually grant, or `--license` left off
@@ -202,6 +202,20 @@ RESOURCE_MEMBERS: tuple[str, ...] = (
202
202
  "event_shape_descriptor",
203
203
  "bookmark",
204
204
  "expected_watermark",
205
+ "table_discovery",
206
+ )
207
+
208
+ # Only the public dependency summary is rendered. An API adding transport or snapshot fields
209
+ # beneath this object must not make signed sessions or private source rows appear in CLI output.
210
+ TABLE_DISCOVERY_MEMBERS: tuple[str, ...] = (
211
+ "table_id",
212
+ "table_version_id",
213
+ "resolution_id",
214
+ "status",
215
+ "source_completed_at",
216
+ "freshness_checked_at",
217
+ "freshness_deadline_at",
218
+ "last_error_code",
205
219
  )
206
220
 
207
221
  #: The members one registered document is rendered under, spelled the way Studio spells them.
@@ -705,7 +719,18 @@ def resource(record: Mapping[str, Any], *, id_member: str) -> dict[str, Any]:
705
719
  rendered: dict[str, Any] = {id_member: record.get(id_member)}
706
720
  for member in RESOURCE_MEMBERS:
707
721
  if member in record:
708
- rendered[member] = record[member]
722
+ if member == "table_discovery":
723
+ summary = record[member]
724
+ if isinstance(summary, Mapping):
725
+ selected = {
726
+ name: value
727
+ for name in TABLE_DISCOVERY_MEMBERS
728
+ if isinstance(value := summary.get(name), str) and 0 < len(value) <= 256
729
+ }
730
+ if "table_id" in selected and "status" in selected:
731
+ rendered[member] = selected
732
+ else:
733
+ rendered[member] = record[member]
709
734
  return rendered
710
735
 
711
736
 
@@ -54,16 +54,16 @@ from mostlyright.data_harness.thin.transport import ThinLaneError
54
54
  #: WHY A LATER NUMBER IS SAFE TO SEND, AND AGAINST WHICH STUDIO. Studio's ``common.schema.json``
55
55
  #: gives command bodies ``request_schema_version`` -- deliberately wider than the ``4.0.0``
56
56
  #: envelope constant every V4 RESPONSE still stamps -- and the pinned copy of it admits ``4.0.0``,
57
- #: ``4.1.0``, ``4.2.0``, ``4.3.0``, ``4.4.0``, ``4.5.0`` and ``4.6.0``. Every command this module
58
- #: sends binds
57
+ #: ``4.1.0``, ``4.2.0``, ``4.3.0``, ``4.4.0``, ``4.5.0``, ``4.6.0`` and ``4.7.0``. Every command
58
+ #: this module sends binds
59
59
  #: that wider definition
60
60
  #: (``test_every_command_this_client_stamps_binds_the_wide_request_version``), so a body carrying
61
61
  #: this number is accepted by a Studio AT OR AFTER THE PINNED COMMIT.
62
62
  #:
63
63
  #: ⚠ AND NOT BY ONE BEFORE IT. Studio validates the body against that enum at request time, so
64
64
  #: this number is a DEPLOY ORDERING and not just a pin: ``4.2.0`` entered the enum in studio#531,
65
- #: ``4.3.0`` in studio#533, ``4.4.0`` in studio#552, ``4.5.0`` in studio#644 and ``4.6.0`` in
66
- #: studio#659, and a release of
65
+ #: ``4.3.0`` in studio#533, ``4.4.0`` in studio#552, ``4.5.0`` in studio#644, ``4.6.0`` in
66
+ #: studio#659 and ``4.7.0`` in Studio's legacy-normalization lane, and a release of
67
67
  #: this package that reaches a Studio older than those turns four commands -- ``dataset activity``,
68
68
  #: ``table rename``, ``table archive`` and ``dataset archive`` -- into ``422``s that used to be
69
69
  #: accepted. That is the whole reason this constant may only be raised in the ``make repin`` commit
@@ -92,6 +92,12 @@ from mostlyright.data_harness.thin.transport import ThinLaneError
92
92
  #: repinned against those bytes stamps it on every command it sends, which is why the enum in
93
93
  #: ``common.schema.json`` had to admit it and why the deploy ordering above covers it too.
94
94
  #:
95
+ #: ⚠ ``4.7.0`` ADDS THE PUBLIC REQUEST/CANCEL SURFACE FOR STUDIO'S SOURCE-FREE LEGACY
96
+ #: NORMALIZER. It is deliberately recorded in ``UNUSED_STUDIO_PATHS``: choosing the two legacy
97
+ #: predecessor coordinates is Studio's recovery flow, and the thin client neither holds nor
98
+ #: lists those reader inputs. The pin still rises because ordinary command bodies stamp the exact
99
+ #: version Studio's common request grammar now declares.
100
+ #:
95
101
  #: ⚠ AND THE WORKER SIDE OF THAT DELETION IS A DEPLOY ORDERING OF ITS OWN, in the OPPOSITE
96
102
  #: direction from the one above. ``run_execution_v4``'s job parser compares an EXACT field set, and
97
103
  #: ``source_collection_protocol``, ``source_collection`` and ``source_collection_inputs`` are no
@@ -99,8 +105,8 @@ from mostlyright.data_harness.thin.transport import ThinLaneError
99
105
  #: ``JOB_INVALID`` outright, which in production was every refresh a collection epoch was offered
100
106
  #: for rather than only the collection runs. This package must not reach a Studio older than the
101
107
  #: commit it pins; ``docs/V4-WORKER-PROTOCOL.md`` states it beside the layout's own ordering rule.
102
- PINNED_V4_OPENAPI_SOURCE_SHA256 = "b88ef1ea9535232f331bb1337d6a9d0def27e4759fb0a2f46085e481dabc395d"
103
- PINNED_V4_CONTRACT_VERSION = "4.6.0"
108
+ PINNED_V4_OPENAPI_SOURCE_SHA256 = "2a83dda60e84c166f95ea18af747985bfeeb5887cb48d2c72d4335e83720dc37"
109
+ PINNED_V4_CONTRACT_VERSION = "4.7.0"
104
110
 
105
111
  # --------------------------------------------------------------------------------------------
106
112
  # The routes
@@ -612,7 +618,7 @@ DECLARED_V4_PATHS: frozenset[str] = frozenset(
612
618
  # authentication alone -- see `CATALOG_SEARCH_PATH` for why a `workspace_id` would state
613
619
  # an isolation the answer does not have -- and the document version it arrived in,
614
620
  # `4.5.0`, was the pinned version when this route was promoted (the pin has since
615
- # moved on to `4.6.0`). `thin/v4_catalog.py` was
621
+ # moved on to `4.7.0`). `thin/v4_catalog.py` was
616
622
  # written against the frozen cross-repo contract while Studio built its half, for the
617
623
  # reason the dataset surface was: a 422 is measured against a shape.
618
624
  CATALOG_SEARCH_PATH,