mostlyright-data 0.19.5__tar.gz → 0.19.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/PKG-INFO +1 -1
  2. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/pyproject.toml +1 -1
  3. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/skills/mr-data-build/SKILL.md +57 -59
  4. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/recipe.py +30 -38
  5. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/router.py +45 -1
  6. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/.gitignore +0 -0
  7. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/README.md +0 -0
  8. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/scripts/hatch_build.py +0 -0
  9. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/skills/mr-data-build/agents/openai.yaml +0 -0
  10. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/skills/mr-data-build/scripts/write_research_notebook.py +0 -0
  11. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/__init__.py +0 -0
  12. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/canonical.py +0 -0
  13. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/formats.py +0 -0
  14. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/hosted_crawler_protocol.py +0 -0
  15. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/key_seam.py +0 -0
  16. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/page_coverage.py +0 -0
  17. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/part_check_evidence.py +0 -0
  18. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/session_probes.py +0 -0
  19. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/skill_assets.py +0 -0
  20. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/table_manifest.py +0 -0
  21. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/__init__.py +0 -0
  22. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/acquire.py +0 -0
  23. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/acquire_cancel.py +0 -0
  24. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/activity.py +0 -0
  25. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/approvals.py +0 -0
  26. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/categories.py +0 -0
  27. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/commands.py +0 -0
  28. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/dataset-categories-v1.json +0 -0
  29. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/download.py +0 -0
  30. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/narrative.py +0 -0
  31. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/parity.py +0 -0
  32. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/probe.py +0 -0
  33. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/progress_vocabulary.py +0 -0
  34. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/propose.py +0 -0
  35. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/research.py +0 -0
  36. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/runs.py +0 -0
  37. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/session.py +0 -0
  38. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/stream.py +0 -0
  39. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/stream_venue.py +0 -0
  40. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/transport.py +0 -0
  41. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/user_agent.py +0 -0
  42. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4.py +0 -0
  43. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_artifacts.py +0 -0
  44. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_catalog.py +0 -0
  45. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_connections.py +0 -0
  46. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_dataset_covers.py +0 -0
  47. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_datasets.py +0 -0
  48. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_handoff.py +0 -0
  49. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_narrative.py +0 -0
  50. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_query.py +0 -0
  51. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_runs.py +0 -0
  52. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_secrets.py +0 -0
  53. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_stream.py +0 -0
  54. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/v4_tables.py +0 -0
  55. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/thin/vocabulary.py +0 -0
  56. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/__init__.py +0 -0
  57. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/attendance.py +0 -0
  58. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/clarification.py +0 -0
  59. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/cloud_auth.py +0 -0
  60. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/commands/__init__.py +0 -0
  61. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/commands/auth.py +0 -0
  62. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/commands/clarify.py +0 -0
  63. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/commands/login.py +0 -0
  64. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/commands/whoami.py +0 -0
  65. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/credential_native.py +0 -0
  66. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/credential_store.py +0 -0
  67. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/credentials.py +0 -0
  68. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/login.py +0 -0
  69. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/path_kind.py +0 -0
  70. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/plain_file.py +0 -0
  71. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/remediation.py +0 -0
  72. {mostlyright_data-0.19.5 → mostlyright_data-0.19.7}/src/mostlyright/data_harness/ux/render.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyright-data
3
- Version: 0.19.5
3
+ Version: 0.19.7
4
4
  Summary: Mostly Right hosted CLI for reviewed datasets
5
5
  Project-URL: Homepage, https://mostlyright.md/
6
6
  Project-URL: Documentation, https://mostlyright.md/docs/guides/cli/
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "mostlyright-data"
3
- version = "0.19.5"
3
+ version = "0.19.7"
4
4
  description = "Mostly Right hosted CLI for reviewed datasets"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.13"
@@ -640,16 +640,16 @@ seconds -- not a paragraph, and not a report. Use exactly this shape, in this or
640
640
  section out only when there is genuinely nothing to say (shown indented here; the document
641
641
  itself carries the headings at column one):
642
642
 
643
- Two or three plain sentences: what the data is about, why it is worth a look, and what is
644
- in it.
643
+ A plain opening paragraph: what the data contains, what question or model it supports, and
644
+ the facts a reader needs before using it.
645
645
 
646
646
  ## Coverage
647
- - **Window:** the exact dates and the time zone.
648
- - **Grain:** one row per what.
649
- - **Cadence:** how often rows arrive, and how many per day or month.
647
+ - Window: the exact dates and the time zone.
648
+ - Grain: one row per what.
649
+ - Cadence: how often rows arrive, and how many per day or month.
650
650
 
651
651
  ## Columns
652
- - `column_name` — what it measures, in which unit.
652
+ - `column_name`: what it measures, in which unit.
653
653
  - (one bullet per column; group unit-sharing columns on one line where that reads better)
654
654
 
655
655
  ## Missing values
@@ -665,43 +665,33 @@ That is six sections: the opening paragraph, and the five headings under it.
665
665
 
666
666
  **Opening paragraph** (everything before the first `##` heading)
667
667
 
668
- Two or three short sentences, 120 to 280 characters, written for somebody who has never heard of
669
- the source, the venue or the format. Sentence one says what the data is about and what makes it
670
- worth a look, in at most 110 characters. The rest says, in the same plain words, what is in it.
671
-
672
- 1. No tickers, station codes, series codes, format names, publisher names, row definitions (the
673
- sentence that begins `Each row is`) or row counts here. Every one of those facts is in the
674
- sections below, which is where a reader who wants it will look.
675
- 2. No Markdown: no bold, no links, no bullets, no code.
676
- 3. A hook is an interesting fact stated plainly, never a pitch. Banned outright: comprehensive,
677
- rich, valuable, powerful, unlock, explore, dive into, perfect for, insights, "this dataset", a
678
- rhetorical question, an exclamation mark.
679
- 4. Every number in it must be one the run established. "About every half hour" and "thousands of
680
- markets" are claims, and a claim that is not on the run does not go here.
681
- 5. The stranger test, before anything else is written: read sentence one alone under the title. If
682
- a curious non-specialist could not say what the data is about and would not want the next line,
683
- rewrite it.
684
- 6. Plain and short. Every sentence under about twenty words, every sentence a fact: the venue, the
685
- subject, the horizon, the cadence, the window. Nothing about what a reader might do with it: no
686
- "so you can", "a way to", "ready to", "for seeing". Say what is there and stop.
687
-
688
- One airport weather table, the shape to avoid and the shape to write. To avoid:
689
-
690
- Each row is one METAR or SPECI weather report from Beijing (ZBAA), since January 2020, via
691
- Iowa Environmental Mesonet. 117,876 reports of temperature, wind, pressure, visibility with
692
- the day's highest and lowest reading.
693
-
694
- To write:
695
-
696
- Beijing airport weather since 2020, a report about every half hour, with the day's high and
697
- low on each row. Temperature, wind, pressure and visibility as the airport recorded them.
698
-
699
- The station code, the report types, the publisher and the count all moved down to Coverage and to
700
- Source and rights, where a reader who wants them looks. "About every half hour" stays only because
701
- Coverage puts a month at 1,302 to 1,591 rows, which is one row every 27 to 33 minutes. Three ways
702
- to fail: the pitch (`Unlock real-time insights into how prediction markets price
703
- the weather!`), the promise (`ready to test a forecast against what actually happened`) and the
704
- bare subject with no reason to care (`Beijing weather data.`).
668
+ Write as an engineer explaining a useful table to another engineer. The opening may have more than
669
+ two sentences. Start with the concrete records the dataset contains. Then say which question the
670
+ records help answer or which model they can support. Include geography, time range, row grain,
671
+ update cadence and material limitations when those facts are known. Put a join, transformation or
672
+ source caveat here only when it changes how somebody should use the data.
673
+
674
+ 1. The first two sentences must distinguish this dataset from every other dataset in the set. If
675
+ both sentences could describe another dataset unchanged, rewrite them.
676
+ 2. Use ordinary verbs such as contains, tracks, joins, updates, records and returns. Vary sentence
677
+ length, and use short sentences for the important facts.
678
+ 3. Preserve every established fact. Never invent coverage, freshness, quality, licensing or an
679
+ intended use.
680
+ 4. Write `Mostly Right`, with a space, except inside a literal identifier that uses another form.
681
+ 5. Use no Markdown in the opening paragraph. Do not use generic openings such as `This dataset
682
+ provides`, marketing claims such as `comprehensive`, `powerful` or `high-quality`, or abstract
683
+ phrases such as `enables insights`, `facilitates analysis` or `serves as a valuable resource`.
684
+ 6. Do not explain page design or metadata fields. Do not use em dashes, semicolons, bold labels or
685
+ fake quotations. Explain the data before implementation details.
686
+
687
+ After the build establishes each fact, an airport-weather opening can read:
688
+
689
+ Hourly weather observations from 20 airport stations across the United States. Each row records
690
+ one station report with temperature, dew point, wind, visibility, and observation time. Use it
691
+ to compare observed conditions across stations and time or train a short-term weather model.
692
+ The table keeps multiple reports within an hour when conditions change, so aggregate it before
693
+ using one row per station-hour. Mostly Right refreshes the recent window and preserves older
694
+ observations.
705
695
 
706
696
  **Body rules** (the five sections under the headings)
707
697
  1. Every sentence must stay true and complete when quoted alone, away from the page. Never
@@ -715,21 +705,26 @@ bare subject with no reason to care (`Beijing weather data.`).
715
705
  5. Everything in the paragraph below still applies.
716
706
 
717
707
  Rules that keep it readable: short sentences; no sentence over about twenty-five words; bullets
718
- rather than comma lists; no tables, no code blocks, no links to internal tools, no headings beyond
708
+ rather than comma lists; no bold labels, em dashes, semicolons or fake quotations; no tables, no
709
+ code blocks, no links to internal tools, no headings beyond
719
710
  the ones above. Spell out the names a reader would search for -- the place, the identifiers in
720
711
  every common form, the programme and the publisher -- once each, under Coverage and under
721
712
  Source and rights, and never as a keyword list. A one-line description is a defect to fix in the
722
713
  revision, not a style choice. Give `table.description` the grain and the window.
714
+ Every statement in the opening and body must preserve an established fact. Never add unsupported
715
+ coverage, freshness, quality, licensing or intended use.
723
716
 
724
717
  **Check before writing.** Fix and re-check until every line passes:
725
718
  - title: at most 60 characters, contains the place, contains no colon, pipe or exclamation mark,
726
719
  no brand, no "dataset" or "data", no version number, no run date
727
- - opening paragraph: two or three sentences, each under about twenty words, 120 to 280 characters,
728
- first full stop before character 110, no Markdown, no identifier, publisher name, row count or
729
- row definition, no banned word, no clause about what a reader might do
720
+ - opening paragraph: at least two distinct sentences, with the concrete records first and the
721
+ question or model they support next; include known geography, time range, grain, update cadence
722
+ and material limitations; no Markdown, banned style or invented fact
730
723
  - category: exactly one fixed ID saved and verified before recipe registration
731
724
  - topics: 3 to 8 descriptive tags, no duplicates, lowercase, at most 40 characters each
732
725
  - licence: an SPDX identifier the sources actually grant, or `--license` left off
726
+ - whole description: every statement preserves an established fact; no bold labels, em dashes,
727
+ semicolons, fake quotations or unsupported claims
733
728
  - every `##` heading in DESCRIPTION.md is one of the five, in that relative order, none repeated
734
729
 
735
730
  **Every dataset gets a cover before its first run is presented.** Once the description is
@@ -938,20 +933,23 @@ finished work rather than a shape test. Where a source was never opened, and its
938
933
  a guess, two thousand rows a source is the fallback: it exercises every column, every cast and
939
934
  every declared check, and it comes back while the person who asked is still watching the page.
940
935
 
941
- **One Reader holds about 3 MiB of one source at a time, and that bounds what a source DELIVERS.**
942
- It is the largest single source rather than the total — one Reader opens one source at a time — and
943
- a source that really delivers more than that dies mid-run with `SANDBOX_MEMORY_LIMIT`. This is
944
- yours to apply to the sizes stage 2 measured, not something the command decides for you:
936
+ **What bounds a source is how much it DECODES, not how much it weighs.** One confined decode may
937
+ hand back 6,845,533 bytes, and a source over that is refused with `READER_BUDGET`, which names both
938
+ sizes. The two quantities come apart constantly: a 10.4 MB document projected to a few columns
939
+ decodes small and runs, while an 8.6 MB table read whole does not, and two members of one 3.6 MB
940
+ archive land on opposite sides. So the fetched size predicts nothing on its own.
941
+
945
942
  `limits.max_source_bytes` states where fetching STOPS rather than how big the source is, so
946
943
  `mr-data recipe` does not refuse a large one. It registers the document and puts one sentence on
947
- the receipt under `warnings`, naming the source and both numbers. Read that as "go and check this
948
- one", not as a failure: a document declaring 8 MiB over a feed that answers 40 KB is fine.
949
-
950
- **If a source really does deliver more than the Reader holds, do not answer it by lowering the
951
- ceiling on that source:** that does not fetch less data, it truncates the sample or refuses the
952
- full run. Split the window across several sources — one per month, per region, or per whatever the
953
- publisher paginates on — and union them in a first transform step. A recipe names up to 256
954
- sources, which is a great many windows. [The recipe document](https://mostlyright.md/docs/reference/recipe/)
944
+ the receipt under `warnings` when the declared number is larger than any decode could return. Read
945
+ that as "go and check this one", not as a failure.
946
+
947
+ **Which remedy applies depends on which way the source is too big, and lowering the ceiling is
948
+ never it** — that does not fetch less data, it truncates the sample or refuses the full run. A WIDE
949
+ source, many columns or long text, is narrowed by projecting fewer columns in the reader's own
950
+ settings; splitting it would not help at all. A LONG source is split across several sources — one
951
+ per month, per region, or per whatever the publisher paginates on — and unioned in a first
952
+ transform step. A recipe names up to 256 sources. [The recipe document](https://mostlyright.md/docs/reference/recipe/)
955
953
  carries the measured numbers behind the ceiling.
956
954
 
957
955
  **A corpus of many pages behind one index is a COLLECTION, not many sources.** See
@@ -69,31 +69,24 @@ REQUIRED_DOCUMENT_MEMBERS: tuple[str, ...] = (
69
69
 
70
70
  #: The declared ``limits.max_source_bytes`` above which one source is worth a word, in bytes.
71
71
  #:
72
- #: WHY THE NUMBER IS WORTH SAYING, AND WHY THIS ONE. The Reader that opens a source's bytes runs
73
- #: inside one Clean room with a fixed memory allowance, and it opens ONE source at a time -- so
74
- #: what decides whether a recipe can run is the largest single source in it, not the total. The
75
- #: number is measured rather than reasoned: a source that DELIVERED 5.76 MB died mid-run with
76
- #: ``SANDBOX_MEMORY_LIMIT``, and one that delivered 1.15 MB was accepted and completed. 3 MiB sits
77
- #: between the two with room on both sides.
72
+ #: ⚠ THIS NUMBER BOUNDS THE FETCH, AND WHAT REFUSES IS THE DECODE. The two are different
73
+ #: quantities and the old value conflated them. 3 MiB came from two observations -- a source that
74
+ #: delivered 5.76 MB and died, one that delivered 1.15 MB and lived -- and neither was a
75
+ #: measurement of anything the run enforces. Measured properly: a 10,449,878-byte JSON source
76
+ #: projected to five columns SUCCEEDS, while an 8,578,215-byte CSV read whole FAILS, and two
77
+ #: members of ONE 3.6 MB archive land on opposite sides -- so the fetched size, which is what this
78
+ #: member states, cannot order the outcomes at all.
78
79
  #:
79
- #: ⚠ AND IT IS A WARNING RATHER THAN A REFUSAL, BECAUSE ``max_source_bytes`` IS NOT A SIZE. It is
80
- #: the point at which the Reader STOPS fetching -- a refusal ceiling, stated by whoever wrote the
81
- #: document -- so a document that writes 8 MiB over a feed answering 40 KB is a document that runs
82
- #: and delivers. Refusing on the declared number would therefore refuse working recipes for
83
- #: something they have not done, and it did: three of this repository's own acceptance recipes
84
- #: declare between 4 and 16 MiB, and every one of them was built, run and delivered. What the
85
- #: ceiling is genuinely about is the DELIVERED size, which this side of the wire cannot know before
86
- #: the run. So the command says so and submits, and the sentence is addressed to the agent that
87
- #: wrote the number, because it is the one party that can go and measure the source.
80
+ #: WHAT ORDERS THEM is how much the Reader DECODES, because the confined decode hands its output
81
+ #: back over a bounded channel and carries it twice there: base64 and as parsed rows. That
82
+ #: ceiling is ``DELIVERABLE_READER_OUTPUT_BYTES`` on the backend, 6,845,533 bytes, and a source
83
+ #: over it is refused by name with ``READER_BUDGET`` naming both sizes.
88
84
  #:
89
- #: ⚠ THE REMEDY IS NOT A SMALLER CEILING ON THE SAME SOURCE. Lowering ``max_source_bytes`` under a
90
- #: window that genuinely holds more bytes does not make the source smaller; on `full` it turns the
91
- #: run into a ``CLAMP_EXCEEDED`` refusal, and on `sample` it truncates, which is worse because a
92
- #: truncated source looks like a complete one until somebody reads ``coverage``. The remedy is to
93
- #: SPLIT the window across several sources -- one per month, per region, or per whatever the
94
- #: publisher paginates on -- and union them in a first transform step, which is what the warning
95
- #: says.
96
- READER_SOURCE_BYTE_CEILING = 3 * 1024 * 1024
85
+ #: So this stays a warning about a DECLARED ceiling being implausibly large, which is worth one
86
+ #: sentence, and it no longer promises to predict a failure it cannot see. It is set at the
87
+ #: backend's own output ceiling: a document declaring more than one decode can return is a
88
+ #: document whose author has not measured the source, whatever the source turns out to weigh.
89
+ READER_SOURCE_BYTE_CEILING = 6_845_533
97
90
 
98
91
  #: How many undescribed columns the warning names before it counts the rest. A table plan admits
99
92
  #: 512 columns, and a sentence that spelt every one of them out would be the wall the `numbered`
@@ -215,11 +208,10 @@ def source_byte_warnings(sources: Any) -> list[str]:
215
208
  will not.
216
209
 
217
210
  ⚠ AND THE SENTENCE NAMES THE FIX, which is the whole reason it is written here rather than left
218
- to a mid-run ``SANDBOX_MEMORY_LIMIT``. That failure costs a registration, a run and the minutes
219
- before it died, and it leaves somebody to guess whether the answer is a smaller ceiling (it is
220
- not -- that truncates or refuses) or a different source (usually not either). Splitting the
221
- window into several sources and unioning them in a first transform step is the answer nearly
222
- every time, so it is said.
211
+ to a mid-run refusal. It also says what the fix is NOT: a smaller declared ceiling, which
212
+ truncates the fetch rather than decoding less. Narrowing what the reader projects is the answer
213
+ when the source is wide; splitting the source is the answer when it is long. Splitting a wide
214
+ source does nothing, which is why the two cases are named separately.
223
215
 
224
216
  A declared ceiling that is not a whole number of bytes is left alone: the document admits no
225
217
  fractional number at all, ``read_document`` has already refused one, and re-deciding here what
@@ -241,16 +233,16 @@ def source_byte_warnings(sources: Any) -> list[str]:
241
233
  named = source.get("name")
242
234
  where = f"{named!r}" if isinstance(named, str) and named else f"sources[{index}]"
243
235
  said.append(
244
- f"source {where} declares limits.max_source_bytes of {declared:,} bytes, and one "
245
- f"Reader has been measured to hold about {READER_SOURCE_BYTE_CEILING:,}. That is a "
246
- "ceiling on fetching rather than a measurement of the source, so this registers and "
247
- "runs -- but if the source really delivers that much, the run dies with "
248
- "SANDBOX_MEMORY_LIMIT. One Reader opens one source at a time in one Clean room, so it "
249
- "is the largest single source that has to fit, not the total. Do not simply lower the "
250
- "ceiling -- that truncates the sample or refuses the full run rather than fetching "
251
- "less data. Measure what the source answers, and if it is over the ceiling: split the "
252
- "window into several sources, one per month or per region or per whatever the "
253
- "publisher paginates on, and union them in a first transform step"
236
+ f"source {where} declares limits.max_source_bytes of {declared:,} bytes, which is "
237
+ f"more than one confined decode can hand back ({READER_SOURCE_BYTE_CEILING:,}). That "
238
+ "is a ceiling on fetching rather than a measurement of the source, so this registers "
239
+ "and runs. What decides the run is how much the Reader DECODES, not how much is "
240
+ "fetched: a wide document projected to a few columns decodes small however large it "
241
+ "arrives, and a compact archive member can decode past the ceiling. A source that "
242
+ "does is refused by name with READER_BUDGET, which states both sizes. Lowering this "
243
+ "ceiling does not help, because it bounds the fetch and truncates rather than "
244
+ "decoding less. Narrow the projection, or split the source and union the parts in a "
245
+ "first transform step"
254
246
  )
255
247
  return said
256
248
 
@@ -508,6 +508,39 @@ def _emit(payload: dict[str, Any], *, as_json: bool) -> None:
508
508
  print(render_human(payload), flush=True)
509
509
 
510
510
 
511
+ #: How many of Studio's own pointer-and-reason details one refusal renders before it counts the
512
+ #: rest. A document refused for one reason is the ordinary case; a document refused for forty is a
513
+ #: document to fix a few at a time, and forty paragraphs is the wall a person stops reading at.
514
+ MAX_RENDERED_DETAILS = 8
515
+
516
+
517
+ def _refusal_details(refusal: BaseException) -> tuple[str, ...]:
518
+ """The sentences Studio wrote about WHICH part of a document it refused.
519
+
520
+ Studio answers a refused write with one entry per fault, each carrying a JSON pointer into the
521
+ document and a reason -- "chart 'share' describes boolean columns; 'event_type' is declared
522
+ string" names the column and the fix in one line. ``_studio_refusal`` already carries them onto
523
+ the refusal; nothing rendered them, so every one of those refusals reached a person as
524
+ "The recipe document is internally inconsistent." and nothing else, and the only way to find
525
+ out which of five hundred members was wrong was to bisect the document by hand.
526
+
527
+ Rendered rather than interpreted: the pointer is Studio's and the reason is Studio's, and a
528
+ client that paraphrased either would be a second opinion about a document it did not validate.
529
+ """
530
+
531
+ details = getattr(refusal, "details", ()) or ()
532
+ said: list[str] = []
533
+ for entry in details:
534
+ if not isinstance(entry, Mapping):
535
+ continue
536
+ reason = entry.get("reason")
537
+ if not isinstance(reason, str) or not reason:
538
+ continue
539
+ pointer = entry.get("pointer")
540
+ said.append(f"{pointer}: {reason}" if isinstance(pointer, str) and pointer else reason)
541
+ return tuple(said[:MAX_RENDERED_DETAILS])
542
+
543
+
511
544
  def _refuse(
512
545
  code: str,
513
546
  detail: str,
@@ -515,6 +548,7 @@ def _refuse(
515
548
  as_json: bool,
516
549
  retry_after_seconds: int | None = None,
517
550
  extra: Mapping[str, Any] | None = None,
551
+ details: Sequence[str] = (),
518
552
  ) -> int:
519
553
  """Report one refusal the way every other ``mr-data`` refusal is reported.
520
554
 
@@ -543,6 +577,8 @@ def _refuse(
543
577
  }
544
578
  if retry_after_seconds is not None:
545
579
  error["retry_after_seconds"] = retry_after_seconds
580
+ if details:
581
+ error["details"] = list(details)
546
582
  if extra:
547
583
  error.update(extra)
548
584
  print(
@@ -551,7 +587,14 @@ def _refuse(
551
587
  flush=True,
552
588
  )
553
589
  else:
554
- print(render_error(detail, code=code, remediation=remediation), file=sys.stderr, flush=True)
590
+ # Studio's own sentences come FIRST, before the generic remediation: they name the member
591
+ # and the fix, and burying them under three paragraphs of lane advice is how a refusal
592
+ # that said exactly what was wrong reached a person saying nothing.
593
+ print(
594
+ render_error(detail, code=code, remediation=tuple(details) + tuple(remediation)),
595
+ file=sys.stderr,
596
+ flush=True,
597
+ )
555
598
  return 2
556
599
 
557
600
 
@@ -789,6 +832,7 @@ def _hosted(command: str, arguments: Sequence[str], *, as_json: bool) -> int:
789
832
  refusal.detail,
790
833
  as_json=as_json or parsed.json,
791
834
  retry_after_seconds=getattr(refusal, "retry_after_seconds", None),
835
+ details=_refusal_details(refusal),
792
836
  )
793
837
  except Exception as error:
794
838
  code = getattr(error, "code", None)