pytest-given 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. {pytest_given-0.3.0 → pytest_given-0.4.0}/AGENTS.md +10 -5
  2. {pytest_given-0.3.0 → pytest_given-0.4.0}/CHANGELOG.md +38 -1
  3. {pytest_given-0.3.0 → pytest_given-0.4.0}/GLOSSARY.md +3 -3
  4. {pytest_given-0.3.0 → pytest_given-0.4.0}/PKG-INFO +2 -1
  5. {pytest_given-0.3.0 → pytest_given-0.4.0}/noxfile.py +22 -1
  6. {pytest_given-0.3.0 → pytest_given-0.4.0}/pyproject.toml +9 -2
  7. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-authoring/SKILL.md +2 -2
  8. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-authoring/references/api.md +7 -9
  9. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-authoring/references/domain-storytelling.md +0 -2
  10. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-authoring/references/glossaries.md +8 -8
  11. pytest_given-0.4.0/src/pytest_given/.agents/skills/pytest-given-authoring/references/scenarios.md +105 -0
  12. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-authoring/references/stories.md +3 -4
  13. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-navigating/references/report-json.md +6 -1
  14. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-reviewing/SKILL.md +12 -10
  15. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-reviewing/references/pairs.md +1 -0
  16. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-reviewing/references/story-coverage.md +6 -1
  17. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/collector.py +39 -31
  18. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/file_glossary.py +15 -10
  19. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/glossary.py +6 -10
  20. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/kind_inference.py +5 -4
  21. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/markdown_glossary.py +3 -5
  22. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/process_state.py +3 -6
  23. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/scenario.py +6 -3
  24. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/source.py +7 -14
  25. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/steps.py +19 -25
  26. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/template.py +5 -7
  27. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/traceback.py +17 -12
  28. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/cli/report.py +7 -3
  29. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/cli/skills.py +1 -5
  30. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/attachments.py +0 -3
  31. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/checks.py +9 -9
  32. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/columns.py +44 -24
  33. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/context.py +2 -5
  34. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/group.py +23 -9
  35. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/ast_rules.py +2 -7
  36. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/runtime_rules.py +20 -23
  37. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/errors.py +3 -4
  38. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/schema.py +16 -4
  39. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/serde.py +10 -25
  40. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/collection.py +3 -7
  41. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/fixtures.py +68 -73
  42. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/options.py +5 -10
  43. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/runtest.py +65 -25
  44. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/session.py +16 -25
  45. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/state.py +9 -24
  46. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/__init__.py +2 -0
  47. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/coverage.py +5 -11
  48. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/glossary_view.py +12 -24
  49. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/html_renderer.py +68 -17
  50. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/inline_markdown.py +3 -4
  51. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/md_renderer.py +41 -19
  52. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/palette.py +1 -7
  53. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/sinks.py +10 -2
  54. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/slugs.py +4 -4
  55. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/source_link.py +8 -4
  56. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/story_view.py +10 -6
  57. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/_macros.html.j2 +13 -4
  58. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/app.js +161 -57
  59. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/report.html.j2 +106 -87
  60. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/styles.css +382 -202
  61. pytest_given-0.4.0/src/pytest_given/report/text.py +24 -0
  62. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_cli.py +3 -2
  63. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_plugin.py +659 -122
  64. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_plugin_lint.py +20 -14
  65. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_collector.py +94 -38
  66. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_file_glossary.py +110 -64
  67. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_glossary.py +72 -78
  68. pytest_given-0.4.0/tests/unit/capture/test_kind_inference.py +240 -0
  69. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_markdown_glossary.py +21 -61
  70. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_step_descriptor.py +69 -66
  71. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_story.py +130 -153
  72. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_template.py +57 -95
  73. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_traceback_parser.py +26 -6
  74. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/lint/test_ast_rules.py +18 -9
  75. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/lint/test_runtime_rules.py +17 -2
  76. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_serde.py +16 -2
  77. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_coverage.py +118 -163
  78. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_glossary_view.py +29 -26
  79. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_html_renderer.py +118 -39
  80. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_md_renderer.py +155 -35
  81. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_palette.py +3 -2
  82. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_sinks.py +2 -1
  83. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_slugs.py +13 -0
  84. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_source_link.py +31 -32
  85. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_story_view.py +54 -78
  86. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_grouping.py +219 -36
  87. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_plugin.py +46 -8
  88. pytest_given-0.4.0/uv.lock +1206 -0
  89. pytest_given-0.3.0/src/pytest_given/.agents/skills/pytest-given-authoring/references/scenarios.md +0 -81
  90. pytest_given-0.3.0/src/pytest_given/report/text.py +0 -8
  91. pytest_given-0.3.0/tests/unit/capture/test_kind_inference.py +0 -331
  92. pytest_given-0.3.0/uv.lock +0 -1027
  93. {pytest_given-0.3.0 → pytest_given-0.4.0}/.gitignore +0 -0
  94. {pytest_given-0.3.0 → pytest_given-0.4.0}/LICENSE.md +0 -0
  95. {pytest_given-0.3.0 → pytest_given-0.4.0}/README.md +0 -0
  96. {pytest_given-0.3.0 → pytest_given-0.4.0}/THIRD-PARTY-LICENSES +0 -0
  97. {pytest_given-0.3.0 → pytest_given-0.4.0}/conftest.py +0 -0
  98. {pytest_given-0.3.0 → pytest_given-0.4.0}/docs/site/configuration/narration-lint.md +0 -0
  99. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/.agents/skills/pytest-given-navigating/SKILL.md +0 -0
  100. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/__init__.py +0 -0
  101. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/__init__.py +0 -0
  102. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/discovery.py +0 -0
  103. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/capture/story.py +0 -0
  104. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/cli/__init__.py +0 -0
  105. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/__init__.py +0 -0
  106. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/percase.py +0 -0
  107. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/grouping/templatize.py +0 -0
  108. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/__init__.py +0 -0
  109. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/base.py +0 -0
  110. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/config.py +0 -0
  111. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/runner.py +0 -0
  112. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/lint/summary.py +0 -0
  113. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/__init__.py +0 -0
  114. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/narration.py +0 -0
  115. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/runtime.py +0 -0
  116. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/steps.py +0 -0
  117. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/model/text.py +0 -0
  118. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/plugin/__init__.py +0 -0
  119. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/py.typed +0 -0
  120. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/alpine.min.js +0 -0
  121. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/fonts/source-code-pro-latin-wght-normal.woff2 +0 -0
  122. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/fonts/source-sans-3-latin-wght-normal.woff2 +0 -0
  123. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/templates/logo.svg +0 -0
  124. {pytest_given-0.3.0 → pytest_given-0.4.0}/src/pytest_given/report/theme.py +0 -0
  125. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/__init__.py +0 -0
  126. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/conftest.py +0 -0
  127. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/__init__.py +0 -0
  128. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_plugin_file_glossary.py +0 -0
  129. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_plugin_session_isolation.py +0 -0
  130. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/integration/test_skills_cli.py +0 -0
  131. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/ubiquitous_language.py +0 -0
  132. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/__init__.py +0 -0
  133. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/__init__.py +0 -0
  134. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_discovery.py +0 -0
  135. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_rootdir_isolation.py +0 -0
  136. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/capture/test_source.py +0 -0
  137. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/lint/__init__.py +0 -0
  138. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/lint/test_config.py +0 -0
  139. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/lint/test_summary.py +0 -0
  140. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/__init__.py +0 -0
  141. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_errors.py +0 -0
  142. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_narration.py +0 -0
  143. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_schema.py +0 -0
  144. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_steps.py +0 -0
  145. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/model/test_text.py +0 -0
  146. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/__init__.py +0 -0
  147. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_inline_markdown.py +0 -0
  148. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/report/test_theme.py +0 -0
  149. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_percase.py +0 -0
  150. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_plugin_traceback.py +0 -0
  151. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_skills_data.py +0 -0
  152. {pytest_given-0.3.0 → pytest_given-0.4.0}/tests/unit/test_skills_scripts.py +0 -0
@@ -19,15 +19,19 @@ uv sync --group dev
19
19
  `uv run nox` runs the default gate — `format`, `lint`, `mypy`, `test`, `coverage` (a 100% target), `audit` (a `pip-audit` of the locked dependencies). **Run it (or at minimum `uv run nox -s format lint mypy test`) before every commit.** The sessions below are on-demand; list them all with `uv run nox -l`.
20
20
 
21
21
  - `uv run nox -s examples` regenerates the JSON, HTML, and Markdown files under `examples/coffeeshop/`, `examples/hotel-booking/`, and `examples/file-glossary-booking/`. Run after changes to the renderer, templates, plugin output schema, or any example test file, and commit the updated outputs.
22
- - `uv run nox -s self_report` regenerates `examples/self-report/` — pytest-given applied to its own backend tests (see [Writing self-report scenarios](#writing-self-report-scenarios)). Run after decorating more tests or changing decorated ones, and commit the updated outputs.
22
+ - `uv run nox -s self_report` regenerates `examples/self-report/` — pytest-given applied to its own backend tests (see [Writing self-report scenarios](#writing-self-report-scenarios)). Run after decorating more tests or changing decorated ones, and after renderer or template changes (its HTML is one of the site's example reports), and commit the updated outputs.
23
23
  - **Only commit a regenerated report when its *content* actually changed.**
24
24
  - Every regeneration rewrites `commit_sha` (to current HEAD, including the SHA-pinned source-link URLs), `timestamp`, and `duration_ms` in the JSON and HTML, so a report your change didn't really touch still shows a diff — `git checkout` those files rather than committing the noise.
25
25
  - **Read the `.md` diff first**: the Markdown carries none of those fields, so it is the behavioral delta of your change in prose. An unchanged `.md` doesn't by itself prove the JSON/HTML are noise-only (glossary and story data never surface in the Markdown); a shifted source line does show up, in the `relpath:line::test_name` anchor under every heading.
26
- - Regenerate only the reports a change can affect — `examples` narrates `examples/**`, `self_report` narrates `tests/**`. A shifted line number in a decorated backend test is therefore a real self-report change worth committing, even when no example changed.
26
+ - Regenerate only the reports a change can affect — `examples` narrates `examples/**`, `self_report` narrates `tests/**`, and a renderer or template change affects all of them. A shifted line number in a decorated backend test is therefore a real self-report change worth committing, even when no example changed.
27
27
  - Both regeneration sessions run the narration lint (`--given-lint`; see [Narration lint](docs/site/configuration/narration-lint.md) and the [design spec](docs/specs/2026-07-05-narration-lint-design.md)). `self_report` fails on any lint error — a real gate, keep the backend suite lint-clean. `examples` tolerates exit 1 for its intentional failures, which masks the lint exit code, so read the printed "narration lint" summary there. A step the lint mis-flags goes on the `given_lint_ignore` list (stale entries fail the run); the rule catalog and ignore mechanics are in the [authoring skill](src/pytest_given/.agents/skills/pytest-given-authoring/references/scenarios.md) under "Mechanical counterparts", and the honest-two-phase test an ignored `missing-phase` has to pass under "Phase structure".
28
- - `uv run nox -s benchmark` generates the large-scenarios suite and renders its JSON + HTML into `benchmarks/` (gitignored). Run it when a change could move report-generation cost; `benchmarks/bench.py` does size sweeps and cProfile runs directly.
28
+ - `uv run nox -s benchmark` generates the large-scenarios suite and renders its JSON + HTML into `benchmarks/` (outputs gitignored). Run it when a change could move report-generation cost; `benchmarks/bench.py` does size sweeps and cProfile runs directly.
29
+ - `uv run nox -s compat` runs the suite on Python 3.14 and 3.15, each against pytest 9.0.x (the dependency floor) and the newest pytest. The default gate tests only the locked pytest on 3.14, so this is the only check of the floor and of 3.15. The manually dispatched [Compat workflow](.github/workflows/compat.yml) runs it on Linux and Windows before every release (see [docs/releasing.md](docs/releasing.md)).
29
30
  - `uv run nox -s build` builds the wheel + sdist and verifies them as a consumer would — `py.typed`, templates and bundled skills present, a real scenario run from a throwaway install, and [library-skills](https://library-skills.io) discovery of every bundled skill. The in-repo suite imports from `src/`, so this session is the only thing that catches a packaging regression. CI and the release workflow both run it.
30
- - `uv run nox -s docs_build` builds the documentation site under `site/` with `zensical build --strict`, after copying the four example reports, the full-size diagram and the report package's two upright fonts into gitignored spots under `docs/site/` (Zensical builds every file in `docs/site/` and cannot exclude any, so those copies are how single-sourced files get in; `CHANGELOG.md` instead reaches the site through a `pymdownx.snippets` include in `docs/site/changelog.md`, between its `site` markers, and the Home-page diagram is inlined the same way from `docs/site/assets/pytest-given-diagram.svg` so page CSS can theme it). The README shows the same diagram as `docs/pytest-given-diagram.svg`, a derived copy with the font embedded and the colour tokens resolved, which the Home page also links to as its full-size view: after editing the source SVG or its tokens, run `uv run nox -s diagram` and commit the result. Run it after changing anything under `docs/site/`; a broken cross-page link fails it. Preview with `uv run --group docs zensical serve` afterwards. The site self-hosts its fonts (`docs/site/assets/fonts/`, latin subsets from Fontsource, OFL notices in `LICENSES.txt` there; the upright faces are the report package's own under `src/pytest_given/report/templates/fonts/`, staged in by the build, only the italics live in the site tree) so a visitor's browser contacts no third party — keep `font = false` in `zensical.toml` and don't `@import` a font CDN; a new face gets its `.woff2` and copyright line added there. `docs_deploy` is the CI-only counterpart that publishes with mike — see [docs/releasing.md](docs/releasing.md#documentation-site).
31
+ - `uv run nox -s docs_build` builds the documentation site under `site/` with `zensical build --strict`; run it after changing anything under `docs/site/` (a broken cross-page link fails it) and preview with `uv run --group docs zensical serve`. `docs_deploy` is the CI-only counterpart that publishes with mike — see [docs/releasing.md](docs/releasing.md#documentation-site).
32
+ - Zensical builds every file in `docs/site/` and cannot exclude any, so single-sourced files get in by copy or include: the build copies the four example reports, the full-size diagram, and the report package's upright fonts and logo into gitignored spots there; `CHANGELOG.md` (between its `site` markers) and the Home-page diagram (`docs/site/assets/pytest-given-diagram.svg`, inlined so page CSS can theme it) are `pymdownx.snippets` includes.
33
+ - `docs/pytest-given-diagram.svg`, the README's diagram and the Home page's full-size view, is derived with the font embedded and the colour tokens resolved: after editing the source SVG or its tokens, run `uv run nox -s diagram` and commit the result.
34
+ - The site self-hosts its fonts so a visitor's browser contacts no third party — keep `font = false` in `zensical.toml` and don't `@import` a font CDN. Only the italics live in `docs/site/assets/fonts/` (latin subsets from Fontsource); a new face gets its `.woff2` there and its copyright line in that folder's `LICENSES.txt`.
31
35
 
32
36
  ## Releasing
33
37
 
@@ -98,6 +102,7 @@ The narration rules live in the **`pytest-given-authoring` skill**, whose canoni
98
102
  **The skill is documentation with the same sync duty as the site** — downstream agents read it instead of the site, version-matched from the wheel. A change to the public API surface or its rules updates the site page under `docs/site/` *and* [references/api.md](src/pytest_given/.agents/skills/pytest-given-authoring/references/api.md); a change to narration/lint semantics updates [references/scenarios.md](src/pytest_given/.agents/skills/pytest-given-authoring/references/scenarios.md) and friends. No check catches prose drift between the site and the skill, so ask "does the skill need this too?" on every user-facing change; only the `python` blocks are covered — `tests/unit/test_skills_scripts.py` runs each against a report built from the model. The reviewing skill restates authoring rules as rubrics, so they share the same duty: its layer 2 mirrors `scenarios.md` ("Keeping it truthful", "Expected raises", "Vocabulary and tags"), and its layer 4 mirrors `glossaries.md` ("Keeping the glossary honest") and `stories.md` — change one, change both.
99
103
 
100
104
  Narration drifts like any other prose — the lint catches several ways step text and body can part company, not all of them — which is why [Quality gates](#quality-gates) has you read the regenerated `.md` rather than trust the text.
105
+
101
106
  Before merging changes to narrated tests, run the `pytest-given-reviewing` skill: it layers the lint, a semantic audit of step text against step bodies, and a hygiene pass over glossary, tags and stories.
102
107
 
103
108
  What is specific to this repo's self-report:
@@ -123,5 +128,5 @@ What is specific to this repo's self-report:
123
128
  - Keep commits coherent: each commit should represent one logical change. Don't split "do X", "tests for X", and "review-fixup for X" into separate commits — squash them before pushing. Don't bundle unrelated changes either.
124
129
  - **A user-facing change adds its `CHANGELOG.md` entry in the same commit**, under `## [Unreleased]`, in the fitting Keep a Changelog category (each category appears at most once per version — extend the existing heading rather than adding a second one). User-facing = public API, CLI flags, report output, lint rules, bundled skills; internal work (refactors, tests, CI, contributor docs) gets no entry.
125
130
  - **One sentence per entry**, written for someone upgrading: name the symbol, flag, or surface and say what changed — no rationale, measurements, or before/after detail. Only a breaking change earns more: the migration it needs. Visual and interaction polish gets one collective bullet per release ("the sidebar and its chips are visually tidied"), never a bullet per restyled element; accessibility fixes stay on their own line. When in doubt, the shorter entry is the right one.
126
- - Plan files under `docs/superpowers/plans/` are scratch artifacts — never commit them. Spec files under `docs/specs/` are committed.
131
+ - Plan files under `docs/superpowers/plans/` are gitignored scratch; spec files under `docs/specs/` are committed.
127
132
  - New specs land under `docs/specs/proposed/`. When a spec's implementation lands, `git mv` it up one level into `docs/specs/` in the same commit, and fix its relative links in the same edit — a `../`-prefixed link to a sibling spec resolves into `docs/` once the file moves. `ls docs/specs/proposed` is the canonical list of outstanding design work.
@@ -11,6 +11,42 @@ form `## [x.y.z] - YYYY-MM-DD`.
11
11
 
12
12
  <!-- --8<-- [start:site] -->
13
13
 
14
+ ## [0.4.0] - 2026-10-04
15
+
16
+ ### Added
17
+
18
+ - Python 3.15 is supported.
19
+ - Hovering a phase in the HTML report's narration, or a column of its parameter table, highlights that phase in both.
20
+ - Scenarios that fail as expected (`xfail`) get their own `xfailed` status, with their reason, steps and error, in every report format.
21
+
22
+ ### Changed
23
+
24
+ - The parameter table orders its columns the way the narration first shows them, and shows its status column only when the cases differ in status.
25
+ - The HTML report's Stories view shows each scenario as a full card that expands in place, and a tag clicked there opens the Scenarios view filtered by it.
26
+ - The HTML report is visually tidied: term refs, the Glossary view, story coverage, narration spacing, hover states and the header row are restyled, and a parametrized scenario is marked by a second status bar.
27
+ - The bundled `pytest-given-authoring` skill asks for one scenario per rule and covers parametrized scenarios as decision tables, and the `pytest-given-reviewing` skill catches more ways scenarios, glossary rows, pins, tags and parameter tables can disagree.
28
+
29
+ ### Fixed
30
+
31
+ - A `@given` fixture parametrized with an unhashable value, such as a list, no longer errors every test that requests it.
32
+ - Editor source links build `{path}` from pytest's rootdir instead of the working directory, so they no longer point at a doubled path when pytest runs from a subdirectory.
33
+ - A scenario deep link stays working when the test's file or directory name contains `+`, `&`, `=` or `#`.
34
+ - A parametrize value of infinity or NaN reaches the JSON report as a string (`"inf"`, `"nan"`), so the report stays valid JSON.
35
+ - A `@given` generator fixture that yields twice errors with pytest's own "more than one 'yield'" message, as it does without pytest-given, instead of passing silently.
36
+ - An `Annotated[..., given(Template(...))]` label whose placeholder is not a bare parametrize column name now fails its scenario with the fix, instead of crashing the HTML report or silently dropping the step.
37
+ - A scenario failing on a pytest-given refusal raised from its test body points at the test's own line, not at pytest-given's.
38
+ - Error messages are clearer: a `FileGlossary` table error names the file, so does a glossary file that is a directory or not UTF-8, and a step nested across phases gets a concrete fix suggested.
39
+ - The Markdown report's note below a parameter table names the row by its short parametrize values or its row number, instead of repeating multiline values.
40
+ - A failure inside the narration lint is summarized under `pytest-given: narration lint failed`, no longer as `report not written` beside a report that was kept.
41
+ - The `tag-shadows-term` lint finding counts a scenario once when it carries two spellings of the tag, such as `Guest` and `guest`.
42
+ - `pytest-given report -o report.markdown` writes Markdown, as `-o report.md` does, instead of refusing the path as an HTML one.
43
+ - `pytest-given report --help` mentions Markdown output and says where each format goes without `-o`.
44
+ - The HTML report's links keep a tag filter whose tag contains a comma, by repeating the parameter per tag or term (`#tag=a&tag=b`), and copying a link no longer replaces the current page's entry in the browser history.
45
+ - The HTML report no longer says "All Scenarios" when every status is filtered out, disables a status filter no scenario has, and the Glossary view says no terms match when every kind is unchecked.
46
+ - The HTML report's view tabs and search boxes show a visible keyboard focus ring, and a turned-off status filter keeps readable contrast.
47
+ - The HTML report's status filters wrap onto a second row instead of being cut off in a narrow sidebar.
48
+ - A parametrized scenario shows a skip reason only when every case was skipped, no longer its first case's reason beside cases that ran.
49
+
14
50
  ## [0.3.0] - 2026-09-27
15
51
 
16
52
  ### Added
@@ -352,7 +388,8 @@ First public release.
352
388
  - Bundled authoring, navigating, and reviewing skills for AI agents, shipped in
353
389
  the wheel and version-matched to the plugin.
354
390
 
355
- [Unreleased]: https://github.com/nwilbert/pytest-given/compare/v0.3.0...HEAD
391
+ [Unreleased]: https://github.com/nwilbert/pytest-given/compare/v0.4.0...HEAD
392
+ [0.4.0]: https://github.com/nwilbert/pytest-given/compare/v0.3.0...v0.4.0
356
393
  [0.3.0]: https://github.com/nwilbert/pytest-given/compare/v0.2.0...v0.3.0
357
394
  [0.2.0]: https://github.com/nwilbert/pytest-given/compare/v0.1.0...v0.2.0
358
395
  [0.1.0]: https://github.com/nwilbert/pytest-given/releases/tag/v0.1.0
@@ -33,7 +33,7 @@ This glossary covers pytest-given's own bounded context. The terminology a *user
33
33
  | Term | Meaning |
34
34
  |---|---|
35
35
  | **Step fixture** | A pytest fixture whose function is wrapped with `@given(text)`. Only `@given` is allowed on fixtures; `@when` / `@then` are rejected. |
36
- | **Plain fixture** | A pytest fixture without a pytest-given decorator. Used by tests but produces no step in the report. |
36
+ | **Plain fixture** | A pytest fixture without a pytest-given decorator. It produces no step in the report unless an `Annotated[..., given(...)]` label on the test's parameter narrates it. |
37
37
  | **Fixture recording** | A captured subtree of steps + attachments produced while a step fixture is being set up. Stored keyed by fixture-instance identity. A generator fixture's teardown records nothing — it refuses steps and attachments. |
38
38
  | **Graft** | Attaching a fixture recording into the active scenario's step tree at the moment its host test starts. |
39
39
 
@@ -79,8 +79,8 @@ Terms follow [Domain Storytelling](https://domainstorytelling.org/quick-start-gu
79
79
  | **Deferred term** | A term handed over before its kind is settled — what `g('foo')` and `g['foo']` on a code glossary return, and every *file glossary* lookup. The handle is the same `TermHandle` the typed registrations (`g.actor(...)` and friends) hand back; `declared_kind is None` is what marks the deferral. The deferral is in the handing over, not the term: a row with an explicit `kind_column` arrives through the same handle already kinded, while the rest stay `None` until *kind inference* runs. |
80
80
  | **Term** | A registered glossary entry: an Actor, Work Object, Activity, or kindless term. Each carries an id (slug), a canonical name, a kind (`None` when kindless), and an optional definition (`str | None`, `None` when undefined). |
81
81
  | **Handle** | The Python object a glossary hands back for a *term* — `TermHandle`, the same type from every accessor (`g.actor(...)`, `g('foo')`, `g['Guest']`, a captured `guest = ...`). It is what steps, scenario titles and sentences interpolate, in one of three surface forms: **bare** (`room` — the canonical display), **`.low`** (`room.low` — lowercased), or **called** (`room('Deluxe Suite')` — an *instance* on an Actor or Work Object, an inflection on an *Activity*). Each form renders as a *term ref*. A story hands out a **sentence handle** the same way: `book_a_room['cancel']` by name or `book_a_room[3]` by number, which `pins=` takes. |
82
- | **Actor** | A glossary term for a participant in the domain (e.g., *Guest*). Carries the actor kind color: a wash in narration, a pill in the Glossary view. |
83
- | **Work Object** | A glossary term for a thing acted on (e.g., *Room*, *Booking*). Carries the work-object kind color: a wash in narration, a pill in the Glossary view. |
82
+ | **Actor** | A glossary term for a participant in the domain (e.g., *Guest*). Carries the actor kind color in the report. |
83
+ | **Work Object** | A glossary term for a thing acted on (e.g., *Room*, *Booking*). Carries the work-object kind color in the report. |
84
84
  | **Activity** | A glossary term for what an actor does (e.g., *book*, *confirm*), which Domain Storytelling draws as an arrow labelled with a verb. Activities accept inflections: calling `book('books')` records *books* as a surface form of the canonical *book*. |
85
85
  | **Term ref** | An occurrence of a term inside narration. Modeled as `NarrationTermRef` in step text and as `ClauseTermRef` inside a clause. |
86
86
  | **Instance** | A named refinement of an Actor or Work Object (e.g., `guest('Alice')` is an instance of the *Guest* actor). Instances aggregate in the Glossary tab's refs block. |
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: pytest-given
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: A pytest plugin that generates interactive HTML reports from Given/When/Then annotated tests.
5
5
  Project-URL: Homepage, https://github.com/nwilbert/pytest-given
6
6
  Project-URL: Repository, https://github.com/nwilbert/pytest-given
@@ -18,6 +18,7 @@ Classifier: Operating System :: OS Independent
18
18
  Classifier: Programming Language :: Python :: 3
19
19
  Classifier: Programming Language :: Python :: 3 :: Only
20
20
  Classifier: Programming Language :: Python :: 3.14
21
+ Classifier: Programming Language :: Python :: 3.15
21
22
  Classifier: Topic :: Software Development :: Documentation
22
23
  Classifier: Topic :: Software Development :: Testing
23
24
  Classifier: Typing :: Typed
@@ -95,6 +95,26 @@ def test(session: nox.Session) -> None:
95
95
  session.run('pytest')
96
96
 
97
97
 
98
+ @nox.session(python=['3.14', '3.15'])
99
+ @nox.parametrize(
100
+ 'pytest_version', ['9.0', 'latest'], ids=['pytest-9.0', 'pytest-latest']
101
+ )
102
+ def compat(session: nox.Session, pytest_version: str) -> None:
103
+ """Run the suite on every supported Python against pytest 9.0.x and the newest.
104
+
105
+ `latest` installs the newest release rather than trusting the lock, which
106
+ only moves on `uv lock --upgrade`. pytest 9.0 predates the `max_warnings`
107
+ option, so there an unknown-option warning replaces the warnings gate.
108
+ """
109
+ _sync(session, 'test', include_project=True)
110
+ if pytest_version == 'latest':
111
+ pytest_install = ['pytest', '--upgrade-package', 'pytest']
112
+ else:
113
+ pytest_install = [f'pytest~={pytest_version}.0']
114
+ session.run('uv', 'pip', 'install', *pytest_install, external=True)
115
+ session.run('pytest')
116
+
117
+
98
118
  @nox.session
99
119
  def coverage(session: nox.Session) -> None:
100
120
  _sync(session, 'coverage', include_project=True)
@@ -184,7 +204,8 @@ def build(session: nox.Session) -> None:
184
204
  assert len(wheels) == 1, f'expected exactly one wheel, got {wheels}'
185
205
  wheel = wheels[0].resolve()
186
206
 
187
- names = zipfile.ZipFile(wheel).namelist()
207
+ with zipfile.ZipFile(wheel) as archive:
208
+ names = archive.namelist()
188
209
  missing = [
189
210
  required
190
211
  for required in _REQUIRED_WHEEL_PATHS
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pytest-given"
3
- version = "0.3.0"
3
+ version = "0.4.0"
4
4
  description = "A pytest plugin that generates interactive HTML reports from Given/When/Then annotated tests."
5
5
  readme = "README.md"
6
6
  # PEP 639: an SPDX expression plus the files it refers to. Deliberately no
@@ -27,6 +27,7 @@ classifiers = [
27
27
  "Programming Language :: Python :: 3",
28
28
  "Programming Language :: Python :: 3 :: Only",
29
29
  "Programming Language :: Python :: 3.14",
30
+ "Programming Language :: Python :: 3.15",
30
31
  "Topic :: Software Development :: Documentation",
31
32
  "Topic :: Software Development :: Testing",
32
33
  "Typing :: Typed",
@@ -115,11 +116,13 @@ extend-exclude = ["benchmarks/test_large_scenarios.py"]
115
116
 
116
117
  [tool.ruff.format]
117
118
  quote-style = "single"
119
+ # Markdown here is generated reports or skill prose whose snippets stay compact.
120
+ exclude = ["*.md"]
118
121
 
119
122
  [tool.ruff.lint]
120
123
  # `explicit-preview-rules` keeps preview mode from pulling in every preview rule
121
124
  # of a selected family: a preview rule is active only when named by its exact
122
- # code below. PLW1514 is the only one that qualifies today.
125
+ # code below: PLW1514 and RUF077 today.
123
126
  preview = true
124
127
  explicit-preview-rules = true
125
128
  select = [
@@ -156,6 +159,8 @@ select = [
156
159
  "BLE", # flake8-blind-except — `except Exception` that swallows everything
157
160
  "PLW1510", # subprocess.run without an explicit check= argument
158
161
  "PLW1514", # open/read_text without encoding=; the locale default is cp1252 on Windows
162
+ "PLR1708", # `raise StopIteration` inside a generator, a RuntimeError when iterated
163
+ "RUF077", # a default on `self`/`cls`, which method binding never uses
159
164
  ]
160
165
 
161
166
  [tool.ruff.lint.per-file-ignores]
@@ -166,6 +171,8 @@ select = [
166
171
 
167
172
  [tool.pytest]
168
173
  testpaths = ["tests"]
174
+ # A warning fails the run, so deprecations surface on the upgrade that adds them.
175
+ max_warnings = "0"
169
176
  given_lint_ignore = [
170
177
  # Honestly two-phase: constructing the descriptor is the action under
171
178
  # test and there is no input to arrange (see the authoring skill's
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: pytest-given-authoring
3
- description: Use when writing or changing @scenario tests, glossary terms, or domain stories in a project that uses pytest-given
3
+ description: Use when writing or changing @scenario tests, glossary terms, or domain stories in a project that uses pytest-given, or converting plain pytest tests into scenarios
4
4
  ---
5
5
 
6
6
  # Authoring pytest-given artifacts
@@ -45,7 +45,7 @@ Each artifact kind stands on its own — scenarios-only is a perfectly good leve
45
45
  ## Verify before committing
46
46
 
47
47
  - Render the touched scenarios and read the output as a spec — every step text must be something its body actually does: `pytest <selection> --given-md`
48
- - If the project enables the narration lint, run it: `pytest <selection> --given-lint`. It catches structural lies (empty steps, a `then` that checks nothing); it cannot check semantic truth — that is the author's job.
48
+ - If the project enables the narration lint, run it over the whole suite: `pytest --given-lint` (a selection fails on a `given_lint_ignore` entry for a test it skips). It catches structural lies (empty steps, a `then` that checks nothing); it cannot check semantic truth — that is the author's job.
49
49
 
50
50
  Auditing someone else's narration rather than writing your own? That is the `pytest-given-reviewing` skill, which restates these rules as a review rubric.
51
51
 
@@ -16,8 +16,8 @@ This is the authoring-relevant surface, version-matched to the installed package
16
16
  - **`@scenario(name, tags=None, *, stories=None, pins=None, group_parametrized=True)`** — marks a test for the report; required for it to appear. `name` is a plain string, a `Template` (for parametrized names), or a t-string whose interpolations are all glossary handles (they render as term refs in the title). `stories=` takes a story or several to narration-match against; `pins=` takes sentence handles (`the_story['name']`, `the_story[3]`) and covers those sentences plus its steps' pins, with no narration matching; `pins=[]` keeps only the steps' pins. Bare numbers and names raise. The decorated function is returned unwrapped, so it keeps its own signature.
17
17
  - **`given(text, *, pins=None)` / `when(text, *, pins=None)` / `then(text, *, pins=None)`** — dual-purpose; `pins=` binds the step to story sentences instead of narration matching, and `pins=[]` opts it out of matching ([stories.md](stories.md)):
18
18
  - **Context manager** in a test body: `with when('…'): result = sut(x)`. Steps nest within a phase (a `when` inside a `when`); crossing phases raises `PytestGivenError` — including a decorated helper of another phase called inside an open step.
19
- - **Fixture decorator** — `@given` only, with `@pytest.fixture` **outermost** (`@pytest.fixture` above `@given('…')`); the other order is rejected at decoration time, `@when`/`@then` on a fixture is rejected at runtime, and the label must be a plain string. Generator fixtures work, at any scope; recording steps after `yield` is not allowed.
20
- - **Helper-function decorator** (any phase): the helper records its own step per call; use `Template` to reference the helper's parameters (`@when(Template('I insert {amount}'))`). Placeholders must name one of the helper's named parameters; `*args`/`**kwargs` placeholders raise at decoration time. An `async def` helper works too: the step stays open across the awaited body.
19
+ - **Fixture decorator** — `@given` only, with `@pytest.fixture` **outermost** (`@pytest.fixture` above `@given('…')`) and a plain-string label. Generator fixtures work, at any scope; a step or attachment after `yield` raises.
20
+ - **Helper-function decorator** (any phase): the helper records its own step per call; use `Template` to reference the helper's parameters (`@when(Template('I insert {amount}'))`). Placeholders must name one of the helper's named parameters. An `async def` helper works too: the step stays open across the awaited body.
21
21
  - **Call-site label** via `Annotated` on a test parameter — `given` only: `def test(text: Annotated[str, given(Template('the name {text}'))])` surfaces a fixture or parametrize value as a `given` step. Plain string or `Template` only; a t-string is rejected here. A label's `pins=` replaces the fixture label's pins; `pins=[]` clears them.
22
22
  - **`when_then(when_text, then_text)`** — one `with` emitting a `when` (wrapping the body) and a sibling `then` (emitted once the body exits cleanly). Pair with a nested `pytest.raises(...)` for expected-raise scenarios. If the body raises uncaught, the `then` is skipped.
23
23
  - **`attach(label, content)`** — attach data to the current step. Must be called with a step open (a call from the test body, outside every `given`/`when`/`then`, raises). `label` is a plain `str` (a t-string or `Template` raises); strings stored verbatim, other types JSON-serialized.
@@ -26,9 +26,9 @@ This is the authoring-relevant surface, version-matched to the installed package
26
26
 
27
27
  | Form | Where | Behavior |
28
28
  |---|---|---|
29
- | Plain string / f-string | anywhere | Rendered verbatim; f-string values are not highlighted. |
29
+ | Plain string / f-string | anywhere | Rendered verbatim; f-string values are not highlighted. In a parametrized scenario, text that varies across cases fails the run — use a t-string. |
30
30
  | T-string `t'a {cup_size} cup'` | test-body steps only | Interpolated at runtime; values color-coded when the expression matches a parametrize column. Full expression syntax allowed. |
31
- | T-string `t'a {guest} checks in'` | `@scenario(...)` name — glossary handles only | Evaluated eagerly at import; each handle renders as a term ref in the title. A value/expression interpolation is rejected (values aren't in scope at import). |
31
+ | T-string `t'a {guest.low} checks in'` | `@scenario(...)` name — glossary handles only | Evaluated eagerly at import; each handle renders as a term ref in the title. A value/expression interpolation is rejected (values aren't in scope at import). |
32
32
  | `Template('… {col} …')` | `@scenario(...)`, helper decorators, `Annotated[..., given(...)]` | Deferred substitution — against parametrize columns (`@scenario`, `Annotated`) or the helper's bound arguments (decorators). |
33
33
 
34
34
  Hard rules (each raises `PytestGivenError`):
@@ -39,7 +39,7 @@ Hard rules (each raises `PytestGivenError`):
39
39
  ## Parametrized tests
40
40
 
41
41
  - All cases group into **one scenario with a parameter table**. T-string interpolations naming a parametrize column render as colored values per row.
42
- - **The baseline case's steps are the template for every row** (the first case that passed) — but only their *structure*: a narrated value or an attachment payload that varies across cases becomes its own parameter-table column. Authoring forms that cannot be rendered honestly against that template raise `PytestGivenError` instead of shipping a wrong report; each message names its fix. The habits that avoid them are in [scenarios.md](scenarios.md) under "Phase structure".
42
+ - **The baseline case's steps are the template for every row** (the first case that passed) — but only their *structure*: a narrated value or an attachment payload that varies across cases becomes its own parameter-table column. Authoring forms that cannot be rendered honestly against that template raise `PytestGivenError` instead of shipping a wrong report; each message names its fix. The habits that avoid them are in [scenarios.md](scenarios.md) under "Parameter tables".
43
43
  - **Narration that genuinely branches per case**: `@scenario(..., group_parametrized=False)` declines the merge and emits one scenario per case, each titled `<name> [<parametrize id>]` with any `Template` placeholders substituted per case first. No parameter table. On an unparametrized test it raises at collection.
44
44
  - Parametrized **scenario name**: `@scenario(Template('Brew {cup_size} ml'))`.
45
45
  - Surface a parametrize value as a `given`: `Annotated[int, given(Template('a {cup_size} ml cup'))]` on the parameter.
@@ -47,11 +47,9 @@ Hard rules (each raises `PytestGivenError`):
47
47
 
48
48
  ## Glossary
49
49
 
50
- - **Code-defined**: `g = Glossary()`, then `guest = g.actor('Guest', definition='…')`, `g.work_object('Room', …)`, `g.activity('book', …)`; `g('foo')` declares a kindless term, `g['Guest']` looks one up (case-insensitive, raises if unknown).
51
- - **File-backed**: `g = FileGlossary(path, *, term_column=0, description_column=1, kind_column=None)` over the file's GFM pipe tables, each column a 0-based index or a case-insensitive header name. The vocabulary is closed: `g('foo')` and `g['foo']` both only look up.
52
- - **Handles** interpolate into t-string steps, `@scenario` titles and sentences in three forms: bare `g['Room']`, `.low` `g['Room'].low`, and called `g['borrow']('borrows')`.
50
+ - `Glossary()` with `g.actor(name, definition=…)`, `g.work_object(…)`, `g.activity(…)` and `g('foo')` (kindless); `FileGlossary(path, *, term_column=0, description_column=1, kind_column=None)`. `g['Guest']` looks a term up, case-insensitive.
53
51
 
54
- Discovery, kinds, and the one-glossary-per-suite rule: [glossaries.md](glossaries.md).
52
+ Handle forms, discovery, kinds, and the one-glossary-per-suite rule: [glossaries.md](glossaries.md).
55
53
 
56
54
  ## Stories
57
55
 
@@ -33,5 +33,3 @@ The step past the method: binding scenarios (`@scenario(..., stories=...)`) turn
33
33
  1. Run Domain Storytelling sessions with stakeholders (a whiteboard or Egon.io is fine — the method works on paper).
34
34
  2. Transfer the agreed stories into `story(...)` code; the glossary emerges from the clause slots (kinds inferred from positions).
35
35
  3. Write scenarios against the stories as behavior gets implemented; uncovered sentences are your living backlog.
36
-
37
- This is why a greenfield project may want stories *before* any scenarios: the domain understanding and vocabulary are established up front, and every later scenario has a place to link into.
@@ -43,14 +43,14 @@ from tests.ubiquitous_language import g # noqa: F401 — plugin discovery
43
43
 
44
44
  `import tests.ubiquitous_language` binds a module, not a glossary: the scan finds nothing and the Glossary tab renders empty. Binding it anyway is the safe habit — it costs one line and survives a later refactor that drops the last story.
45
45
 
46
- **One glossary per suite.** Two distinct `Glossary` instances reaching the report — via stories or via conftests — raise `PytestGivenError`. Splitting vocabulary across bounded contexts means splitting the test suite too.
46
+ **One glossary per suite.** Stories reaching two distinct `Glossary` instances raise `PytestGivenError`, and so do two in conftests; but once the stories reach one, a different conftest glossary is silently ignored — keep one instance and import it everywhere.
47
47
 
48
48
  ## Using terms in narration
49
49
 
50
50
  Look up handles by name — `g['Room']` (case-insensitive) — or use the captured variables from a code-defined glossary. Both work in t-string steps, `@scenario(...)` titles, and story sentences:
51
51
 
52
52
  ```python
53
- with when(t'a {g["Guest"]} {g["book"]("books")} a {g["Room"]}'):
53
+ with when(t'a {g["Guest"].low} {g["book"]("books")} a {g["Room"].low}'):
54
54
  ...
55
55
  ```
56
56
 
@@ -70,13 +70,13 @@ Pick the lightest surface form for the word you need — the same three forms on
70
70
  Terms are actors, activities, or work objects. Three ways a term gets its kind:
71
71
 
72
72
  1. **Explicit** — `g.actor(...)` / `g.activity(...)` / `g.work_object(...)`, or a `kind_column` in the glossary file.
73
- 2. **Inferred from stories** — a term with no declared kind takes one from its clause slot positions: position 0 → actor, odd positions (the verb slots) → activity, even positions ≥ 2 → work object. A term seen in both actor and noun slots resolves to actor (an actor can be the target of a hand-off); a term seen in a verb slot *and* any other slot raises — add a kind column to disambiguate. A term used only in steps stays kindless (neutral wash).
74
- **A declared kind is checked against those same positions, never overridden** — whether it came from a typed handle (`g.work_object(...)`) or a `kind_column` row, putting the term in a slot its kind forbids raises `PytestGivenError` at `sentence(...)` construction, naming the term and its declared kind. Inference then handles only the undeclared terms. So an explicit kind settles ambiguity but does not license a mismatch. Note `kind_column` is opt-in: a column headed "Kind" is ignored unless you pass `kind_column=`, and those terms fall back to slot inference.
73
+ 2. **Inferred from stories** — a term with no declared kind takes one from its clause slot positions: position 0 → actor, odd positions (the verb slots) → activity, even positions ≥ 2 → work object. A term seen in both actor and noun slots resolves to actor (an actor can be the target of a hand-off); a term seen in a verb slot *and* any other slot raises — add a kind column to disambiguate. A term used only in steps stays kindless (grey).
74
+ **A declared kind is never overridden:** putting the term in a slot its kind forbids raises at `sentence(...)`. `kind_column` is opt-in: a column headed "Kind" is ignored unless you pass `kind_column=`.
75
75
  3. **Deliberately deferred** — `g('foo')` declares a term the team hasn't classified yet: it lands in the *Uncategorized* bucket and shows an *Undefined* badge until a definition arrives. Use it as a triage bucket, not a resting place. Code-defined glossaries only: a `FileGlossary` is a **closed vocabulary** — `g('foo')` and `g['foo']` both merely look up and raise on unknown names; new vocabulary is added as a row in the file.
76
76
 
77
77
  ## Keeping the glossary honest
78
78
 
79
- - **Don't dilute the glossary — keep it sharp.** A term earns its row by being vocabulary the team actually speaks: something someone would look up, with a meaning specific to the domain. Never add terms to make sentences render more term refs or to improve lint metrics; a generic word in a sentence is better left a bare string. When a row genuinely doesn't earn its place, deleting it beats manufacturing a reference to it — but that is a judgment about the term, not about whether anything happens to reference it.
80
- - **Watch the size — a glossary is read whole, never sampled.** Authors and reviewers absorb every term in one pass; that is how a near-duplicate term gets caught before it is coined. A glossary that outgrows one comfortable reading is speaking for more than one bounded context: alert the user rather than start reading it piecemeal. The structural fix is one glossary per context — but a suite supports only one glossary, so that means splitting the suite as well; raise it as a design question, not a mechanical edit.
81
- - **Tags never duplicate terms** — filter a feature area via its term, and keep tags for what the glossary can't carry (behavior, mechanism). The `tag-shadows-term` lint rule enforces this.
82
- - **A term nothing references is normal, and usually needs no action.** A glossary documents the domain, not the test suite's coverage — real vocabulary can sit in the file with no scenario narrating it yet, and every file term appears in the report either way. That is why `dead-term` defaults to `off`. Opt in only where the glossary is meant to be fully exercised, and read what it reports as a prompt to look at the term, not a defect to clear — if an undecorated test already demonstrates the term's behavior, decorating that test is the fix.
79
+ - **Don't dilute the glossary.** A term earns its row by being vocabulary the team speaks: something someone would look up, with a meaning specific to the domain. Never add terms to render more term refs or to satisfy the lint; a generic word stays a bare string. When a row doesn't earn its place, delete it rather than manufacture a reference to it.
80
+ - **Watch the size — a glossary is read whole, never sampled.** Reading every term in one pass is how a near-duplicate gets caught before it is coined. A glossary that outgrows one comfortable reading speaks for more than one bounded context: raise it with the user as a design question, since one glossary per context means splitting the suite too.
81
+ - **A definition that asserts behavior is a spec sentence.** A row saying "must be unique" or "produces no step" makes a claim: back it with a scenario, and update the row when the behavior changes.
82
+ - **A term nothing references is normal.** A glossary documents the domain, not the suite's coverage, which is why `dead-term` is off by default. Where you opt in, read a finding as a prompt to look at the term: if an undecorated test already demonstrates its behavior, decorate that test; if a step says the term as plain text, add the term ref.
@@ -0,0 +1,105 @@
1
+ # Writing scenarios
2
+
3
+ When decorating a test with `@scenario`, the goal is a report that reads as a truthful behavioral spec.
4
+
5
+ ## What to decorate
6
+
7
+ - **Convert behavior, not plumbing.** Decorate tests that assert a rule (a calculation, a validation, a dispatch decision). Leave trivial getters, constructors, and dataclass round-trips as plain tests — they add report noise, not behavior.
8
+ - **One scenario per rule, not per branch.** Decorate the test that best states the rule — the one whose body shows what its name claims, so the end-to-end test when a unit test covers the same rule — and leave its edge cases plain. A user-visible rule you add or change needs its scenario, or the report never shows it.
9
+ - **Sibling scenarios of one rule are one table.** Merge scenarios that share a step skeleton and vary one arrangement, or whose names negate each other ("the reviewer confirms, so it is rejected" / "the reviewer declines, so it is accepted"), into one parametrized scenario (see [Parameter tables](#parameter-tables)); plain tests of the rule that fit its steps become rows. Before you merge or delete a scenario, account for everything only it had:
10
+ - an assertion: keep it, or name the scenario that still makes it;
11
+ - an input the table holds fixed: a booking in another building is not just another "no clash" row;
12
+ - a step that covers a story sentence or holds a term's last term ref: re-run the coverage check from [stories.md](stories.md).
13
+
14
+ ## Phase structure
15
+
16
+ - **Every step maps to load-bearing code.** `given` arranges, `when` performs the one call under test, `then` asserts its result. Never write a placeholder step like `with given(...): pass` — a step with no code is a lie in the report. Delete it.
17
+ - **Put the system-under-test call in `when`, not folded into the `then` assertion.** Prefer `with when(...): result = sut(x)` then `with then(...): assert result == …` over `with then(...): assert sut(x) == …`.
18
+ - **One outcome per `then`.** When an action has two independent outcomes (a return value *and* a state change), give each its own sibling `then` step. A `then` narration that needs an "and" is two steps written as one.
19
+ - **When the construction *is* the action under test, `when` builds it and `given` shows only the input.** If a scenario asserts a property of a freshly loaded, parsed, or built object (`Order(lines)` computes its total, `parse(doc)` yields all rows), `given` holds and, when useful, `attach`es the raw input; `when` runs the constructor; `then` asserts on the result. A `given` that both arranges the input *and* constructs the object under test is the most common missed `when`.
20
+ - **Two phases is fine when honest — but rarer than it looks.** If the assertion inspects a *static property of the arranged state* (not a return value of an action), `given` + `then` is truthful and you shouldn't invent a `when`. Likewise a pure "constructing X raises" check needs no `given`.
21
+
22
+ ## Parameter tables
23
+
24
+ A parametrized scenario renders as one narrated tree over a parameter table, and a reader uses that table as a decision table: one row per input combination, with its outcomes.
25
+
26
+ - **Step structure must not depend on parameter values.** Every row renders against the baseline case's step structure, so a conditional `with given/when/then(...)` on a parametrize value fails the run. When the steps genuinely differ per case, decline the merge with `@scenario(..., group_parametrized=False)`; when the cases are different behaviors rather than one behavior narrated two ways, split them into separate scenarios.
27
+ - **Narrate what varies with a t-string interpolating a bare name.** `when(t'the drink costs {price} euros')` leaves a `{price}` token in the step and a column holding each case's value. Keep everything else identical across cases — the sentence, the `attach` labels, the term refs — and bind a value you derive from a column to its own local before narrating it (`price = cup_size * 0.01`, then `{price}`).
28
+ - **A column must read on its own.** When the outcome depends on how a varied input relates to a fixed one, the column holds the relation (`hours_later`, `same_room`), not the varied input's own value (`existing_start=50` or `existing_room='B'` beside a new booking fixed in room A at hours 10 to 20 in another step). This applies to a merged table too: parametrizing the value an old step text happened to name is how such columns arise. Arrange the fixed side first, so the relation has something to refer to. Whole objects as columns fail the same way: a reader can't compare two `Booking(...)` reprs at a glance.
29
+ - **Never turn a label back into data.** The body may unpack a relation column the way its narration says (`'A' if same_room else 'B'` under "same room: {same_room}"). It never maps a descriptive label to the input (`{'40 hours later': 40}[shift]`), and no label column sits beside the data it describes. Either way, the report shows a claim that nothing checks against the input. A document is the exception, since it can't read as a column: parametrize a label, look the document up, and `attach` it, so the row shows both.
30
+ - **Contrast rows show what decides.** For each input that decides the outcome, include a row that changes only that input and flips the outcome, at the boundary where there is one (moved 10 hours later still shares an hour, 11 hours later does not). A table whose rows all share one outcome can't show which column matters, even when another scenario holds the flipped case. Only a column listing forms the rule treats alike needs no flip, and a flip that needs other steps stays in its own scenario.
31
+ - **A refusal is a row too.** `when_then` can't narrate a raise per row, so when accepted and refused inputs share the steps, catch the refusal in the `when` and assert it like any outcome:
32
+ ```python
33
+ with when(t'the {g["Guest"].low} books {nights} nights'):
34
+ try:
35
+ booking, refusal = book(room, nights), ''
36
+ except BookingRefused as error:
37
+ booking, refusal = None, str(error)
38
+ with then(t'the booking is accepted: {accepted}'):
39
+ assert (booking is not None) == accepted
40
+ with then('a refusal names the minimum stay'):
41
+ assert ('minimum stay' in refusal) == (not accepted)
42
+ ```
43
+ Keep `when_then` for a scenario whose every row refuses.
44
+ - **Outcomes are columns, asserted against.** An outcome that varies gets a column, named in a `then` that holds for every row (`assert (0 in result.clashes) == clashes` under "it clashes: {clashes}"), not an `if` on the column inside the step. The headline outcome gets its own column even when another column implies it (`settled_by=None` implies "no clash", but a reader looks for the clash). Columns follow the order the narration first shows them, so an input narrated in a `given` precedes an outcome in a `then`.
45
+
46
+ ## Arrangement
47
+
48
+ - **Surface the arrangement as a `given`, don't hide it in the `when` or the assertion.** A module constant or a value built on the fly (a document string, a path, a list of rows) that the action consumes is bound in a `with given(...)` block, not passed as a literal into the `when`/`then`/`pytest.raises` call.
49
+ - State-mutating setup calls (inserting credit, seeding a database) are arrangement too: a scenario with two `when` steps usually hides an arrangement in the first one.
50
+ - **Decide a fixture's `given` by what it holds.** Would you write a `given` for this value if you constructed it inline? A domain value the scenario is about (an actor, a document, an entity it acts on) is arrangement: decorate the fixture `@given('…')` under `@pytest.fixture`. Infrastructure (a `Glossary()`, `tmp_path`, a connection) stays a bare `@pytest.fixture`. Either way, an arrangement reads the same whether a scenario builds it inline or pulls a fixture.
51
+ - **A parametrized value can be a `given` too.** Its column already shows in the parameter table. When it reads as an arrangement the reader should see named up front, surface it via `Annotated[..., given(Template('… {col} …'))]` on that parameter; for a one-arg pure function either reading is honest.
52
+ - **Attach the concrete artifact a step can only describe.** When a step handles a multi-line value its text can only abstract — a source document, a config snippet, a rendered result — `attach('label', text)` it onto that step.
53
+ - **An input document goes on the arranging `given`; an output document goes on the `then` that checks it.** Output is the case most often missed: the assertions only sample the result, so the attachment is the only place a reader sees all of it.
54
+ - **Attach documents, not object dumps**, and skip it when the step text or the parameter table already carries the value, or the artifact is too big to read inline (a full HTML page). The exception is a genuine before/after: where a step digests a rich input to a small output, the input makes "only the first line survives" checkable.
55
+ - Across parametrize cases, an attachment whose content varies becomes a column headed by its label.
56
+
57
+ ## Expected raises
58
+
59
+ - **Keep pytest-given narration and real assertions separate.** Nest the vanilla `pytest.raises` inside the narration, never inside a narration-named helper, and exempt tests from lint rules that forbid nested `with` statements (ruff `SIM117`).
60
+ - **Narrate an expected raise as `when_then`.** `with when_then('the action', 'an `InsufficientCredit` error is raised'), pytest.raises(Exc, match=…): sut(x)` emits a `when` wrapping the call and a sibling `then`, recorded once `pytest.raises` swallows the error.
61
+ - **Write the `then` as a real outcome.** Name the exception type and, when `match=` pins a specific message, say what it reports in domain terms (`'the shortfall amount is reported'`) — never a bare `'it raises'`.
62
+ - **Check a message with several details in its own `then`**: `pytest.raises(E) as excinfo` under the `when_then`, then `with then('the error names …'): assert '…' in str(excinfo.value)`.
63
+ - **The pin must be as specific as the `then`.** A `then` promising message details (the offender, a suggestion, a file:line) needs a `match=` that only a message with that detail passes; `match='column'` under "names the missing column" lets the detail regress unnoticed. An alternation (`match=r'odd|dangling|ends'`) is only as strong as its weakest branch: pin the detail that tells sibling refusals apart.
64
+
65
+ ## Vocabulary and tags
66
+
67
+ - **Narrate in glossary vocabulary.** Reference terms through handles in t-string step text — `t'a {g["Room"].low} is booked'` ([glossaries.md](glossaries.md)) — so they render as kind-colored words and feed the Glossary tab's per-term filter. Use `.low` mid-sentence and the bare handle to start a sentence. Pick a term for its meaning, not its word: a ref whose definition isn't what the sentence means links the reader to the wrong row. (Skip this rule if the project has no glossary yet.)
68
+ - **The code speaks the language too.** Each referenced term should be reflected in the naming within the step: the body and the SUT names it directly calls. `File glossary` over a `FileGlossary` call matches by design (term names are natural language), but `{g["Reservation"]}` over code that only knows `Booking` is language drift — rename one side.
69
+ - **Tag orthogonally to the glossary.** A tag that restates a term is redundant — filter by the term instead. `tag-shadows-term` catches literal collisions, not a tag naming a feature area the glossary covers under a different word (`markdown` over scenarios referencing `File glossary`).
70
+ - **Expect tagging to be sparse.** Tags carry only what the glossary can't: behavior (`validation`) and mechanism (`parametrization`). A tag must cut across modules (one confined to a test file repeats the module grouping) and stay a minority of the suite — `happy-path` on most scenarios filters nothing. Once used, a tag goes on every scenario it describes, or filtering by it under-reports.
71
+ - **A `/` nests a tag.** The Tags sidebar files `ticket/ABC-123` under a `ticket` heading that filters like a package, so a family of tags costs one row. Ticket tags are the family worth having: they open exactly one ticket's work in the report.
72
+ - **Keep step text short.** A t-string with two or three term refs reaches your line-length limit fast; move detail into the node structure rather than one long sentence.
73
+
74
+ ## Keeping it truthful
75
+
76
+ Step text may abstract; it must never overstate.
77
+
78
+ - **A value in the text must match the body.** A quantity, date, or amount you narrate is a claim about the data the step actually holds: `'three copies'` over `catalog={'Dune': 1}` is a lie even when every assertion passes.
79
+ - **Everything a `then` claims must be asserted in it.** A `then` reading "…and recorded in the ledger" with no such assertion is fabricated behavior — assert it, or drop the clause.
80
+ - **What the `when` names must be what the body calls.** Narrate the action the step performs, not the one the scenario is loosely about. The frequent slip: a `when` reading as more arrangement over a body that arranges *and* makes the call the `then` reports, so the action reaches the report nowhere. Move the setup into a `given`.
81
+ - **A universal quantifier is a claim about every item.** A `then` or scenario name saying "each", "every", "both", or "all" must assert every item it ranges over — "each slot becomes a term ref" backed by assertions on two of three slots overstates. Assert them all, or narrow the text to what is checked.
82
+ - **Mark a scenario written ahead of its implementation `xfail(strict=True)`** (or set `xfail_strict`). It reports as an expected failure, with its steps and error, and fails the run once it passes, so the mark comes off. A non-strict mark stays on working behavior and turns a later regression into an expected failure. An `xfail` on one `pytest.param` row is fine too.
83
+ - **Stale narration is a stale assertion.** When a change touches a step *body*, re-read that step's *text* (and the scenario name) in the same edit and update it if the behavior shifted.
84
+ - **When narration and body disagree, the narration is not automatically the mistake.** A name stating the intended behavior over a body that checks something weaker, or passes for a different reason, has found a broken test: fix the body, keep the words. Editing the narration down launders a bad test into a truthful-looking spec. Ask whether the body would fail if the name's promise were broken.
85
+
86
+ ## Mechanical counterparts
87
+
88
+ The narration lint (`pytest --given-lint`, or `given_lint = true` in `[tool.pytest]`; `--no-given-lint` turns an ini-enabled lint off for one run) enforces the structural subset of these rules. A *warn* finding prints in the terminal summary; only an *error* finding fails the run.
89
+
90
+ | Rule | Default | Catches |
91
+ |---|---|---|
92
+ | `empty-step` | error | A step whose body does nothing — only constants/`pass`, or, for `when`/`then`, only an `attach(...)` call. |
93
+ | `then-without-check` | error | A `then` with no `assert` statement and no checking call — a call whose name starts with `assert` (`assert_totals(order)`, `result.assert_outcomes(...)`), or `pytest.raises` / `pytest.warns` / `pytest.fail`. Nothing else counts, `pytest.approx` included. |
94
+ | `missing-phase` | warn | A passed scenario that doesn't cover all three phases. Fixture `@given`s and `Annotated[..., given(...)]` parameters count. |
95
+ | `check-outside-then` | warn | An `assert` inside a `given` or `when` (the `when` half of a `when_then` pair is exempt). |
96
+ | `action-in-then` | warn | No `when` performs an action and a `then` folds the action into its assertion. |
97
+ | `unused-interpolation` | warn | A `with`-anchored step whose narration interpolates `{name}` that the step body never uses — a t-string interpolation, or a parametrize placeholder in a grouped scenario. |
98
+ | `tag-shadows-term` | warn | A scenario tag whose slug duplicates a glossary term. |
99
+ | `dead-term` | off | A glossary term referenced by no scenario narration, no step narration and no story sentence. Off by default because an unreferenced term is normal; opt in where the glossary is meant to be fully exercised. |
100
+
101
+ Severities are overridable per rule via the `given_lint_rules` ini; that and the rest of the setup are in the project documentation at <https://nwilbert.github.io/pytest-given/latest/configuration/narration-lint/>.
102
+
103
+ **A `then-without-check` finding on a step that does check is usually a naming problem.** The rule knows an assertion helper only by its `assert` prefix, so rename `check_totals` to `assert_totals` before reaching for the ignore list. A third-party matcher under another name (`result.stdout.fnmatch_lines(...)`) needs a real `assert` beside it, or an ignore entry.
104
+
105
+ **Treat a `missing-phase` warning as a prompt to restructure, not to suppress.** It nearly always marks a hidden `when`: apply the constructor rule above, or surface a parametrized input as `Annotated[..., given(Template('…'))]` so the inline block can narrate the action. A `given_lint_ignore` entry (`missing-phase: <node-id glob>`) is a last resort, and costs more than it looks: an entry that suppresses nothing is a `stale-ignore` error that can't be downgraded or ignored, so `pytest tests/one_file.py --given-lint` fails whenever the selection skips the suppressed test.
@@ -15,7 +15,7 @@ book_a_group_trip = story('Book a Group Trip', [
15
15
  ```
16
16
 
17
17
  - A sentence reads left-to-right: **actor → activity → work object**, with optional connective words (`'for'`, `'to'`) between parts. Structurally each clause is a strict node/edge alternation of odd length ≥ 3: even positions are entity nodes (position 0 is the acting actor), odd positions are edges (an activity or a connective) — the verb slots.
18
- - **A bare word consumes a position.** Write a connective as one string in an edge slot (`'to the'`, `'with a'`); never insert a standalone article before a noun — it shifts the noun into a verb slot and construction fails.
18
+ - **A bare word consumes a position.** Write a connective as one string in an edge slot (`'to the'`, `'with a'`); never insert a standalone article before a noun — it shifts the noun into a verb slot.
19
19
  - Handles come from the glossary; calling one supplies an instance or inflection — `organizer('Carol')`, `confirm('confirms')`.
20
20
  - Any part may be a **bare string** instead of a glossary handle — the right place for a verb that is just sentence prose (*searches for*, *submits*; see [Authoring workflow](#authoring-workflow)). But a sentence needs at least two distinct glossary terms to be matched by narration; under-anchored sentences render as "not coverage-tracked" unless a step pins them (below).
21
21
  - A sentence with several arrow chains under one number — an actor handing a work object to two recipients, two actors working side by side — takes one `clause(...)` per chain:
@@ -40,7 +40,7 @@ def test_select_suite(carol):
40
40
  ...
41
41
  ```
42
42
 
43
- Coverage is matched **per step**: a sentence is covered when a *single step's* term refs include all of the sentence's terms — refs spread across several steps don't add up. The Stories tab shows a coverage chip per sentence with the scenarios that touch it; the JSON report carries the same result under `coverage[]` (below).
43
+ Name a story in `stories=` only when a step can match or pin one of its sentences: a scenario that covers none of it still lists under the story. Coverage is matched **per step**: a sentence is covered when a *single step's* term refs include all of the sentence's terms — refs spread across several steps don't add up. The Stories tab shows a coverage chip per sentence with the scenarios that touch it; the JSON report carries the same result under `coverage[]` (below).
44
44
 
45
45
  What the rule means when you write:
46
46
 
@@ -60,7 +60,7 @@ jq -r '.coverage[] | select(.tracked and .scenario_ids == [])
60
60
 
61
61
  `coverage[]` holds `{story_id, sentence_id, tracked, scenario_ids}` per sentence, computed by the same code as the Stories tab; `tracked: false` is the "not coverage-tracked" chip. The full shape is in the navigating skill's `references/report-json.md`.
62
62
 
63
- A step can also **pin** a sentence explicitly — `given(text, pins=book_a_group_trip['confirm'])`, taking a sentence handle (or a list of them). A pin *replaces* narration matching for that step rather than adding to it: the step covers exactly the sentences it names, in any story, however well its text fits others; `pins=[]` opts a step out of matching without pinning anything. A pin is also the only thing that reaches an under-anchored sentence: the two-term rule gates narration matching, not pins. Use a pin when the sentence is phrased above the vocabulary the step narrates (e.g. a process-level sentence implemented by a technical test), and keep it on the one step that genuinely demonstrates the sentence. `@scenario(pins=...)` pins the whole scenario: it covers those sentences plus its steps' pins, and none of its steps is narration-matched; `@scenario(pins=[])` keeps only the steps' pins.
63
+ A step can also **pin** sentences — `given(text, pins=book_a_group_trip['confirm'])`, a sentence handle or a list. A pin *replaces* narration matching for that step: it covers exactly the sentences it names, in any story; `pins=[]` opts a step out of matching. Only a pin reaches an under-anchored sentence. Pin when the sentence is phrased above the vocabulary the step narrates (a process-level sentence implemented by a technical test), on the one step that demonstrates it. `@scenario(pins=...)` pins the whole scenario: none of its steps is narration-matched, and their own pins add up; `@scenario(pins=[])` keeps only the steps' pins.
64
64
 
65
65
  **Pin by name, not by number.** Sentence numbers are positions, so inserting a row renumbers every row after it, and `the_story[5]` silently lands on a different sentence. Name a sentence you pin (`sentence(..., name='confirm')`) and pin `the_story['confirm']`.
66
66
 
@@ -74,4 +74,3 @@ Write a story for flows with distinguishable actors and hand-offs — a user and
74
74
 
75
75
  - **Keep sentences at domain granularity** — what the actor does ("submits payment for the booking"), never what the code does ("calls `submit_payment()`"). If a sentence only makes sense to someone reading the implementation, it's too fine.
76
76
  - **Grow the glossary from the sentences — but only with real vocabulary.** A slot gets a term when the word is domain language someone would look up; a word that is just sentence prose (generic verbs like *tells*, *reviews*) stays a bare string. Don't mint glossary rows to satisfy the grammar. With a file glossary and no kind column, term kinds are inferred from slot positions for free (see [glossaries.md](glossaries.md)); unclassified vocabulary can enter as `g('loyalty points')` and be triaged later.
77
- - **Derive stories from Domain Storytelling sessions** where you can: transfer the sentences recorded with stakeholders into `sentence(...)` rows, then write scenarios against them — see [domain-storytelling.md](domain-storytelling.md).
@@ -18,8 +18,9 @@ coverage[] one entry per story sentence — which scenarios cover it
18
18
  | `narration.text` | The scenario name (grouped template for parametrized scenarios) — for a *step* in a grouped parametrized scenario this is the template too (`the drink costs {price} euros`), not the first case's rendering |
19
19
  | `module` | Python module the test lives in |
20
20
  | `tags[]` | `tags=` from `@scenario` — report metadata, **not** pytest marks |
21
- | `status` | `passed` / `failed` / `skipped` |
21
+ | `status` | `passed` / `failed` / `skipped` / `xfailed` (failed as expected: an `xfail` mark or `pytest.xfail()`) |
22
22
  | `skip_reason` | `null`, or the reason a skipped scenario carries instead of a traceback |
23
+ | `xfail_reason` | `null`, or the reason an xfailed scenario was expected to fail; it keeps its steps and error as well |
23
24
  | `duration_ms` | Wall-clock time for the test |
24
25
  | `steps[]` | Recursive step tree (see below) |
25
26
  | `parameters` | `null`, or `{columns: [{id, name, kind}], cases: [{values, status, error}]}` for parametrized scenarios. `kind` is `param` / `derived` / `attachment`; a case's `values` is positionally aligned with `columns`, and an `attachment` cell is an `{label, content, content_type}` object (or `null` for a case with no value). A scenario opted out of grouping with `group_parametrized=False` has `parameters: null` like any unparametrized one |
@@ -76,6 +77,10 @@ jq -r '.scenarios[] | select(.status == "failed")
76
77
  | "[" + (.values | map(tostring) | join(", ")) + "] " + .error.message]
77
78
  | join("; ")))' report.json
78
79
 
80
+ # Expected failures with their reasons — planned behavior not yet working
81
+ jq -r '.scenarios[] | select(.status == "xfailed")
82
+ | .narration.text + (if .xfail_reason then " — " + .xfail_reason else "" end)' report.json
83
+
79
84
  # Scenarios whose narration references a term (any step depth: use recursion for nested steps)
80
85
  jq -r '.scenarios[] | select([.. | .term_id? // empty] | index("waitlist"))
81
86
  | .narration.text' report.json