design-playbook 0.24.4 → 0.25.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/README.md +20 -6
  2. package/codex/AGENTS.md +1 -1
  3. package/codex/install_skills.py +9 -0
  4. package/commands/component-distill.md +8 -0
  5. package/commands/run-status.md +84 -0
  6. package/lib/index.js +4 -2
  7. package/mcp/evidence/README.md +14 -4
  8. package/mcp/evidence/capture_runtime.py +120 -10
  9. package/mcp/evidence/containment.py +91 -0
  10. package/mcp/evidence/disclosure.py +2 -3
  11. package/mcp/evidence/evidence_preflight.py +22 -2
  12. package/mcp/evidence/path_syntax.py +26 -0
  13. package/mcp/evidence/server.py +26 -2
  14. package/mcp/preview/control.css +178 -145
  15. package/mcp/preview/control.html +80 -86
  16. package/mcp/preview/control.js +286 -63
  17. package/mcp/preview/control.py +10 -4
  18. package/mcp/preview/control.review.js +64 -77
  19. package/mcp/preview/i18n.py +49 -43
  20. package/mcp/preview/integrity.py +2 -0
  21. package/mcp/preview/ledger.py +65 -0
  22. package/mcp/preview/pin_bridge.py +11 -1
  23. package/mcp/preview/review_session.py +29 -3
  24. package/mcp/preview/server.py +34 -3
  25. package/mcp/preview/transaction.py +50 -1
  26. package/mcp/run_console/actions.py +133 -7
  27. package/mcp/run_console/app.css +87 -0
  28. package/mcp/run_console/app.js +275 -15
  29. package/mcp/run_console/diagnostic_export.py +403 -0
  30. package/mcp/run_console/export_transaction.py +208 -0
  31. package/mcp/run_console/http_server.py +106 -6
  32. package/mcp/run_console/request_security.py +19 -2
  33. package/mcp/run_console/session.py +22 -2
  34. package/mcp/run_console/snapshot_builder.py +2 -1
  35. package/mcp/run_console/source_registry.py +23 -1
  36. package/package.json +2 -2
  37. package/scripts/adapter_matrix.py +2 -2
  38. package/scripts/audit_preferences.py +9 -0
  39. package/scripts/component_candidates.py +420 -0
  40. package/scripts/context_budget.py +9 -0
  41. package/scripts/contract_v1.py +15 -0
  42. package/scripts/doctor.py +10 -1
  43. package/scripts/eval_decisions.py +312 -0
  44. package/scripts/evidence_manifest.py +280 -0
  45. package/scripts/frontend_review.py +495 -0
  46. package/scripts/g10_design_decisions.py +36 -13
  47. package/scripts/g11_coverage.py +23 -0
  48. package/scripts/g1_spec.py +131 -2
  49. package/scripts/g2_g4_pointback.py +77 -4
  50. package/scripts/g5_preview.py +102 -3
  51. package/scripts/g6_evidence.py +145 -11
  52. package/scripts/g6_records.py +47 -0
  53. package/scripts/g6_warnings.py +36 -17
  54. package/scripts/g7_contract_drift.py +9 -0
  55. package/scripts/g8_run_registry.py +29 -7
  56. package/scripts/g9_shaping.py +50 -10
  57. package/scripts/generate_adapter.py +16 -1
  58. package/scripts/governance_jsonl.py +42 -0
  59. package/scripts/method_semantics.py +7 -1
  60. package/scripts/promotion_governance.py +185 -0
  61. package/scripts/run_console.py +9 -0
  62. package/scripts/run_continuation.py +21 -4
  63. package/scripts/run_handoff.py +9 -0
  64. package/scripts/run_metadata.py +0 -7
  65. package/scripts/run_profile.py +10 -0
  66. package/scripts/run_status.py +45 -0
  67. package/scripts/shaping_log.py +48 -2
  68. package/scripts/stages.py +16 -0
  69. package/scripts/stdio_encoding.py +42 -0
  70. package/scripts/validate_run.py +18 -0
  71. package/skills/component-distill/SKILL.md +52 -0
  72. package/skills/craft-guard/SKILL.md +2 -2
  73. package/skills/craft-guard/references/detectors.md +1 -0
  74. package/skills/design-baseline/SKILL.md +10 -4
  75. package/skills/design-baseline/scripts/design_baseline.py +250 -2
  76. package/skills/design-playbook/SKILL.md +26 -6
  77. package/skills/design-playbook/references/observe-ops.md +3 -1
  78. package/skills/design-playbook/references/preview-ops.md +2 -1
  79. package/skills/ui-evaluator/SKILL.md +2 -2
  80. package/skills/ui-evaluator/references/a11y-tree.md +4 -0
  81. package/skills/ui-picker/SKILL.md +1 -1
  82. package/skills/ux-spec/SKILL.md +6 -4
  83. package/skills/ux-spec/references/spec-template.md +2 -1
  84. package/mcp/evidence/test_capture_contract.py +0 -355
  85. package/mcp/evidence/test_containment.py +0 -733
  86. package/mcp/evidence/test_delivery_matrix.py +0 -159
  87. package/mcp/evidence/test_disclosure.py +0 -309
  88. package/mcp/evidence/test_evidence_preflight.py +0 -288
  89. package/mcp/evidence/test_handoff.py +0 -672
  90. package/mcp/evidence/test_handoff_i18n.py +0 -125
  91. package/mcp/evidence/test_ledger_syntax.py +0 -251
  92. package/mcp/evidence/test_page_defects.py +0 -90
  93. package/mcp/evidence/test_server_stdio.py +0 -821
  94. package/mcp/preview/test_browser_control.py +0 -1284
  95. package/mcp/preview/test_i18n_labels.py +0 -92
  96. package/mcp/preview/test_integrity.py +0 -237
  97. package/mcp/preview/test_preview_usability.py +0 -247
  98. package/mcp/preview/test_server_stdio.py +0 -200
  99. package/mcp/preview/test_transaction.py +0 -810
  100. package/mcp/preview/test_versions.py +0 -625
  101. package/mcp/preview/test_versions_freeze.py +0 -176
  102. package/mcp/run_console/test_actions.py +0 -723
  103. package/mcp/run_console/test_contract.py +0 -668
  104. package/mcp/run_console/test_diagnostic_export_gate.py +0 -1540
  105. package/mcp/run_console/test_http_server.py +0 -1041
  106. package/mcp/run_console/test_parity.py +0 -1129
  107. package/mcp/run_console/test_read_only_trial.py +0 -583
  108. package/mcp/run_console/test_repair_packet.py +0 -840
  109. package/mcp/run_console/test_request_security.py +0 -325
  110. package/mcp/run_console/test_role_attestation_gate.py +0 -945
  111. package/mcp/run_console/test_session.py +0 -336
  112. package/mcp/run_console/test_snapshot_builder.py +0 -912
  113. package/mcp/run_console/test_source_registry.py +0 -950
  114. package/mcp/run_console/test_ui.py +0 -444
  115. package/mcp/run_console/test_ui_actions.py +0 -295
  116. package/mcp/run_console/test_ui_browser.py +0 -951
package/README.md CHANGED
@@ -1,6 +1,12 @@
1
1
  # design-playbook
2
2
 
3
- Agent plugin: **Design I/O** for product UI (Claude Code / Codex).
3
+ Agent plugin for **evidence-backed UI delivery** in existing Web products
4
+ (Claude Code / Codex). **Design I/O** is the declaration and contract mechanism.
5
+
6
+ The project is in maintainer self-use and maintenance. Catalog submissions
7
+ and recruitment are paused; installed paths remain available. Missing required
8
+ proof stays `blocked`; an explicitly skipped evaluator records `audited: false`,
9
+ not an audited Pass. Evaluator review does not replace human semantic approval.
4
10
 
5
11
  Declarations + contracts — not a style CSV pack. Compose with [ui-ux-pro-max](https://github.com/nextlevelbuilder/ui-ux-pro-max-skill) and Anthropic `frontend-design` for aesthetics; this package owns pipeline and acceptance.
6
12
 
@@ -52,6 +58,7 @@ After install, skills and commands are **namespaced** by the plugin name:
52
58
  | `/design-playbook:craft-guard` | Craft / anti-slop skill |
53
59
  | `/design-playbook:native-craft` | Native-feel desktop declaration skill |
54
60
  | `/design-playbook:ui-evaluator` | Point-back acceptance skill |
61
+ | `/design-playbook:component-distill` | Cross-run component/token proposal skill and command; report-only, promotion requires a user decision |
55
62
  | `/design-playbook:design-io` | Full pipeline command |
56
63
  | `/design-playbook:ux-spec` | Spec-only command |
57
64
  | `/design-playbook:ui-review` | Review command |
@@ -75,8 +82,8 @@ pi has no plugin namespace — skills are `/skill:<name>`, commands are bare `/<
75
82
  | Invoke | Role |
76
83
  | --- | --- |
77
84
  | `/skill:design-playbook` | Orchestrator skill (model-invoked) |
78
- | `/skill:ux-spec` … `/skill:ui-evaluator` | Same eight skills as above |
79
- | `/design-io` · `/ux-spec` · `/ui-review` · `/run-review` · `/run-status` · `/run-handoff` · `/doctor` | Pipeline / spec-only / review / cross-run / status / handoff / health commands |
85
+ | `/skill:ux-spec` … `/skill:component-distill` | Same nine skills as above |
86
+ | `/design-io` · `/ux-spec` · `/ui-review` · `/run-review` · `/run-status` · `/run-handoff` · `/doctor` · `/component-distill` | Pipeline / spec-only / review / cross-run / status / handoff / health / proposal commands |
80
87
 
81
88
  pi ships no built-in MCP, so `preview*` and `observe*` skip by default (ADR-0009 absent→skip; the pipeline still runs spec → picker → fill → craft → accept). To enable both gates, install an MCP adapter and register the bundled servers in your project `.mcp.json`:
82
89
 
@@ -113,7 +120,7 @@ npx design-playbook init <agent> # e.g. cursor, gemini-cli, windsurf
113
120
  npx design-playbook --list # all 30 agents, shows which have renderers
114
121
  ```
115
122
 
116
- See the root [README](../../README.md#install-on-other-agents) for the tier table and capability notes.
123
+ See the root [README](../../README.md#-install-on-other-agents) for the tier table and capability notes.
117
124
 
118
125
  ## Stack with other skills
119
126
 
@@ -131,7 +138,7 @@ See the root [README](../../README.md#install-on-other-agents) for the tier tabl
131
138
  .mcp.json ← bundled MCP servers, launched via ${CLAUDE_PLUGIN_ROOT} (ADR-0009)
132
139
  mcp/{preview,evidence}/← MCP adapter runtimes (preview_prototype / execute_capture_plan)
133
140
  skills/<name>/SKILL.md ← model-invoked skills
134
- commands/<name>.md ← slash commands (design-io, ux-spec, ui-review, run-review, run-status, run-handoff, doctor)
141
+ commands/<name>.md ← slash commands; see invocation tables above
135
142
  codex/AGENTS.md ← Codex bridge notes
136
143
  examples/ ← self-authored onboarding samples
137
144
  LICENSE · NOTICE ← authored-only scope
@@ -159,6 +166,13 @@ python <pkg>/scripts/run_status.py --list # newest runs under .
159
166
 
160
167
  The status command reuses the packaged validator’s G5 confirm rules. It is part of the installed package — not monorepo-only tooling. For an eligible run it also reports an explicit `open-console` continuation command for the local Run Console (it never starts a server itself), with the blocking reason and a safe fallback when the run or Console prerequisites are ineligible. The Console’s current claim is **local, experimental, and trial-gated** — `run-status --json` reports the same capability receipt (`publicClaim: experimental`) — and no authorized external trial or public release is claimed until the separately authorized read-only trial gate passes (ADR-0043).
161
168
 
169
+ The source checkout adds a bounded self-use report:
170
+ `python <pkg>/scripts/run_status.py <run> --scope path:P1 --json`.
171
+ It links explicit declarations, summarizes evidence gaps, and retains the original
172
+ owner's re-verification requirements. Unknown impact cannot safely narrow that
173
+ scope. See [scope inputs, source refresh, and limits](commands/run-status.md).
174
+ This is not a release or measured productivity claim.
175
+
162
176
  ### Static handoff
163
177
 
164
178
  ```text
@@ -176,7 +190,7 @@ python <pkg>/scripts/doctor.py --json
176
190
 
177
191
  One packaged diagnosis for interpreter, package surface, optional Playwright, and run-root configuration. Distinguishes `ok` / `degraded` / `broken` with repair actions. Those three states describe **install and runtime health** of what is present locally — they are not a public capability-maturity verdict; maturity vocabulary (`stable` / `experimental` / `blocked-by-gate` / `not-shipped`) stays with the `run-status` capability receipt, and doctor reads existing facts rather than adding a new health or capability-state authority.
178
192
 
179
- **Bundled MCP (v0.3+):** Preview (`mcp/preview/`) and Evidence (`mcp/evidence/`) runtimes ship inside this package and are registered by `.mcp.json` (`${CLAUDE_PLUGIN_ROOT}`). Sibling monorepo dirs remain compatibility launchers/docs. The orchestrator still **probes** MCP `tools/list` and skips `preview*` / `observe*` when tools are absent. Evidence provider writes artifacts only — never the manifest. **`DESIGN_PLAYBOOK_RUN_ROOT`:** default `"."` in `.mcp.json` is the **MCP process cwd**, not the chat workspace — for a host-app dogfood, set an **absolute** path to `.scratch/<run>/` (see [`mcp/evidence/README.md`](mcp/evidence/README.md)). Capture responses include `written_path` (absolute) so mis-rooted writes are visible without a filesystem search.
193
+ **Bundled MCP (v0.3+):** Preview (`mcp/preview/`) and Evidence (`mcp/evidence/`) runtimes ship inside this package and are registered by `.mcp.json` (`${CLAUDE_PLUGIN_ROOT}`). Sibling monorepo dirs remain compatibility launchers/docs. The orchestrator still **probes** MCP `tools/list` and skips `preview*` / `observe*` when tools are absent. Evidence provider writes artifacts only — never the manifest. **`DESIGN_PLAYBOOK_RUN_ROOT`:** `.mcp.json` passes this variable through without pinning a default. With no explicit root, artifacts resolve under the **MCP process cwd**, not necessarily the chat workspace. For a host-app run, use an absolute `.scratch/<run>/` root; per-call overrides and marker requirements are documented in [`mcp/evidence/README.md`](mcp/evidence/README.md). Capture responses include `written_path` (absolute) so mis-rooted writes are visible without a filesystem search.
180
194
 
181
195
  What is **deterministically enforced** today: repository install/structure CI checks and the run-artifact shape (`scripts/validate_run.py` — L1–L6 present; every top-level L6 item ordered `Given -> When -> Then`; one non-empty four-field evidence ledger row per `L6.<n>` with allowed results; four non-empty finding fields with non-empty source; exactly one explicit `## Verdict` of `Pass` or `Recirculate`; Pass requires every evidence result to be `pass` and exactly one issue-linked `0 blocking` closure per blocking finding; exit 0/`RUN OK`, exit 1/`RUN INVALID`, exit 2/`RUN ERROR`; regression-tested by `tests/test_validate_run.py`, which also validates the showcase artifacts directly; **G5** is a *conditional* preview-confirm gate — enforced only when preview artifacts exist / `--preview-dir` is used; **G6** is a *conditional* evidence-binding gate — enforced only when a ledger `observed` references an `evidence/` artifact / `--evidence-dir` is used; opt-in **strict mode** via `--require-preview` / `--require-evidence` / `--strict`). The `observe*` step probes MCP tool `execute_capture_plan` and is skipped when absent. Everything else in the pipeline is agent-executed craft judgment, not a machine gate.
182
196
 
package/codex/AGENTS.md CHANGED
@@ -1,4 +1,4 @@
1
- <!-- generated-by design-playbook v0.24.4 -->
1
+ <!-- generated-by design-playbook v0.25.1 -->
2
2
  # design-playbook for Codex
3
3
 
4
4
  ## Install (path of record)
@@ -66,4 +66,13 @@ def main(argv: list[str] | None = None) -> int:
66
66
 
67
67
 
68
68
  if __name__ == "__main__":
69
+ # One pipe-encoding seam (T-105): UTF-8 on piped stdout/stderr
70
+ # regardless of the host code page. See scripts/stdio_encoding.py.
71
+ for _candidate in Path(__file__).resolve().parents:
72
+ if (_candidate / "design_playbook.py").is_file():
73
+ sys.path.insert(0, str(_candidate))
74
+ break
75
+ from design_playbook.scripts.stdio_encoding import configure_piped_utf8
76
+
77
+ configure_piped_utf8()
69
78
  sys.exit(main())
@@ -0,0 +1,8 @@
1
+ ---
2
+ description: Cross-run component backflow — derive recurring components from iterated pages and render a propose-only DESIGN.md promotion proposal (report-only, user-gated)
3
+ ---
4
+
5
+ Cross-run **component backflow** over `.scratch/<run>/` decision reports in the user project. Not a step of a single Design I/O run. Run skill **component-distill**. Proposal header **`component-distill/v1`**; report-only — never writes `DESIGN.md` or any authority; promotion is a user decision recorded in the governance log and executed only by `design_baseline.py promote`.
6
+
7
+ Output path:
8
+ $ARGUMENTS
@@ -18,6 +18,90 @@ python <plugin>/scripts/run_status.py [.scratch/<run>] [--json] [--list] [--scra
18
18
 
19
19
  For an eligible run the continuation names an explicit `open-console` command for the local, experimental, trial-gated Run Console — no authorized external trial or public release is claimed until the separately authorized read-only trial gate passes (ADR-0043). `run-status` never starts a server, daemon, or background process itself; an ineligible run reports the blocking reason and a safe fallback instead. On Windows the emitted command line is PowerShell syntax (single-quoted literals behind the `&` call operator) — paste it into PowerShell, not cmd.exe.
20
20
 
21
+ ## Explicit frontend scope review
22
+
23
+ The source checkout also provides a bounded, read-only report for self-use. It
24
+ has no separate Console panel or persistent mapping, and no measured time-saving
25
+ or released-package claim.
26
+
27
+ ```text
28
+ python <plugin>/scripts/run_status.py <run> --scope path:P1 --scope page:checkout
29
+ python <plugin>/scripts/run_status.py <run> --scope component:Retry --json
30
+ python <plugin>/scripts/run_status.py <run> --scope path:P1 --git-root <repo> --base-revision <base> --head-revision <head> --json
31
+ python <plugin>/scripts/run_status.py <run> --scope path:P1 --git-root <repo> --base-revision <base> --worktree --json
32
+ ```
33
+
34
+ - A run path and at least one `--scope` are required. Scope mode does not discover
35
+ runs and cannot combine with `--list`; `--scratch` only applies to discovery.
36
+ Its JSON is a scope report, not the legacy status JSON or Run Snapshot v1. It
37
+ carries `schemaVersion: 1`; every member of `evidence_gaps[].bindings[]` has the
38
+ same keys `integrity`, `reasons`, `source`, `content_hash`, and an unbound or
39
+ unreadable entry reports `source: null` / `content_hash: null` rather than
40
+ dropping those keys.
41
+ - Repeated scopes accept `path:P1`, `page:<id>`, `component:<id>`, and
42
+ `file:<relative/path>`. Paths must stay within the explicit Git root, or the
43
+ selected run if no Git root was given. Absolute paths and traversal are rejected.
44
+ - Known links require an L3 `Path | Steps` row and an L6 `(path: P1)` reference.
45
+ Page links additionally require an exact L2 `Page | Duty` entry and an exact
46
+ step token separated by `->` or an arrow. Free-form prose is not a code
47
+ dependency map. Assumptions stay `assumed`; component and file clues currently
48
+ remain `unknown`. No association means unknown, never unaffected.
49
+ - Git is optional. When requested, supply the actual repository root, a base,
50
+ and exactly one of a head revision or `--worktree`. Worktree mode includes
51
+ untracked paths. The report neither guesses a base nor scans source code.
52
+
53
+ Evidence requirements are read only where an L6 item already declares a precise
54
+ continuation, for example:
55
+
56
+ ```text
57
+ - Given checkout fails When retry is offered Then recovery is visible (path: P1)
58
+ Required evidence: screenshot; state=error; viewport=390x844
59
+ ```
60
+
61
+ The recognized proof values are `screenshot`, `a11y_tree`, and
62
+ `interaction_trace`; `state` and `viewport` are optional and appear in that order.
63
+ This optional read convention adds no mandatory spec field. Missing or
64
+ unrecognized prose stays `unknown`; conflicting declarations stay `inconsistent`.
65
+ The separate sampling summary enumerates only nonblank L5 state cells.
66
+ `reported` means a sampling claim, not a verified proof binding.
67
+
68
+ The report separates source availability, binding integrity, evidence gaps, and
69
+ the original evaluator result. Missing required proof is `blocked`; skipping the
70
+ evaluator remains `unaudited`. An owner ledger that exists but cannot be
71
+ projected is never read as an absent one: every gap gains the
72
+ `pointback-malformed` reason, and sources that are still current report the
73
+ evaluation as `inconsistent` rather than `unknown`. Binding timestamps order
74
+ by instant, so mixed `Z` / `+HH:MM` stamps pick the real latest capture: a
75
+ stamp that is missing, offset-less, or not ISO-8601 reports
76
+ `invalid-binding-timestamp`, and two distinct entries sharing the latest
77
+ instant report `conflicting-bindings`; both leave the binding `inconsistent`.
78
+ Not-applicable and unreviewed reasons are referenced at their source, not
79
+ copied as private prose.
80
+ It does not start a Provider, read browser login state, emit capture URLs or
81
+ source code, write run artifacts, or grant semantic approval. Treat scope names
82
+ and relative file paths as local data.
83
+
84
+ Reverification candidates retain the original Repair Packet's invalidated set,
85
+ owner, resume stage, and recapture requirement. **They cannot safely narrow the
86
+ original owner scope**: declaration links do not prove exhaustive code impact.
87
+ No repair or acceptance command is executed.
88
+
89
+ Before copying an owner command from an earlier report, repeat the same inputs
90
+ with its `source_hash`:
91
+
92
+ ```text
93
+ python <plugin>/scripts/run_status.py <run> --scope path:P1 --expected-source-hash sha256:<previous-hash> --json
94
+ ```
95
+
96
+ Changed or owner-unverified sources make the report `stale` and remove the copyable command.
97
+ Unavailable contracts also suppress it. Refresh the report and use the existing
98
+ owner workflow; a source check does not lock files against later edits.
99
+ Exit 0 means a report was produced, not that the run passed acceptance.
100
+ Invalid inputs or unreadable/out-of-root sources exit 2.
101
+
21
102
  ## Done when
22
103
 
23
104
  The command names completed stage markers, any active blocker (preview floor, baseline gate, recirculate verdict), and the single next valid resume action. It reuses `validate_run` judgments for G5 confirm validity rather than inventing a second state machine. Stale, partial, hash-mismatched, malformed, or inconsistent runs stay visible as those states and are never replaced by an older successful snapshot.
105
+
106
+ In scope mode, known links are traceable to their declaration and hash, uncertain
107
+ links remain explicit, and diagnostic gaps never replace the evaluator verdict.
package/lib/index.js CHANGED
@@ -8,8 +8,9 @@
8
8
  * `skills/` directory. The plugin locates the directory via `__dirname`,
9
9
  * so no `!!js` expression and no cwd-dependent resolution is involved.
10
10
  *
11
- * 2. Slash commands (ctx.commands) — `design-io`, `doctor`,
12
- * `run-handoff`, `run-review`, `run-status`, `ui-review`, `ux-spec` —
11
+ * 2. Slash commands (ctx.commands) — `component-distill`, `design-io`,
12
+ * `doctor`, `run-handoff`, `run-review`, `run-status`, `ui-review`,
13
+ * `ux-spec` —
13
14
  * that load the matching `commands/<name>.md` prompt, substitute
14
15
  * `$ARGUMENTS` with the raw trailing input, and inject it as a
15
16
  * user-role follow-up turn via `agent.followup()`.
@@ -87,6 +88,7 @@ exports.createUserMessageFromPrompt = createUserMessageFromPrompt
87
88
  * substituted) is injected as a user follow-up turn.
88
89
  */
89
90
  const COMMAND_NAMES = [
91
+ 'component-distill',
90
92
  'design-io',
91
93
  'doctor',
92
94
  'run-handoff',
@@ -6,11 +6,21 @@ Runtime for the optional **observe\*** step. Writes capture artifacts only — *
6
6
 
7
7
  | Setting | Meaning |
8
8
  | --- | --- |
9
- | Unset | Artifact paths resolve under the **MCP process cwd** |
10
- | `"."` (default in package `.mcp.json`) | Same — relative to process cwd, **not** the chat workspace root |
9
+ | Unset | Artifact paths resolve under the **MCP process cwd** (the packaged `.mcp.json` passes the variable through, it does not pin a value) |
10
+ | `"."` (a host config that pins it) | Same — relative to process cwd, **not** the chat workspace root |
11
11
  | Absolute path | Preferred for cross-repo dogfood: set to the run root (e.g. `/path/to/host-app/.scratch/playbook-smoke/<run>`) so `evidence/L6.*.png` lands next to `manifest.jsonl` |
12
12
 
13
- Relative values are resolved with `Path(value).resolve()` at process start semantics (cwd-relative). If captures appear under the plugin monorepo instead of the host run, check cwd and this env — the tool also returns **`written_path`** (absolute) so mis-roots are obvious without a filesystem search.
13
+ Relative values are resolved with `Path(value).resolve()` at process start semantics (cwd-relative). If captures appear under the plugin monorepo instead of the host run, check cwd and this env — the tool also returns **`written_path`** (absolute) plus a `warnings` entry so mis-roots are obvious without a filesystem search.
14
+
15
+ **Cannot restart the server (`--plugin-dir` dev load)?** Pass `run_root` on the tool call — an absolute `.scratch/<run>/` that already carries a run marker (`plan.md` / `point-back.md`). Resolution order is `run_root` → `DESIGN_PLAYBOOK_RUN_ROOT` → process cwd, and the argument binds only that one call, so a mistyped or markerless root fails the capture instead of scattering evidence.
16
+
17
+ ## What each capture type writes
18
+
19
+ | `type` | Bytes | Name it |
20
+ | --- | --- | --- |
21
+ | `screenshot` | PNG (plus a `<stem>.probe.json` sidecar when the adapter can probe) | `evidence/<leaf>.png` |
22
+ | `a11y tree` | JSON envelope `{"format": "aria_snapshot", "tree": "…"}` — `tree` is Playwright's indentation text, not a node/role tree | `evidence/<leaf>.json` |
23
+ | `interaction trace` | Playwright trace **ZIP** (actions + snapshots inside) | `evidence/<leaf>.trace.zip` — the Provider refuses a non-`.zip` name |
14
24
 
15
25
  Example (host run):
16
26
 
@@ -20,7 +30,7 @@ Example (host run):
20
30
  }
21
31
  ```
22
32
 
23
- Plugin auto-load: [`../.mcp.json`](../.mcp.json) (under `packages/design-playbook/`).
33
+ Plugin auto-load: [`../../.mcp.json`](../../.mcp.json) (under `packages/design-playbook/`).
24
34
  Codex / manual: sibling [`mcp.example.toml`](../../../design-playbook-evidence/mcp.example.toml).
25
35
 
26
36
  ## Return shape
@@ -11,7 +11,8 @@ from __future__ import annotations
11
11
  import json
12
12
  import math
13
13
  import os
14
- from pathlib import Path
14
+ from contextvars import ContextVar
15
+ from pathlib import Path, PurePosixPath, PureWindowsPath
15
16
  from typing import Any, Protocol
16
17
 
17
18
  from design_playbook.mcp.evidence import containment
@@ -22,6 +23,7 @@ from design_playbook.mcp.evidence.action_params import (
22
23
  from design_playbook.mcp.evidence.capture_contract import parse_capture_contract
23
24
  from design_playbook.mcp.evidence.path_syntax import (
24
25
  probe_sidecar_rel,
26
+ trace_artifact_error,
25
27
  trimmed_relpath,
26
28
  )
27
29
  from design_playbook.mcp.evidence.disclosure import (
@@ -50,6 +52,7 @@ ALLOWED_ARGUMENTS = frozenset(
50
52
  "viewport",
51
53
  "freeze",
52
54
  "storage_state",
55
+ "run_root",
53
56
  }
54
57
  )
55
58
  RUN_ROOT_ENV = "DESIGN_PLAYBOOK_RUN_ROOT"
@@ -110,6 +113,10 @@ def _failed(
110
113
  }
111
114
  if request is not None:
112
115
  payload["request"] = request
116
+ if written_path and _run_root_misrooted():
117
+ # Same misroot warning as the success payload: a failed write outside
118
+ # the run tree is exactly where the orchestrator needs the hint.
119
+ payload["warnings"] = [_MISROOTED_WARNING]
113
120
  return payload
114
121
 
115
122
 
@@ -138,6 +145,10 @@ def _captured(
138
145
  "written_path": written_path,
139
146
  "request": request,
140
147
  }
148
+ if _run_root_misrooted():
149
+ # The stderr warning fires once per process and may never reach the
150
+ # model; the payload is what the orchestrator actually reads.
151
+ payload["warnings"] = [_MISROOTED_WARNING]
141
152
  if probe_artifact:
142
153
  payload["probe_artifact"] = probe_artifact
143
154
  return payload
@@ -270,20 +281,87 @@ def _apply_freeze(page: Any, freeze: dict[str, Any]) -> None:
270
281
  _RUN_MARKERS = ("plan.md", "point-back.md")
271
282
  _warned_run_root = False
272
283
 
284
+ # Per-call run root (DEF-4): a `--plugin-dir` dev host cannot edit the shipped
285
+ # DESIGN_PLAYBOOK_RUN_ROOT after the MCP process starts, so
286
+ # `execute_capture_plan` accepts a `run_root` argument that binds the
287
+ # resolution for the duration of one call and nothing longer.
288
+ _CALL_RUN_ROOT: ContextVar["Path | None"] = ContextVar(
289
+ "design_playbook_call_run_root", default=None
290
+ )
291
+
292
+ _MISROOTED_WARNING = (
293
+ "run root resolved to a markerless cwd (DESIGN_PLAYBOOK_RUN_ROOT "
294
+ "unset or '.'); written_path is outside the run tree — pass "
295
+ "run_root=<abs .scratch/<run>/> on the recapture (no server restart "
296
+ "needed) or set DESIGN_PLAYBOOK_RUN_ROOT before launch"
297
+ )
298
+
299
+
300
+ def _has_run_marker(root: Path) -> bool:
301
+ """True when ``root`` carries one of the run's own marker artifacts."""
302
+ return any((root / marker).is_file() for marker in _RUN_MARKERS)
303
+
304
+
305
+ def _validated_call_run_root(value: object) -> Path:
306
+ """Normalize an explicit ``run_root`` argument.
307
+
308
+ Absolute, existing, and marker-carrying: a mistyped root must fail the
309
+ call rather than scatter evidence into a directory that is not a run.
310
+ """
311
+ if not isinstance(value, str) or not value.strip():
312
+ raise ValueError("run_root must be a non-empty absolute path string")
313
+ raw = value.strip()
314
+ if not (
315
+ Path(raw).is_absolute()
316
+ or PureWindowsPath(raw).is_absolute()
317
+ or PurePosixPath(raw).is_absolute()
318
+ ):
319
+ raise ValueError(f"run_root must be an absolute path (got {raw!r})")
320
+ root = Path(raw).expanduser().resolve()
321
+ if not root.is_dir():
322
+ raise ValueError(
323
+ f"run_root {root} is not an existing directory; pass the run root "
324
+ ".scratch/<run>/"
325
+ )
326
+ if not _has_run_marker(root):
327
+ raise ValueError(
328
+ f"run_root {root} carries no run marker "
329
+ f"({' / '.join(_RUN_MARKERS)}); pass the run root, not the "
330
+ "workspace root or the evidence/ directory"
331
+ )
332
+ return root
333
+
334
+
335
+ def _run_root_misrooted() -> bool:
336
+ """True when the run root fell back to a markerless cwd.
337
+
338
+ The shipped .mcp.json default (DESIGN_PLAYBOOK_RUN_ROOT=".") makes the
339
+ server resolve artifacts under its process cwd; in a host workspace that
340
+ cwd is the repo root, so captures silently land outside the run tree.
341
+ A per-call ``run_root`` is marker-validated before use, so it is never
342
+ misrooted.
343
+ """
344
+ if _CALL_RUN_ROOT.get() is not None:
345
+ return False
346
+ configured = os.environ.get(RUN_ROOT_ENV)
347
+ if configured and configured != ".":
348
+ return False
349
+ return not _has_run_marker(Path.cwd().resolve())
350
+
273
351
 
274
352
  def _run_root() -> Path:
353
+ explicit = _CALL_RUN_ROOT.get()
354
+ if explicit is not None:
355
+ return explicit
275
356
  configured = os.environ.get(RUN_ROOT_ENV)
276
357
  if not configured or configured == ".":
277
- # cwd-relative default silently mis-roots multi-run workspaces (the
278
- # root .mcp.json ships DESIGN_PLAYBOOK_RUN_ROOT="."). Warn only when
279
- # cwd does not look like a run dir (no run marker file) — the shipped
280
- # default resolving to a real run dir is correct usage, not a
281
- # misconfig — and only once per process to avoid per-capture spam.
358
+ # Warn only when cwd does not look like a run dir (no run marker
359
+ # file) — the shipped default resolving to a real run dir is correct
360
+ # usage, not a misconfig — and only once per process to avoid
361
+ # per-capture spam.
282
362
  root = Path.cwd().resolve()
283
363
  global _warned_run_root
284
- if not _warned_run_root and not any(
285
- (root / marker).is_file() for marker in _RUN_MARKERS
286
- ):
364
+ if not _warned_run_root and _run_root_misrooted():
287
365
  _warned_run_root = True
288
366
  _log(
289
367
  "WARNING: DESIGN_PLAYBOOK_RUN_ROOT is unset or '.' "
@@ -291,7 +369,8 @@ def _run_root() -> Path:
291
369
  f"({' / '.join(_RUN_MARKERS)}); artifacts resolve under "
292
370
  f"{root}/evidence/. Set DESIGN_PLAYBOOK_RUN_ROOT to the run "
293
371
  "root when the host workspace is not the intended run "
294
- "directory."
372
+ "directory, or pass run_root=<abs run root> on the capture "
373
+ "call when the server cannot be restarted."
295
374
  )
296
375
  return root
297
376
  return Path(configured).resolve()
@@ -765,6 +844,34 @@ def _validate_runtime_object(
765
844
  def execute_capture_plan(
766
845
  args: dict[str, Any],
767
846
  browser_adapter: BrowserAdapter | None = None,
847
+ ) -> dict[str, Any]:
848
+ """Capture one artifact, then report the result (never a verdict).
849
+
850
+ Run root resolution order: an explicit ``run_root`` argument (bound to
851
+ this call only) → ``DESIGN_PLAYBOOK_RUN_ROOT`` → process cwd. The explicit
852
+ root is marker-validated before use, so a dev host that cannot restart the
853
+ MCP server still lands evidence inside the run tree (DEF-4).
854
+ """
855
+ raw_run_root = args.get("run_root")
856
+ if raw_run_root is None:
857
+ return _capture(args, browser_adapter)
858
+ artifact = args.get("artifact_path")
859
+ label = artifact if isinstance(artifact, str) else ""
860
+ try:
861
+ root = _validated_call_run_root(raw_run_root)
862
+ except (ValueError, OSError) as exc:
863
+ # OSError is in range: Path.resolve() rejects host-illegal characters.
864
+ return _failed(label, str(exc))
865
+ token = _CALL_RUN_ROOT.set(root)
866
+ try:
867
+ return _capture(args, browser_adapter)
868
+ finally:
869
+ _CALL_RUN_ROOT.reset(token)
870
+
871
+
872
+ def _capture(
873
+ args: dict[str, Any],
874
+ browser_adapter: BrowserAdapter | None,
768
875
  ) -> dict[str, Any]:
769
876
  unknown = sorted(set(args) - ALLOWED_ARGUMENTS)
770
877
  if unknown:
@@ -797,6 +904,9 @@ def execute_capture_plan(
797
904
  raise ValueError("storage_state must be a string path when provided")
798
905
 
799
906
  rel = artifact_path.strip()
907
+ trace_error = trace_artifact_error(cap_type, rel)
908
+ if trace_error:
909
+ return _failed(rel, trace_error, request=request)
800
910
  try:
801
911
  out_path = _resolve_artifact_path(rel)
802
912
  except ValueError as exc:
@@ -23,6 +23,11 @@ consume it instead of mirroring the escape classes. The ``evidence/``
23
23
  operations remain the ADR-0026 contract surface - same reason codes, same
24
24
  existence timing - now expressed as specializations of the one resolver.
25
25
 
26
+ ADR-0044 adds the Diagnostic export write boundary as a third specialization:
27
+ ``trial_export_write_target`` confines one bare filename under
28
+ ``<run_root>/trial-export/`` (the trial-export subtree), with the same
29
+ reason-code discipline and TOCTOU limit.
30
+
26
31
  Threat-model limit (ADR-0026, explicit): this module resolves and validates
27
32
  the path; it does NOT perform the write. Path resolution alone cannot close
28
33
  the TOCTOU gap - a concurrent untrusted filesystem actor that replaces a
@@ -50,6 +55,9 @@ REASON_RESOLUTION_FAILURE = "resolution_failure"
50
55
  REASON_CANONICAL_ESCAPE = "canonical_escape"
51
56
  REASON_SYMLINK_ESCAPE = "symlink_escape"
52
57
  REASON_NOT_REGULAR_FILE = "not_regular_file"
58
+ # Trial-export targets (ADR-0044) accept bare filenames only; a name that
59
+ # carries any separator or reserved form is rejected before resolution.
60
+ REASON_RESERVED_NAME = "reserved_name"
53
61
 
54
62
  # Every resolution-time escape reason (the classes the ADR requires both
55
63
  # operations to reject at resolution time). The Provider treats all of these
@@ -203,3 +211,86 @@ def read_artifact(artifact_path: str, run_root: Path) -> ContainmentResult:
203
211
  must not bind a directory or a missing path).
204
212
  """
205
213
  return _resolve(artifact_path, run_root, require_existing_file=True)
214
+
215
+
216
+ # The Diagnostic export write boundary (ADR-0044): one subtree, sibling of
217
+ # evidence/, owned here so the export transaction cannot disagree with the
218
+ # one containment authority on where trial exports may land.
219
+ TRIAL_EXPORT_SUBDIR = "trial-export"
220
+
221
+ # Names the export subtree may never carry. ``manifest.jsonl`` is reserved
222
+ # across the run tree (the Evidence Manifest authority); the current-directory
223
+ # name is a no-op write and is refused as malformed rather than silently
224
+ # permitted.
225
+ _TRIAL_EXPORT_RESERVED_NAMES = frozenset({"manifest.jsonl", "", ".", ".."})
226
+
227
+ # Win32 name quirks that CreateFile folds but Path.resolve does not: a
228
+ # trailing dot or space vanishes on write (the on-disk name would diverge
229
+ # from the reviewed one), and the reserved device names are never regular
230
+ # files. The export writes exactly the reviewed pair, so both classes are
231
+ # rejected as reserved.
232
+ _WIN32_DEVICE_NAMES = frozenset(
233
+ {"CON", "PRN", "AUX", "NUL",
234
+ *(f"COM{i}" for i in range(1, 10)),
235
+ *(f"LPT{i}" for i in range(1, 10))},
236
+ )
237
+
238
+
239
+ def trial_export_write_target(filename: str, run_root: Path) -> ContainmentResult:
240
+ """Resolve a Diagnostic export write target under ``trial-export/``.
241
+
242
+ ``filename`` must be a bare filename - no directory separators (native,
243
+ POSIX, or Windows), no drive form, no ``..`` segment, and not one of the
244
+ reserved names, Win32 fold forms (trailing dot or space), or reserved
245
+ device stems - anything the platform would write under a different name
246
+ than the one reviewed. Same resolution-time escape
247
+ rejection and TOCTOU limit as every operation in this module; the
248
+ transaction performs the staged write and rollback around this resolution.
249
+ """
250
+ if not isinstance(filename, str) or filename == "":
251
+ return ContainmentResult(None, REASON_RESERVED_NAME)
252
+ if filename in _TRIAL_EXPORT_RESERVED_NAMES:
253
+ return ContainmentResult(None, REASON_RESERVED_NAME)
254
+ # Bare-filename precondition: any separator, drive, or traversal form
255
+ # fails before the generic resolver can even see it.
256
+ if (
257
+ PurePosixPath(filename).is_absolute()
258
+ or PureWindowsPath(filename).is_absolute()
259
+ or "/" in filename
260
+ or "\\" in filename
261
+ or any(part == ".." for part in PurePosixPath(filename).parts)
262
+ or any(part == ".." for part in PureWindowsPath(filename).parts)
263
+ or ":" in filename
264
+ ):
265
+ return ContainmentResult(None, REASON_ABSOLUTE_PATH)
266
+ # Win32 fold classes: a trailing dot/space or a reserved device stem
267
+ # would make the on-disk name differ from the reviewed name.
268
+ if filename != filename.rstrip(" ."):
269
+ return ContainmentResult(None, REASON_RESERVED_NAME)
270
+ if filename.split(".", 1)[0].upper() in _WIN32_DEVICE_NAMES:
271
+ return ContainmentResult(None, REASON_RESERVED_NAME)
272
+ # The boundary subtree itself must be a real child of the run root: a
273
+ # ``trial-export`` symlink pointing outside the run root would make the
274
+ # generic under-boundary check pass while writing outside the selected
275
+ # run, so the resolved boundary is required to stay inside the resolved
276
+ # run root before anything else is resolved against it.
277
+ try:
278
+ resolved_root = run_root.resolve(strict=False)
279
+ boundary = (run_root / TRIAL_EXPORT_SUBDIR).resolve(strict=False)
280
+ Path(os.path.realpath(boundary)).relative_to(
281
+ Path(os.path.realpath(resolved_root))
282
+ )
283
+ except (OSError, ValueError):
284
+ return ContainmentResult(None, REASON_SYMLINK_ESCAPE)
285
+ result = _resolve_candidate(
286
+ run_root,
287
+ f"{TRIAL_EXPORT_SUBDIR}/{filename}",
288
+ run_root / TRIAL_EXPORT_SUBDIR,
289
+ require_existing_file=False,
290
+ )
291
+ # The generic resolver also folds "." segments; a name like "a/." or
292
+ # "a.." is a file name here, but a name that Path normalizes to
293
+ # something other than itself inside the subtree must not pass.
294
+ if result.ok and result.path is not None and result.path.name != filename:
295
+ return ContainmentResult(None, REASON_RESERVED_NAME)
296
+ return result
@@ -1,8 +1,7 @@
1
1
  """Static-handoff disclosure review builder (Stage 9 evidence).
2
2
 
3
3
  Produces the ``disclosure-review.json`` delivery credential defined by the
4
- Static Handoff spec (docs/specs/2026-08-22-interactive-review-and-static-handoff
5
- -implementation-plan.md §4.2): a single authoritative payload binding the run
4
+ Static Handoff contract (ADR-0034): a single authoritative payload binding the run
6
5
  identity, verdict, profile, decision authority, the five standard viewport
7
6
  layout metrics (``sw`` / ``innerH`` / ``hOverflow`` / ``disclosure.inFold``),
8
7
  and the G1–G8 gate count — so a front-end/QA consumer can reproduce and audit
@@ -16,7 +15,7 @@ Two seams keep the contract builder pure and testable without a browser:
16
15
  The probe JS string is also exposed (``LAYOUT_PROBE_JS``) for a static
17
16
  syntax/structure check.
18
17
  * ``build_disclosure(...)`` — deterministic pure builder: no I/O, no browser.
19
- It only normalizes caller-supplied facts into the §4.2 shape.
18
+ It only normalizes caller-supplied facts into the disclosure shape.
20
19
 
21
20
  ``build_handoff_zip()`` packages the disclosure credential plus any caller-
22
21
  supplied snapshot artifacts into a single ZIP. The Evidence-side builder
@@ -25,6 +25,7 @@ import json
25
25
  import re
26
26
  import sys
27
27
  from dataclasses import dataclass
28
+ from pathlib import Path
28
29
  from typing import Any
29
30
 
30
31
  try:
@@ -37,8 +38,10 @@ try:
37
38
  parse_capture_contract,
38
39
  )
39
40
  from design_playbook.mcp.evidence.path_syntax import ( # noqa: E402
41
+ TRACE_SUFFIX,
40
42
  lexical_posix_key,
41
43
  probe_sidecar_rel,
44
+ trace_artifact_error,
42
45
  trimmed_relpath,
43
46
  )
44
47
  except ImportError: # standalone execution: same-dir seam (rules_registry pattern)
@@ -52,8 +55,10 @@ except ImportError: # standalone execution: same-dir seam (rules_registry patte
52
55
  )
53
56
  from capture_contract import parse_capture_contract # noqa: E402
54
57
  from path_syntax import ( # noqa: E402
58
+ TRACE_SUFFIX,
55
59
  lexical_posix_key,
56
60
  probe_sidecar_rel,
61
+ trace_artifact_error,
57
62
  trimmed_relpath,
58
63
  )
59
64
 
@@ -133,6 +138,14 @@ def preflight_entry(request: object, entry: int) -> list[PreflightFact]:
133
138
  expected="relative path starting with "
134
139
  f"{ARTIFACT_PREFIX!r}",
135
140
  actual=artifact))
141
+ elif isinstance(capture_type, str) and (
142
+ name_error := trace_artifact_error(capture_type, artifact)
143
+ ):
144
+ # Same rule the Provider rejects with (path_syntax), reported here
145
+ # before a browser starts (DEF-6).
146
+ facts.append(_error("bad_artifact_extension", name_error, entry,
147
+ expected=f"name ending in {TRACE_SUFFIX}",
148
+ actual=artifact))
136
149
 
137
150
  actions = request.get("actions")
138
151
  if actions is not None:
@@ -291,8 +304,15 @@ def preflight_plan(plan: object) -> list[PreflightFact]:
291
304
 
292
305
 
293
306
  def main(argv: list[str] | None = None) -> int:
294
- if sys.stdout and hasattr(sys.stdout, "reconfigure"):
295
- sys.stdout.reconfigure(encoding="utf-8") # Windows GBK consoles
307
+ # One pipe-encoding seam (T-105): UTF-8 on piped stdout/stderr
308
+ # regardless of the host code page. See scripts/stdio_encoding.py.
309
+ for _candidate in Path(__file__).resolve().parents:
310
+ if (_candidate / "design_playbook.py").is_file():
311
+ sys.path.insert(0, str(_candidate))
312
+ break
313
+ from design_playbook.scripts.stdio_encoding import configure_piped_utf8
314
+
315
+ configure_piped_utf8()
296
316
  args = sys.argv[1:] if argv is None else argv
297
317
  md = "--md" in args
298
318
  paths = [a for a in args if a != "--md"]