design-playbook 0.20.0 → 0.20.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
  {"event": "answered", "question_id": "Q1", "answer": "运行状态跨页可查、全局控制可用、失败可批量治理", "ts": "2026-08-14T09:02:00Z"}
3
3
  {"event": "asked", "question_id": "Q9", "batch": 2, "tier": "T3", "text": "顶栏计数区升级为全局运行控制台——视觉方向(构成级)?", "impact": "D3 decision-report(成形只登记路由,D3 裁决)", "ts": "2026-08-14T09:03:00Z"}
4
4
  {"event": "assumption_staged", "field": "sim.control_scope", "tier": "T2", "reason": "全局暂停的作用范围未答", "risk": "误伤手动单次重试", "fallback": "全局暂停/恢复不作用于手动单次重试", "ts": "2026-08-14T09:04:00Z"}
5
- {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l1.goal 修订(supersedes D-0001)", "l6.c4-c6 表述"], "ts": "2026-08-14T09:08:00Z"}
5
+ {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": [{"field": "l1.goal", "value": "修订(supersedes D-0001)"}, "l6.c4", "l6.c5", "l6.c6"], "ts": "2026-08-14T09:08:00Z"}
6
6
  {"event": "confirm_presented", "batch": "CP-C", "kind": "assumption", "items": ["sim.control_scope"], "ts": "2026-08-14T09:08:00Z"}
7
7
  {"event": "item_confirmed", "batch": "CP-A", "field": "l1.goal", "ts": "2026-08-14T09:12:00Z"}
8
8
  {"event": "item_confirmed", "batch": "CP-A", "field": "l6.c4", "ts": "2026-08-14T09:12:00Z"}
@@ -8,8 +8,8 @@
8
8
  {"event": "assumption_staged", "field": "export.row_cap", "tier": "T1", "reason": "规模上限未答(队列满,转显式风险确认)", "risk": "上限过低阻断真实导出或过高拖垮同步窗口", "fallback": "50000 行", "ts": "2026-08-14T09:34:00Z"}
9
9
  {"event": "assumption_staged", "field": "export.sync_window", "tier": "T2", "reason": "同步导出限时未逐字确认", "risk": "超时体验未定义", "fallback": "60 秒", "ts": "2026-08-14T09:34:00Z"}
10
10
  {"event": "assumption_staged", "field": "export.column_scope", "tier": "T4", "reason": "文件命名与列范围为局部实现选择", "risk": "隐藏列误出", "fallback": "当前视图列(不含隐藏列)", "ts": "2026-08-14T09:34:00Z"}
11
- {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l1.goal", "l1.target_user", "l6.c1-c3 表述", "l1.non_goals"], "ts": "2026-08-14T09:40:00Z"}
12
- {"event": "confirm_presented", "batch": "CP-B", "kind": "structural", "items": ["入口 IA 两案:A 全局工具栏 / B 主列表行内批量"], "ts": "2026-08-14T09:40:00Z"}
11
+ {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l1.goal", "l1.target_user", "l6.c1", "l6.c2", "l6.c3", "l1.non_goals"], "ts": "2026-08-14T09:40:00Z"}
12
+ {"event": "confirm_presented", "batch": "CP-B", "kind": "structural", "items": [{"field": "l2.entry_choice", "value": "入口 IA 两案:A 全局工具栏 / B 主列表行内批量"}], "ts": "2026-08-14T09:40:00Z"}
13
13
  {"event": "confirm_presented", "batch": "CP-C", "kind": "assumption", "items": ["export.row_cap", "export.sync_window", "export.column_scope", "l1.scenes"], "ts": "2026-08-14T09:40:00Z"}
14
14
  {"event": "item_confirmed", "batch": "CP-A", "field": "l1.goal", "ts": "2026-08-14T09:52:00Z"}
15
15
  {"event": "item_confirmed", "batch": "CP-A", "field": "l1.target_user", "ts": "2026-08-14T09:52:00Z"}
@@ -1,6 +1,6 @@
1
1
  {"event": "asked", "question_id": "Q1", "batch": 1, "tier": "T1", "text": "「只导出选中列」的完成判据如何表述?", "impact": "l6.c4", "ts": "2026-08-14T12:41:00Z"}
2
2
  {"event": "answered", "question_id": "Q1", "answer": "Given 运营在主列表圈选 2 列 When 触发行内导出 Then CSV 仅含圈选列且顺序与列表一致(证据:交互记录)", "ts": "2026-08-14T12:42:00Z"}
3
- {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l6.c4 表述"], "ts": "2026-08-14T12:43:00Z"}
3
+ {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l6.c4"], "ts": "2026-08-14T12:43:00Z"}
4
4
  {"event": "item_confirmed", "batch": "CP-A", "field": "l6.c4", "ts": "2026-08-14T12:44:00Z"}
5
5
  {"event": "projected", "ts": "2026-08-14T12:45:00Z", "order": ["promote_fields", "append_decision", "apply_decisions", "spec", "bind_first"], "decisions": ["D-0101"], "assumed": [], "mappings": [{"decision": "D-0101", "field": "l6.c4", "spec_section": "L6"}]}
6
6
  {"event": "archived", "ts": "2026-08-14T12:46:00Z"}
@@ -2,7 +2,7 @@
2
2
  {"event": "answered", "question_id": "Q1", "answer": "任务全程可见、跨页保持、结果可获知", "ts": "2026-08-14T10:44:00Z"}
3
3
  {"event": "asked", "question_id": "Q9", "batch": 2, "tier": "T3", "text": "导出进行中的状态呈现——视觉方向(构成级)?", "impact": "D3 decision-report(成形只登记路由,D3 裁决)", "ts": "2026-08-14T10:46:00Z"}
4
4
  {"event": "assumption_staged", "field": "export.task_persist_ttl", "tier": "T2", "reason": "状态条目保留时长未答", "risk": "条目堆积或过早消失", "fallback": "run 内持久(≤10 条)", "ts": "2026-08-14T10:47:00Z"}
5
- {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": ["l1.goal 修订(supersedes D-0001)", "l6.c4-c6 表述"], "ts": "2026-08-14T10:50:00Z"}
5
+ {"event": "confirm_presented", "batch": "CP-A", "kind": "intent", "items": [{"field": "l1.goal", "value": "修订(supersedes D-0001)"}, "l6.c4", "l6.c5", "l6.c6"], "ts": "2026-08-14T10:50:00Z"}
6
6
  {"event": "confirm_presented", "batch": "CP-C", "kind": "assumption", "items": ["export.task_persist_ttl"], "ts": "2026-08-14T10:50:00Z"}
7
7
  {"event": "item_confirmed", "batch": "CP-A", "field": "l1.goal", "ts": "2026-08-14T10:55:00Z"}
8
8
  {"event": "item_confirmed", "batch": "CP-A", "field": "l6.c4", "ts": "2026-08-14T10:55:00Z"}
@@ -256,13 +256,12 @@ def _action_wait_for_state(page: Any, action: dict, index: int, do: str) -> None
256
256
  if not isinstance(state, str) or not state:
257
257
  raise ValueError(f"actions[{index}].state required for wait_for_state")
258
258
  selector = action.get("selector")
259
- if isinstance(selector, str) and selector:
260
- page.wait_for_selector(selector, timeout=10_000)
261
- else:
262
- page.wait_for_selector(
263
- f'[data-state="{state}"]',
264
- timeout=10_000,
265
- )
259
+ target = (
260
+ f'{selector}[data-state="{state}"]'
261
+ if isinstance(selector, str) and selector
262
+ else f'[data-state="{state}"]'
263
+ )
264
+ page.wait_for_selector(target, timeout=10_000)
266
265
 
267
266
 
268
267
  def _action_wait(page: Any, action: dict, index: int, do: str) -> None:
@@ -121,9 +121,37 @@ def _structured(call_response: dict) -> dict:
121
121
  return json.loads(text)
122
122
 
123
123
 
124
+ class _RecordingPage:
125
+ def __init__(self) -> None:
126
+ self.waits: list[tuple[str, int]] = []
127
+
128
+ def wait_for_selector(self, selector: str, *, timeout: int) -> None:
129
+ self.waits.append((selector, timeout))
130
+
131
+
124
132
  class EvidencePurePathTests(unittest.TestCase):
125
133
  """No-chromium transport and direct runtime-interface tests."""
126
134
 
135
+ def test_wait_for_state_combines_selector_and_state(self) -> None:
136
+ page = _RecordingPage()
137
+ capture_runtime._action_wait_for_state(
138
+ page,
139
+ {"do": "wait_for_state", "selector": "body", "state": "ready"},
140
+ 0,
141
+ "wait_for_state",
142
+ )
143
+ self.assertEqual(page.waits, [('body[data-state="ready"]', 10_000)])
144
+
145
+ def test_wait_for_state_without_selector_targets_any_matching_state(self) -> None:
146
+ page = _RecordingPage()
147
+ capture_runtime._action_wait_for_state(
148
+ page,
149
+ {"do": "wait_for_state", "state": "ready"},
150
+ 0,
151
+ "wait_for_state",
152
+ )
153
+ self.assertEqual(page.waits, [('[data-state="ready"]', 10_000)])
154
+
127
155
  def test_parse_capture_contract_requires_schema_and_viewport(self) -> None:
128
156
  from design_playbook.mcp.evidence import capture_contract
129
157
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "design-playbook",
3
- "version": "0.20.0",
3
+ "version": "0.20.1",
4
4
  "description": "Design I/O for coding agents: controllable UI generation via declarations (spec/domain/craft/design/components/template) and contracts (skill/evaluator). Use for product UI—console, dashboard, agent-ops, CJK-first apps.",
5
5
  "keywords": [
6
6
  "pi-package",
@@ -43,7 +43,20 @@ stay protocol-side. Gate policy lives in ``g10_design_decisions.py``.
43
43
 
44
44
  Flow-map values inside ``- {k: v, ...}`` items may not contain ASCII commas
45
45
  or braces (use full-width punctuation in prose values); this is the declared
46
- shape, same contract style as the seven-column audit rows.
46
+ shape, same contract style as the seven-column audit rows. Items may fold
47
+ across lines (issue #44): a ``- {`` item that does not close its brace on
48
+ the marker line continues on the following lines until the braces balance —
49
+ single-line items parse exactly as before. Folds break at commas (the
50
+ schema example's shape): a break that does not end the accumulated text
51
+ with a comma (or the opening brace) would merge the next key into the
52
+ previous value, so the join records a :class:`FoldIssue` and G10 reports it
53
+ (:code:`G10.fold_break_not_comma`); a fold that never balances by the end
54
+ of the block records :code:`G10.fold_unterminated` (fail-closed, with the
55
+ remaining shape errors still firing).
56
+
57
+ ``dd:`` is the R3 challenge channel, never an observation link: values
58
+ carried by positive (S0) findings are excluded from the challenge face and
59
+ reported as structural errors by G10 (issue #44).
47
60
  """
48
61
  from __future__ import annotations
49
62
 
@@ -51,6 +64,8 @@ import re
51
64
  from dataclasses import dataclass, field
52
65
  from typing import Any
53
66
 
67
+ from design_playbook.scripts.g2_g4_pointback import _findings
68
+
54
69
  # --- closed enums (design-prototype 4.1 machine face) -----------------------
55
70
 
56
71
  DD_TIERS = frozenset({"record", "compare", "explore"})
@@ -120,6 +135,7 @@ class DDEntry:
120
135
  supersedes: str = ""
121
136
  stale: str = ""
122
137
  stale_review: dict[str, str] = field(default_factory=dict)
138
+ fold_issues: tuple["FoldIssue", ...] = ()
123
139
  block: str = ""
124
140
 
125
141
  # -- convenience accessors ------------------------------------------
@@ -168,6 +184,23 @@ class DDEntry:
168
184
  return [item.get("id", "") for item in self.candidates]
169
185
 
170
186
 
187
+ @dataclass(frozen=True)
188
+ class FoldIssue:
189
+ """A fold defect found while joining folded flow-map items.
190
+
191
+ ``kind`` is ``"break_not_comma"`` (the accumulated fold text did not end
192
+ with a comma — or the opening brace — when a continuation was appended,
193
+ so the next key merges into the previous value) or ``"unterminated"``
194
+ (the braces never balanced before the block ended). ``line`` is the
195
+ 1-based line of the fold-opening marker inside the entry block;
196
+ ``tail`` carries the offending text tail for the error face.
197
+ """
198
+
199
+ kind: str
200
+ line: int
201
+ tail: str = ""
202
+
203
+
171
204
  @dataclass(frozen=True)
172
205
  class ESignals:
173
206
  """Machine-judgeable E-criterion signals gathered from run artifacts."""
@@ -208,6 +241,55 @@ def _flow_map(item: str) -> dict[str, str]:
208
241
  return {"value": _scalar(item)}
209
242
 
210
243
 
244
+ def _join_folded_flow_maps(
245
+ lines: list[str]) -> tuple[list[str], tuple[FoldIssue, ...]]:
246
+ """Join folded ``- {...}`` flow-map items onto their marker line.
247
+
248
+ Issue #44: an item that opens a ``{`` without closing it on the marker
249
+ line continues on the following lines (the fold the entry schema
250
+ example already shows) until its braces balance; continuation text is
251
+ appended to the marker line so the rest of the parser sees one logical
252
+ line. Values never contain ASCII commas or braces (declared shape), so
253
+ brace balance is an unambiguous fold terminator. Folds break at commas:
254
+ a continuation appended to accumulated text that does not end with a
255
+ comma (or the opening brace) would silently merge the next key into the
256
+ previous value, so the break is recorded as a :class:`FoldIssue`
257
+ (fail-closed; G10 reports it). An unterminated fold stays joined and
258
+ records its own issue plus the downstream shape checks (fail-closed).
259
+ Returns ``(joined lines, fold issues)``.
260
+ """
261
+ out: list[str] = []
262
+ issues: list[FoldIssue] = []
263
+ fold: str | None = None
264
+ fold_line = 0
265
+ for lineno, raw_line in enumerate(lines, 1):
266
+ stripped = raw_line.strip()
267
+ if fold is not None:
268
+ if not fold.rstrip().endswith((",", "{")):
269
+ issues.append(FoldIssue(
270
+ kind="break_not_comma",
271
+ line=fold_line,
272
+ tail=fold.strip()[-60:],
273
+ ))
274
+ fold = fold.rstrip() + " " + stripped
275
+ if fold.count("{") <= fold.count("}"):
276
+ out.append(fold)
277
+ fold = None
278
+ continue
279
+ if (
280
+ stripped.startswith("- ")
281
+ and stripped.count("{") > stripped.count("}")
282
+ ):
283
+ fold = raw_line.rstrip()
284
+ fold_line = lineno
285
+ continue
286
+ out.append(raw_line)
287
+ if fold is not None:
288
+ issues.append(FoldIssue(kind="unterminated", line=fold_line))
289
+ out.append(fold)
290
+ return out, tuple(issues)
291
+
292
+
211
293
  def _entry_blocks(text: str) -> list[tuple[str, str]]:
212
294
  """Split the report into (id, body) blocks by DD entry heading."""
213
295
  matches = list(DD_HEADING.finditer(text))
@@ -231,7 +313,8 @@ def _parse_entry(entry_id: str, body: str) -> DDEntry:
231
313
  current_section: str | None = None
232
314
  current_list_key: str | None = None
233
315
 
234
- for raw_line in block.splitlines():
316
+ joined, fold_issues = _join_folded_flow_maps(block.splitlines())
317
+ for raw_line in joined:
235
318
  stripped = raw_line.strip()
236
319
  if not stripped or stripped.startswith("```"):
237
320
  continue
@@ -307,6 +390,7 @@ def _parse_entry(entry_id: str, body: str) -> DDEntry:
307
390
  supersedes=fields.get("supersedes", ""),
308
391
  stale=fields.get("stale", ""),
309
392
  stale_review=_scalars("stale_review"),
393
+ fold_issues=fold_issues,
310
394
  block=block,
311
395
  )
312
396
 
@@ -319,20 +403,74 @@ def parse_dd_entries(text: str) -> list[DDEntry]:
319
403
  ]
320
404
 
321
405
 
406
+ # ``none`` value token for the verbatim top block: a trailing same-line
407
+ # note after the token is tolerated commentary, never a declared change
408
+ # (issue #44). The token must end at whitespace or punctuation so values
409
+ # like ``nonempty`` and ``none_x`` stay fail-closed non-none (underscore
410
+ # joins the lookalike continuation class with ``-``, e.g. ``none-such``).
411
+ NONE_VALUE = re.compile(r"none(?=$|[^0-9A-Za-z_-])", re.I)
412
+
413
+
322
414
  def top_block_baseline_change(text: str) -> bool:
323
- """True when the verbatim top block declares ``baseline-changes != none``."""
415
+ """True when the verbatim top block declares ``baseline-changes != none``.
416
+
417
+ ``none`` may carry a trailing same-line note (anything after the value
418
+ token); notes never turn ``none`` into a declared change. Substantive
419
+ commentary belongs on its own line.
420
+ """
324
421
  match = re.search(r"^baseline-changes:[ \t]*(\S.*)$", text, re.M)
325
422
  if match is None:
326
423
  return False
327
- return match.group(1).strip().casefold() != "none"
424
+ return NONE_VALUE.match(match.group(1).strip()) is None
425
+
426
+
427
+ def is_positive_finding(parsed: dict[str, list[str]]) -> bool:
428
+ """True when a parsed point-back finding sits on the S0 (info) axis."""
429
+ values = parsed.get("severity") or [""]
430
+ return values[0].strip().casefold() == "s0"
431
+
432
+
433
+ def positive_dd_refs(
434
+ text: str) -> tuple[tuple[int, tuple[str, ...]], ...]:
435
+ """``(finding index, dd refs)`` for every positive finding carrying ``dd:``.
436
+
437
+ Issue #44: ``dd:`` is the R3 challenge channel and never rides a
438
+ positive observation. These are structural errors G10 reports
439
+ (fail-closed) instead of silently reading them as challenges.
440
+ """
441
+ out: list[tuple[int, tuple[str, ...]]] = []
442
+ for index, parsed in enumerate(_findings(text), 1):
443
+ refs = tuple(
444
+ value.strip().rstrip(",") for value in parsed.get("dd", [])
445
+ if value.strip())
446
+ if refs and is_positive_finding(parsed):
447
+ out.append((index, refs))
448
+ return tuple(out)
449
+
450
+
451
+ def _positive_dd_block(block: str) -> bool:
452
+ return any(
453
+ parsed.get("dd") and is_positive_finding(parsed)
454
+ for parsed in _findings(block)
455
+ )
328
456
 
329
457
 
330
458
  def dd_refs_in_pointback(text: str) -> tuple[str, ...]:
331
- """Collect ``dd:`` targets from finding field lines / invalidated rows."""
459
+ """Collect ``dd:`` challenge targets from finding field lines.
460
+
461
+ Issue #44: ``dd:`` values carried by positive (S0) findings record
462
+ observation links, not challenges — their paragraphs are skipped so a
463
+ positive observation can never fire a false re-entry / E3 signal.
464
+ Paragraphs that are not findings (e.g. a bare ``dd:`` line) keep the
465
+ legacy raw face.
466
+ """
332
467
  targets: list[str] = []
333
- for match in re.finditer(r"^dd:[ \t]*(\S+)", text, re.I | re.M):
334
- ref = match.group(1).strip().rstrip(",")
335
- targets.append(ref)
468
+ for block in re.split(r"\n\s*\n", text):
469
+ if _positive_dd_block(block):
470
+ continue
471
+ for match in re.finditer(r"^dd:[ \t]*(\S+)", block, re.I | re.M):
472
+ ref = match.group(1).strip().rstrip(",")
473
+ targets.append(ref)
336
474
  return tuple(targets)
337
475
 
338
476
 
@@ -41,6 +41,7 @@ import re
41
41
  from dataclasses import dataclass
42
42
 
43
43
  from design_playbook.scripts._diagnostics import Finding, finding
44
+ from design_playbook.scripts.dd_entries import is_positive_finding
44
45
  from design_playbook.scripts.g2_g4_pointback import _findings
45
46
  from design_playbook.scripts.repair_rounds import is_blocking
46
47
 
@@ -167,9 +168,16 @@ def check_routes(text: str) -> list[Finding]:
167
168
 
168
169
 
169
170
  def dd_targets(text: str) -> tuple[str, ...]:
170
- """dd: references carried by findings (R3 challenge face)."""
171
+ """dd: references carried by findings (R3 challenge face).
172
+
173
+ Issue #44: positive (S0) findings carry observation links, not
174
+ challenges — their ``dd:`` values never fire E3 (G10 reports the
175
+ misuse as a structural error instead).
176
+ """
171
177
  targets: list[str] = []
172
178
  for parsed in _findings(text):
179
+ if is_positive_finding(parsed):
180
+ continue
173
181
  targets.extend(value.strip() for value in parsed.get("dd", [])
174
182
  if value.strip())
175
183
  return tuple(targets)
@@ -6,8 +6,9 @@ declared by the protocol — entry completeness, tier/status enums, tier
6
6
  recording obligations (R one-line rationale / C trade-off record / E user
7
7
  confirmation), supersedes existence + acyclicity, registry rule-reference
8
8
  cross-check, preview transaction linkage (decision_id), R3 re-entry
9
- resolution (dd: challenges must end invalidated with an E-tier revision),
10
- and the baseline-drift stale review (three exits: keep / revise / escalate).
9
+ resolution (dd: challenges must end invalidated with an E-tier revision;
10
+ ``dd:`` on a positive finding is a shape error, issue #44), and the
11
+ baseline-drift stale review (three exits: keep / revise / escalate).
11
12
 
12
13
  Comparison-matrix quality, trade-off sufficiency, and tier-grading
13
14
  judgement calls (composition change, identity drift beyond declared
@@ -35,6 +36,7 @@ from design_playbook.scripts.dd_entries import (
35
36
  is_cross_run_ref,
36
37
  local_dd_id,
37
38
  parse_dd_entries,
39
+ positive_dd_refs,
38
40
  )
39
41
 
40
42
  # rules.md ships inside the package (read-only protocol consumption, the
@@ -102,6 +104,11 @@ def _entry_checks(entries: list[DDEntry]) -> list[Finding]:
102
104
  ))
103
105
  seen[label] = index
104
106
 
107
+ # fold defects first (issue #44 follow-up): a named unterminated /
108
+ # comma-less fold error outranks the indirect missing_* findings it
109
+ # causes downstream, so the error face points at the real defect.
110
+ errs += _fold_checks(entry)
111
+
105
112
  for key in ("id", "tier", "question", "status"):
106
113
  if not entry.fields.get(key, "").strip():
107
114
  errs.append(finding(
@@ -143,6 +150,45 @@ def _entry_checks(entries: list[DDEntry]) -> list[Finding]:
143
150
  return errs
144
151
 
145
152
 
153
+ def _fold_checks(entry: DDEntry) -> list[Finding]:
154
+ """Fold defects on ``- {…}`` items (issue #44 follow-up, fail-closed).
155
+
156
+ A fold break without a comma merges the next key into the previous
157
+ value (the parse alone would accept it silently); a fold that never
158
+ balances swallows the rest of the block and only indirect missing_*
159
+ errors would fire. Both get a named error up front, with the block line
160
+ of the fold-opening marker; the remaining shape errors still fire.
161
+ """
162
+ errs: list[Finding] = []
163
+ for issue in entry.fold_issues:
164
+ if issue.kind == "unterminated":
165
+ errs.append(finding(
166
+ "G10.fold_unterminated",
167
+ f"G10 decisions: {entry.id} opens a folded flow-map item at "
168
+ f"block line {issue.line} that never closes its brace — the "
169
+ "fold swallowed the rest of the entry block",
170
+ owner=_fmt(entry.id),
171
+ expected="braces balance inside the entry block",
172
+ actual=f"unterminated fold from block line {issue.line}",
173
+ repair="Close the brace, or unfold to the canonical "
174
+ "single-line item",
175
+ ))
176
+ else:
177
+ errs.append(finding(
178
+ "G10.fold_break_not_comma",
179
+ f"G10 decisions: {entry.id} folds a flow-map item at block "
180
+ f"line {issue.line} without a comma at the break — the next "
181
+ "key merges into the previous value",
182
+ owner=_fmt(entry.id),
183
+ expected="fold breaks end with a comma (or the opening "
184
+ "brace)",
185
+ actual=f"break after {issue.tail!r}",
186
+ repair="End the folded line with a comma before continuing "
187
+ "the item on the next line",
188
+ ))
189
+ return errs
190
+
191
+
146
192
  def _candidate_checks(entry: DDEntry) -> list[Finding]:
147
193
  errs: list[Finding] = []
148
194
  label = entry.id
@@ -669,6 +715,31 @@ def _reentry_checks(
669
715
  return errs
670
716
 
671
717
 
718
+ def _positive_dd_checks(pointback_text: str | None) -> list[Finding]:
719
+ """``dd:`` on a positive (S0) finding is a shape error (issue #44).
720
+
721
+ ``dd:`` is the R3 challenge channel; riding it on a positive
722
+ observation reads as a challenge downstream. Fail closed with a
723
+ precise error instead of silently ignoring the line.
724
+ """
725
+ if not pointback_text:
726
+ return []
727
+ errs: list[Finding] = []
728
+ for index, refs in positive_dd_refs(pointback_text):
729
+ errs.append(finding(
730
+ "G10.dd_on_positive_finding",
731
+ f"G10 decisions: positive finding {index} carries "
732
+ f"dd: {', '.join(refs)} — dd: is the R3 challenge channel and "
733
+ "never rides a positive observation",
734
+ owner=f"point-back.md#finding.{index}",
735
+ expected="dd: only on non-positive (S1-S3) findings",
736
+ actual="dd: on severity S0",
737
+ repair="Drop the dd: line and record the observation link as "
738
+ "prose (e.g. an evidence note line)",
739
+ ))
740
+ return errs
741
+
742
+
672
743
  def _preview_link_checks(
673
744
  entries: list[DDEntry],
674
745
  preview_dir: Path | None) -> list[Finding]:
@@ -929,6 +1000,7 @@ def check_g10(
929
1000
  report_text=report_text,
930
1001
  )
931
1002
  errs += _reentry_checks(entries, signals.dd_targets)
1003
+ errs += _positive_dd_checks(pointback_text)
932
1004
  errs += _preview_link_checks(entries, preview_dir)
933
1005
  errs += _stale_checks(entries, baseline_state)
934
1006
  errs += _signal_checks(entries, signals, run_profile_tier)
@@ -36,9 +36,9 @@ from design_playbook.scripts.g2_g4_pointback import FIELD_LINE
36
36
  MIN_DISTINCT_RUNS = 3
37
37
  MIN_DISTINCT_CONTEXTS = 2
38
38
  MAX_UNEXPLAINED_FALSE_POSITIVES = 0
39
- # Positive observations (S0) are not defect signals the derivation
40
- # never lets them enter the candidate queue.
41
- EXCLUDED_SEVERITIES = frozenset({"S0"})
39
+ # Candidate derivation fails closed: only defect severities enter history.
40
+ # S0 is a positive observation; blank, legacy, and unknown values are invalid.
41
+ CANDIDATE_SEVERITIES = frozenset({"S3", "S2", "S1"})
42
42
 
43
43
  UNSPECIFIED_CONTEXT = "(unspecified)"
44
44
 
@@ -134,8 +134,8 @@ def derive_candidates(
134
134
  """
135
135
  groups: dict[str, list[Occurrence]] = {}
136
136
  for occurrence in occurrences:
137
- if occurrence.severity.strip() in EXCLUDED_SEVERITIES:
138
- continue # positive observations are not defect signals
137
+ if occurrence.severity.strip() not in CANDIDATE_SEVERITIES:
138
+ continue
139
139
  key = normalize(occurrence.issue)
140
140
  if key:
141
141
  groups.setdefault(key, []).append(occurrence)
@@ -28,6 +28,7 @@ from dataclasses import dataclass, field
28
28
  RUN_PROFILE_MARKER = re.compile(r"<!--\s*run-profile(?::\s*v(\d+))?\s*-->")
29
29
  FENCED_BLOCK = re.compile(r"```[a-zA-Z]*\n(.*?)```", re.S)
30
30
  TIERS = frozenset({"P1", "P2", "P3"})
31
+ SUPPORTED_RUN_PROFILE_VERSIONS = frozenset({1})
31
32
 
32
33
 
33
34
  @dataclass(frozen=True)
@@ -105,6 +106,10 @@ def validate_run_profile(profile: RunProfile | None) -> list[str]:
105
106
  "skipping the rest of the plan body is legal, skipping the "
106
107
  "profile block is not)"]
107
108
  errors: list[str] = []
109
+ if profile.version not in SUPPORTED_RUN_PROFILE_VERSIONS:
110
+ errors.append(
111
+ f"run-profile version v{profile.version} is unsupported; only v1 is accepted"
112
+ )
108
113
  if profile.tier not in TIERS:
109
114
  errors.append(
110
115
  f"run-profile tier {profile.tier!r} not in P1|P2|P3 "
@@ -2,12 +2,16 @@
2
2
  """Derive Design I/O run status / resume hints from existing artifacts.
3
3
 
4
4
  Does **not** create a second run-state SSOT. Reads only files agents already
5
- write under a run root (default: discover newest ``.scratch/*/``).
5
+ write under a run root (default: discover newest ``.scratch/*/``). Fill
6
+ surfaces may live outside the run root: when ``plan.md`` registers them with
7
+ ``fill:`` field lines, the fill stage is also judged on those declared paths
8
+ (issue #44; the stage registry itself is unchanged).
6
9
  """
7
10
  from __future__ import annotations
8
11
 
9
12
  import argparse
10
13
  import json
14
+ import re
11
15
  import sys
12
16
  from dataclasses import dataclass
13
17
  from pathlib import Path
@@ -202,16 +206,63 @@ class StageState:
202
206
  evidence: list[str]
203
207
 
204
208
 
209
+ # Issue #44: fill surfaces may land in the host tree (product side) instead
210
+ # of the run root. When plan.md registers those paths as ``fill:`` field
211
+ # lines, the fill stage is also judged on their existence — one path per
212
+ # line, run-root-relative or host-project-relative (the orchestrating cwd).
213
+ # Only unfenced column-0 field lines are declarations; fenced blocks are
214
+ # prose/examples and are never read as declarations.
215
+ PLAN_FILL_LINE = re.compile(r"^fill:[ \t]*(\S+)")
216
+
217
+
218
+ def _plan_fill_artifacts(run_root: Path) -> list[str]:
219
+ """Declared fill artifact paths from plan.md that exist on disk.
220
+
221
+ Fenced code blocks (```` ``` ````) are skipped while scanning: an
222
+ example or prose block citing ``fill: spec.md`` is not a declaration
223
+ (fail-closed — the fill stage stays unchecked rather than counting a
224
+ narrated example).
225
+ """
226
+ try:
227
+ text = (run_root / "plan.md").read_text(encoding="utf-8")
228
+ except (OSError, UnicodeError):
229
+ return []
230
+ found: list[str] = []
231
+ fenced = False
232
+ for line in text.splitlines():
233
+ if line.lstrip().startswith("```"):
234
+ fenced = not fenced
235
+ continue
236
+ if fenced:
237
+ continue
238
+ match = PLAN_FILL_LINE.match(line)
239
+ if match is None:
240
+ continue
241
+ declared = match.group(1).strip().rstrip(",")
242
+ candidate = Path(declared)
243
+ bases = (
244
+ [candidate] if candidate.is_absolute()
245
+ else [run_root / candidate, Path.cwd() / candidate]
246
+ )
247
+ if any(base.is_file() for base in bases) and declared not in found:
248
+ found.append(declared)
249
+ return found
250
+
251
+
205
252
  def inspect_run(
206
253
  run_root: Path, preview_snapshot: PreviewSnapshot | None = None,
207
254
  run_facts: RunFacts | None = None,
208
255
  ) -> list[StageState]:
209
256
  facts = run_facts or capture_run_facts(run_root=run_root)
210
257
  snapshot = preview_snapshot or facts.preview or inspect_preview(run_root / "preview")
258
+ plan_fills = _plan_fill_artifacts(run_root)
211
259
  states: list[StageState] = []
212
260
  for stage in STAGES:
213
261
  if stage.key == "preview":
214
262
  found = [f"preview/{source}" for source in snapshot.occurrence_sources]
263
+ elif stage.key == "fill":
264
+ found = [marker for marker in stage.markers if marker in facts.existing_paths]
265
+ found += [declared for declared in plan_fills]
215
266
  else:
216
267
  found = [marker for marker in stage.markers if marker in facts.existing_paths]
217
268
  states.append(StageState(
@@ -93,21 +93,11 @@ def load_shaping_facts(run_root: Path) -> ShapingFacts | None:
93
93
  return ShapingFacts(events=tuple(events), queue=queue)
94
94
 
95
95
 
96
- def _pending_ids(events: list[dict[str, Any]]) -> set[str]:
97
- """Item ids referenced by ask/stage events without a terminal answer."""
98
- opened: set[str] = set()
99
- closed: set[str] = set()
100
- for event in events:
101
- kind = event.get("event")
102
- item = event.get("question_id") or event.get("item_id")
103
- if not isinstance(item, str) or not item:
104
- continue
105
- if kind in ("asked", "assumption_staged", "confirm_presented"):
106
- opened.add(item)
107
- elif kind in ("answered", "item_confirmed", "item_rejected",
108
- "item_revised"):
109
- closed.add(item)
110
- return opened - closed
96
+ def _item_id(item: Any) -> Any:
97
+ """Identity of a confirmation-batch item (``field`` for dict items)."""
98
+ if isinstance(item, dict):
99
+ return item.get("field")
100
+ return item
111
101
 
112
102
 
113
103
  def derive_queue(events: list[dict[str, Any]]) -> dict[str, Any]:
@@ -116,6 +106,13 @@ def derive_queue(events: list[dict[str, Any]]) -> dict[str, Any]:
116
106
  Derived view (shaping-prototype 4.1): pending questions (asked, not
117
107
  answered), staged assumptions (staged, not confirmed/rejected/revised),
118
108
  and confirmation batches (presented, not fully decided).
109
+
110
+ Batches close at item granularity: ``item_confirmed`` /
111
+ ``item_rejected`` / ``item_revised`` events settle only the presented
112
+ item whose id (dict ``field`` or the bare string) they carry, so a
113
+ partially decided batch stays listed with its undecided items and a
114
+ batch leaves ``open_confirmations`` only once every presented item has
115
+ a terminal decision.
119
116
  """
120
117
  pending: list[dict[str, Any]] = []
121
118
  staged: list[dict[str, Any]] = []
@@ -154,11 +151,23 @@ def derive_queue(events: list[dict[str, Any]]) -> dict[str, Any]:
154
151
  item for item in staged
155
152
  if item.get("field") != field_path
156
153
  ]
157
- if isinstance(event.get("batch"), str):
158
- confirmations = [
159
- batch for batch in confirmations
160
- if batch.get("batch") != event.get("batch")
161
- ]
154
+ if (
155
+ isinstance(event.get("batch"), str)
156
+ and isinstance(field_path, str) and field_path
157
+ ):
158
+ batch_name = event.get("batch")
159
+ settled: list[dict[str, Any]] = []
160
+ for batch in confirmations:
161
+ if batch.get("batch") != batch_name:
162
+ settled.append(batch)
163
+ continue
164
+ remaining = [
165
+ item for item in batch.get("items", [])
166
+ if _item_id(item) != field_path
167
+ ]
168
+ if remaining:
169
+ settled.append({**batch, "items": remaining})
170
+ confirmations = settled
162
171
  return {
163
172
  "derived_from": "shaping-log.jsonl",
164
173
  "pending_questions": pending,
@@ -51,6 +51,8 @@ For implemented UI, evaluate the applicability predicates of the registry entrie
51
51
 
52
52
  `Applicability` is the entry's three-state predicate outcome: `applicable`, `not-applicable`, or `blocked`. **not-applicable and blocked both require an observable reason** (blank is invalid — never a silent skip). `Result` is `clear|hit` only for applicable rows, `-` otherwise; `Positive fix` is required on hit rows. Missing rendered or source proof is `blocked`, not a silent clear. Unknown registry IDs fail at this stage. Audit rows are advisory: record evidence, exception check, and positive fix; leave declaration source, severity, and verdict to `ui-evaluator`.
53
53
 
54
+ The host model may have no vision (text-only input): renders stay **path-bound** (the `Rendered evidence` column cites the artifact path; never read the image), and assertions run on the **text face** — HTML/CSS source, `a11y tree` text, interaction-trace JSON. A no-vision run evaluates the same registry protocol this way without degrading it; genuinely missing rendered proof still records `blocked` honestly.
55
+
54
56
  | Push toward | Instead of default sludge |
55
57
  | --- | --- |
56
58
  | Accent on key noun + primary CTA | Purple–blue gradient wallpaper |
@@ -146,6 +146,8 @@ Implement structure from the decision report + `spec` + confirmed project `DESIG
146
146
 
147
147
  If a reused host component conflicts with spec L5, record the conflict and recirculate to `spec` via the authoritative map in `ui-evaluator` before choosing a minimal patch or explicit acceptance.
148
148
 
149
+ **Fill artifact location:** the Fill surface may live in the host tree (product side) instead of the run root. When it does, register the path(s) in `plan.md` as `fill: <path>` field lines (one per line; run-root-relative or host-project-relative; unfenced column-0 lines — fenced example/prose blocks are never read as declarations) — `run-status` judges the fill stage on those declared paths in addition to `filled-ui.*` in the run root. An out-of-run Fill surface with no registered path leaves the fill stage unchecked.
150
+
149
151
  Load on demand (only if the fill needs them):
150
152
 
151
153
  - domain / risk / sensitive fields → `ui-picker/references/domain.md`
@@ -47,6 +47,8 @@ Write `manifest.json` under `.scratch/<run>/reference/` using the shape in [`ref
47
47
 
48
48
  Read the sources. Fill **every required heading** from [`references/contract-template.md`](references/contract-template.md) (SSOT for section names and bullet prompts). Do not invent alternate headings.
49
49
 
50
+ The host model may have no vision (text-only input): image sources are registered by **path and metadata only** (locator, `sha256`, `captured_at` in `manifest.json`) — reading the image is never a required intake action. The observed/inferred split then rides the text the session can actually cite (user-provided notes, URL page text, file facts); visual points nobody can verify stay `inferred` or move to Unresolved questions. A no-vision host runs this skill end to end without degrading the protocol.
51
+
50
52
  Mark every claim as **observed** or **inferred**. Unlabeled claims are invalid; rewrite them before emit.
51
53
 
52
54
  **Done when:**
@@ -57,6 +57,8 @@ Evidence is captured, not judged. A manifest entry records that an artifact was
57
57
 
58
58
  For implemented UI, visible-state proof is a rendered inspection at the declared target viewport; behavior proof is an interaction trace or automated check; code-health proof is the relevant available test, type/lint, or affected build result. Planning-only proof is declaration coverage and must not claim a render or test occurred. Non-L6 declaration checks may be supporting observations or findings; they do not enter the machine ledger.
59
59
 
60
+ The host model may have no vision (text-only input). Reading a screenshot would break such a session — never make viewing an artifact a review action. Render artifacts stay bound as path references (manifest + ledger `observed`), and machine assertions judge the **text face**: HTML/CSS source, `a11y tree` text, and interaction-trace JSON. A no-vision run reviews this way end to end without degrading the protocol; note it once in the Limitations statement ("this run was reviewed on text-face evidence").
61
+
60
62
  **Done when:** every bound row was considered; every L6 criterion has exactly one non-empty `criterion / required / observed / result` row keyed as `L6.<n>`; results use only `pass|fail|blocked|N/A`; unavailable required proof is `blocked`, not skipped.
61
63
 
62
64
  ### 3. Emit point-back findings
@@ -79,7 +81,7 @@ disposition: blocking|advisory|info (severity x fact/judgment class x confiden
79
81
  evidence: <artifact path or source ref — may repeat>
80
82
  assumes: <assumed contract field paths the finding depends on, if any>
81
83
  rule: <registry ID@version refs, when a registry rule is involved>
82
- dd: <decision-report entry ref, when a design decision is involved>
84
+ dd: <decision-report entry ref, when a design decision is challenged — never on positive (S0) findings>
83
85
  ```
84
86
 
85
87
  Severity and disposition are **two axes**: a judgment-class S3 (subjective / semantic / representativeness) is never directly blocking — list it in the Limitations "pending user adjudication" sub-block with the three options (change declaration / accept risk / promote to the rule-registry queue). Only fact-class S3 (reproducible, evidence-bound) takes `disposition: blocking` and enters G4 closure.
@@ -115,7 +117,9 @@ The report artifact remains `point-back.md` (no new file). The machine face is u
115
117
  explicit unreviewed list; G11 checks existence)
116
118
  ## Limitations statement (judgment-class dimensions, no-user-evidence scope,
117
119
  pass scope, assumed dependencies, machine-face
118
- boundary, pending-user-adjudication sub-block)
120
+ boundary, text-face review note when no visual
121
+ evidence was read, pending-user-adjudication
122
+ sub-block)
119
123
  ## Verdict (exactly one Pass|Recirculate + closure lines)
120
124
  ```
121
125
 
@@ -51,6 +51,8 @@ baseline-changes: none | <explicitly approved change>
51
51
  risks: …
52
52
  ```
53
53
 
54
+ `baseline-changes: none` tolerates a trailing same-line note (the machine face reads the `none` value token); substantive commentary goes on its own line instead.
55
+
54
56
  **Done when:** the report exists, records the bound baseline or explicit waiver, and coding has not started without it.
55
57
 
56
58
  #### DD entry blocks (append after the top block)
@@ -59,7 +59,7 @@ stale: <reason + ts> # optional: baseline drift marked this entry
59
59
  stale_review: {exit: keep, note: <review line + new sha256>} # keep | revise | escalate
60
60
  ```
61
61
 
62
- Rules: ids are zero-padded `DD-####` and never repeat inside a run; flow-map values must not contain ASCII commas or braces (use full-width punctuation in prose); comparison cells carry facts and statements with source references — **numeric scores, weighted sums, and ranking points are forbidden**; incommensurable axes get an explicit trade-off line instead.
62
+ Rules: ids are zero-padded `DD-####` and never repeat inside a run; flow-map values must not contain ASCII commas or braces (use full-width punctuation in prose); a `- {…}` item may fold onto following lines while its braces stay unbalanced, breaking at a comma — the folded line must end with a comma before the continuation (the schema above shows the fold — single-line items are canonical; a comma-less break or an unterminated fold is a G10 structural error, never a silent merge); comparison cells carry facts and statements with source references — **numeric scores, weighted sums, and ranking points are forbidden**; incommensurable axes get an explicit trade-off line instead.
63
63
 
64
64
  ## Candidates and providers
65
65
 
@@ -71,6 +71,6 @@ E tier is confirmed by the user in batches (≤2 items, 2-3 candidates each, ver
71
71
 
72
72
  ## Re-entry (R3) and baseline drift
73
73
 
74
- An R3 finding (failed assumption, unrecorded trade-off, baseline conflict) names the challenged entry with an additional field line `dd: DD-0003` (same backward-compatible channel as `rule:` lines). Revision = a new entry with `supersedes: DD-0003`; the old entry is retired (`status: invalidated`) and stays parsable — history is never rewritten. The minimal invalidation set is the entry + the Fill surface consuming it + the evidence depending on its assumptions (recorded in the point-back `invalidated:` block); re-review runs only that set plus adjacent main path, inside the same run. A revision re-grades by current criteria — a challenged direction is still direction-level, so the revision lands at E tier with a fresh user confirmation (a new preview round when riding, `round_n` incremented). Baseline-conflict revisions have exactly two legal exits: pick a baseline-conforming candidate, or take an explicit `baseline-changes` approval.
74
+ An R3 finding (failed assumption, unrecorded trade-off, baseline conflict) names the challenged entry with an additional field line `dd: DD-0003` (same backward-compatible channel as `rule:` lines). The field rides only non-positive findings (S1-S3): a positive observation records its decision link as prose (e.g. an evidence note line) — `dd:` on an S0 finding is a G10 structural error, never a challenge signal. Revision = a new entry with `supersedes: DD-0003`; the old entry is retired (`status: invalidated`) and stays parsable — history is never rewritten. The minimal invalidation set is the entry + the Fill surface consuming it + the evidence depending on its assumptions (recorded in the point-back `invalidated:` block); re-review runs only that set plus adjacent main path, inside the same run. A revision re-grades by current criteria — a challenged direction is still direction-level, so the revision lands at E tier with a fresh user confirmation (a new preview round when riding, `round_n` incremented). Baseline-conflict revisions have exactly two legal exits: pick a baseline-conforming candidate, or take an explicit `baseline-changes` approval.
75
75
 
76
76
  When `design-baseline verify` detects source-hash drift, entries citing the old sha are marked `stale: <reason>` and re-reviewed under the new baseline with exactly one exit recorded in `stale_review`: **keep** (review line citing the new sha256 — clears the mark), **revise** (a superseding entry, confirmed per its tier), or **escalate** (the drift returns the question to direction level — user decision). No calendar expiry: stale is triggered by structural events only. G10 checks that drifted entries carry the mark and that every marked entry records a valid exit.
@@ -61,6 +61,8 @@ Use the headings in [`references/spec-template.md`](references/spec-template.md)
61
61
 
62
62
  Evidence is criterion-shaped: visible states require rendered inspection at named target viewports; behavior requires an interaction trace or automated check; implementation health uses the relevant tests, type/lint checks, or affected build when available. Planning-only work names the future proof instead of claiming it exists. Where the proof is a runtime state, name the **capture seed** — the state to capture (e.g. "error-state screenshot") and the capture type. This is the seed the `observe*` step derives a capture plan from (`Given`/`When` → state+actions, `Then` → required); do not write selectors, URLs, or actions here — those are derived later. Capture contract v1 requires `schemaVersion: 1` and an explicit viewport at observe time (ADR-0018).
63
63
 
64
+ The host model may have no vision (text-only input): render artifacts are bound by **path reference** (manifest + ledger), never by viewing them. Machine assertions for review use the **text face** — HTML/CSS source, `a11y tree` text, interaction-trace JSON. A no-vision run follows this mode end to end; it is not a protocol downgrade, so do not write L6 proof that requires the composing model to look at a screenshot.
65
+
64
66
  **Done when:** L5 is not a single word (“loading”); every L6 item is a top-level list item that uses `Given -> When -> Then` in that order, can be ticked pass/fail without taste debate, and says what evidence will prove it (naming the capture seed where the proof is a runtime state); every L6 item references a reachable path row; L6 items stay user-risk units rather than one row per evidence type.
65
67
 
66
68
  ### 4. Emit