@opengsd/gsd-core 1.5.0-rc.3 → 1.5.0-rc.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/agents/gsd-advisor-researcher.md +1 -1
  3. package/agents/gsd-assumptions-analyzer.md +1 -1
  4. package/agents/gsd-code-fixer.md +1 -1
  5. package/agents/gsd-code-reviewer.md +1 -1
  6. package/agents/gsd-codebase-mapper.md +1 -1
  7. package/agents/gsd-debugger.md +1 -1
  8. package/agents/gsd-doc-writer.md +1 -1
  9. package/agents/gsd-eval-auditor.md +1 -1
  10. package/agents/gsd-executor.md +1 -1
  11. package/agents/gsd-integration-checker.md +1 -1
  12. package/agents/gsd-nyquist-auditor.md +1 -0
  13. package/agents/gsd-phase-researcher.md +1 -1
  14. package/agents/gsd-plan-checker.md +1 -1
  15. package/agents/gsd-planner.md +1 -1
  16. package/agents/gsd-project-researcher.md +1 -1
  17. package/agents/gsd-research-synthesizer.md +1 -1
  18. package/agents/gsd-roadmapper.md +55 -2
  19. package/agents/gsd-security-auditor.md +1 -0
  20. package/agents/gsd-ui-auditor.md +1 -1
  21. package/agents/gsd-ui-checker.md +1 -1
  22. package/agents/gsd-ui-researcher.md +1 -1
  23. package/agents/gsd-verifier.md +13 -2
  24. package/bin/install.js +36 -57
  25. package/commands/gsd/progress.md +2 -1
  26. package/gemini-extension.json +1 -1
  27. package/gsd-core/bin/gsd-tools.cjs +167 -3
  28. package/gsd-core/bin/lib/active-workstream-store.cjs +6 -0
  29. package/gsd-core/bin/lib/capability-state.cjs +97 -3
  30. package/gsd-core/bin/lib/capability-writer.cjs +354 -0
  31. package/gsd-core/bin/lib/config.cjs +80 -24
  32. package/gsd-core/bin/lib/edge-probe.cjs +25 -2
  33. package/gsd-core/bin/lib/frontmatter.cjs +53 -1
  34. package/gsd-core/bin/lib/git-base-branch.cjs +194 -0
  35. package/gsd-core/bin/lib/init.cjs +28 -6
  36. package/gsd-core/bin/lib/install-profiles.cjs +55 -0
  37. package/gsd-core/bin/lib/installer-migration-report.cjs +1 -0
  38. package/gsd-core/bin/lib/phase.cjs +28 -12
  39. package/gsd-core/bin/lib/plan-drift-guard.cjs +117 -0
  40. package/gsd-core/bin/lib/probe-core.cjs +117 -1
  41. package/gsd-core/bin/lib/roadmap-parser.cjs +13 -3
  42. package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +246 -0
  43. package/gsd-core/bin/lib/runtime-artifact-layout.cjs +34 -2
  44. package/gsd-core/bin/lib/state.cjs +240 -59
  45. package/gsd-core/bin/lib/verify.cjs +73 -4
  46. package/gsd-core/bin/lib/worktree-safety.cjs +2 -1
  47. package/gsd-core/references/edge-probe.md +11 -0
  48. package/gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json +14 -0
  49. package/gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json +4 -0
  50. package/gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json +32 -0
  51. package/gsd-core/references/prohibition-probe.md +248 -0
  52. package/gsd-core/templates/spec.md +14 -0
  53. package/gsd-core/workflows/complete-milestone.md +1 -5
  54. package/gsd-core/workflows/execute-phase.md +4 -3
  55. package/gsd-core/workflows/execute-plan.md +21 -6
  56. package/gsd-core/workflows/help/modes/full.md +4 -0
  57. package/gsd-core/workflows/next.md +50 -2
  58. package/gsd-core/workflows/pause-work.md +7 -1
  59. package/gsd-core/workflows/plan-phase.md +2 -0
  60. package/gsd-core/workflows/plan-review-convergence.md +14 -4
  61. package/gsd-core/workflows/pr-branch.md +4 -2
  62. package/gsd-core/workflows/quick.md +3 -2
  63. package/gsd-core/workflows/resume-project.md +17 -1
  64. package/gsd-core/workflows/settings.md +27 -1
  65. package/gsd-core/workflows/ship.md +1 -5
  66. package/gsd-core/workflows/spec-phase.md +75 -0
  67. package/gsd-core/workflows/verify-phase.md +14 -4
  68. package/hooks/dist/gsd-ensure-canonical-path.js +305 -0
  69. package/hooks/dist/gsd-statusline.js +1 -1
  70. package/hooks/dist/managed-hooks-registry.cjs +1 -0
  71. package/hooks/gsd-ensure-canonical-path.js +305 -0
  72. package/hooks/gsd-statusline.js +1 -1
  73. package/hooks/hooks.json +1 -0
  74. package/hooks/managed-hooks-registry.cjs +1 -0
  75. package/package.json +3 -3
  76. package/scripts/build-hooks.js +7 -0
  77. package/scripts/changeset/new.cjs +17 -3
  78. package/scripts/fix-slash-commands.cjs +15 -3
  79. package/scripts/gen-capability-registry.cjs +14 -1
  80. package/scripts/lint-allow-test-rule-refs.allowlist.json +2 -1
  81. package/scripts/lint-test-file-count.allowlist.json +6 -0
  82. package/scripts/mutation-matrix.cjs +108 -7
  83. package/scripts/pr-target-policy.cjs +63 -0
  84. package/scripts/research-profiles.cjs +5 -5
  85. package/scripts/run-tests.cjs +107 -6
@@ -701,6 +701,7 @@ function reapOrphanWorktrees(repoRoot, deps = {}) {
701
701
  const readFileSafe = deps.readFileSafe || defaultReadFileSafe;
702
702
  const mtimeSafe = deps.mtimeSafe || defaultMtimeSafe;
703
703
  const reapMtimeGuardMs = deps.reapMtimeGuardMs !== undefined ? deps.reapMtimeGuardMs : REAP_MTIME_GUARD_MS;
704
+ const nowMs = deps.nowMs ?? Date.now();
704
705
  const results = [];
705
706
  // 1. Discover the .git/worktrees/ admin directory.
706
707
  const gitDir = execGit(['rev-parse', '--git-dir'], { cwd: repoRoot });
@@ -803,7 +804,7 @@ function reapOrphanWorktrees(repoRoot, deps = {}) {
803
804
  }
804
805
  // 4a. Stale-lock guard: skip if lock is too fresh (PID recycling / race).
805
806
  const lockMtime = mtimeSafe(lockedFile);
806
- if (!lockMtime || Date.now() - lockMtime.getTime() < reapMtimeGuardMs) {
807
+ if (!lockMtime || nowMs - lockMtime.getTime() < reapMtimeGuardMs) {
807
808
  results.push({ path: worktreePath, status: 'skipped', reason: 'lock_too_fresh' });
808
809
  continue;
809
810
  }
@@ -77,6 +77,17 @@ Two rules keep the probe honest and prevent an "everything is N/A" failure mode:
77
77
  2. **Dismissal requires a reason string.** "N/A — input is a bounded enum, no boundary
78
78
  exists" is valid; silence is not. The reason string is the audit trail.
79
79
 
80
+ **Zero-classification surfaces an `unclassified` candidate (#1110).** The relevance filter is
81
+ a heuristic over prose cues, so a requirement whose wording *is* edge-relevant but matches no
82
+ shape cue would otherwise classify to zero shapes → zero edges and vanish from coverage with
83
+ no signal — the same silent blind spot the probe exists to catch. Instead, a requirement with
84
+ non-empty prose, no authored `shapes`, and zero matched shapes surfaces exactly one soft
85
+ `unclassified — review manually` candidate (`category: "unclassified"`, `status: "unresolved"`).
86
+ It is a dismissible nudge — resolve it, or dismiss it with a reason (e.g. a genuinely edge-free
87
+ static-asset requirement) — never a hard block. `unclassified` is a review signal, **not** a
88
+ ninth taxonomy category: the closed eight above are unchanged, and an explicit `shapes: []`
89
+ opt-out stays silent (the author's deliberate "no edge surface").
90
+
80
91
  Each raised edge carries two orthogonal axes — a resolution **lifecycle** and, when
81
92
  resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model):
82
93
 
@@ -0,0 +1,14 @@
1
+ {
2
+ "items": [
3
+ {
4
+ "requirement_id": "R1",
5
+ "category": "values",
6
+ "status": "resolved",
7
+ "verification": "judgment",
8
+ "resolution": null,
9
+ "reason": null,
10
+ "statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing"
11
+ }
12
+ ],
13
+ "coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } }
14
+ }
@@ -0,0 +1,4 @@
1
+ {
2
+ "items": [],
3
+ "coverage": { "applicable": 0, "resolved": 0, "unresolved": 0, "byVerification": { "test": 0, "judgment": 0 } }
4
+ }
@@ -0,0 +1,32 @@
1
+ {
2
+ "items": [
3
+ {
4
+ "requirement_id": "R1",
5
+ "category": "fairness",
6
+ "status": "resolved",
7
+ "verification": "judgment",
8
+ "resolution": null,
9
+ "reason": null,
10
+ "statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate"
11
+ },
12
+ {
13
+ "requirement_id": "R1",
14
+ "category": "privacy",
15
+ "status": "resolved",
16
+ "verification": "test",
17
+ "resolution": null,
18
+ "reason": null,
19
+ "statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log"
20
+ },
21
+ {
22
+ "requirement_id": "R1",
23
+ "category": "transparency",
24
+ "status": "resolved",
25
+ "verification": "judgment",
26
+ "resolution": null,
27
+ "reason": null,
28
+ "statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)"
29
+ }
30
+ ],
31
+ "coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } }
32
+ }
@@ -0,0 +1,248 @@
1
+ # Prohibition-Probe — Spec-Completeness Must-NOT Reference
2
+
3
+ Shared reference for the spec/requirements phase. Companion to
4
+ `@~/.claude/gsd-core/references/edge-probe.md`: `edge-probe` reaches the
5
+ **data/behavior-shape axis** (boundaries, adjacency, encoding, ordering) — the things a
6
+ feature must *do*. This reference reaches the orthogonal **must-NOT axis** (product, values,
7
+ safety, ethics) — the things a feature must *never silently become*. The edge-probe caught
8
+ 0/8 of these in controlled testing because it is the wrong instrument: a shape taxonomy
9
+ cannot surface "the reminder must not shame the user." Walk each requirement through the
10
+ two-stage recall→precision protocol below and resolve each surfaced prohibition to exactly
11
+ one state.
12
+
13
+ This doc is written in generic `requirements → checks → verifier` terms with no
14
+ tool-specific vocabulary, so it is portable: copy it into any spec/requirements process.
15
+ A short mapping table at the end binds it to common host structures.
16
+
17
+ ## Why front-of-pipeline
18
+
19
+ A goal-backward verifier only checks assertions that exist; an assertion only exists for a
20
+ requirement that was written down. The class of constraint this probe targets — the
21
+ *"must-NOT"* the author assumed but never wrote — is invisible to the verifier in exactly
22
+ the same way an omitted edge is, but with a sharper failure mode: a `✅ done` that means
23
+ "the code matches the words in the spec" can still ship a feature that does what the author
24
+ explicitly would *not* want. The manipulative-streak reminder, the loan model that proxies
25
+ on zip code, the audit log that stores raw SSN — each one passes a literal spec while
26
+ violating the intent. The fix is not a better verifier; it is **spec completeness**: surface
27
+ the omitted prohibition into an explicit, checkable acceptance criterion *before* any code
28
+ exists, after which the verifier reliably enforces it.
29
+
30
+ The technique is adversarial elicitation, not deterministic computation. Unlike the edge
31
+ taxonomy (a closed eight categories a classifier can apply), the recall stage is inherently
32
+ model-driven: it asks an open question and reads prose. There is **no compiled
33
+ `prohibition-probe.cjs` engine** — the recall stage is an LLM prose pass, and only the
34
+ schema/projection layer is real code (ADR-550 Decision 7b). Building a deterministic
35
+ recall adapter would be the scope-creep the maintainer flags.
36
+
37
+ ## Inputs
38
+
39
+ A list of requirements, each a `{ id, text }` record where `text` is a testable statement.
40
+ There is no shape override and no taxonomy classifier — the recall stage reads the prose
41
+ directly and the precision stage filters its raw output. The probe runs **after** the
42
+ edge-probe in the spec phase, over the same requirement list.
43
+
44
+ ## Two-stage protocol (recall → precision)
45
+
46
+ The probe is a two-pass pipeline per requirement. Stage 1 maximizes recall (cast wide);
47
+ Stage 2 restores precision (drop the noise). Running them in this order — wide then narrow —
48
+ is what keeps the surfaced list both complete and short.
49
+
50
+ **Stage 1 — Recall (adversarial probe).** Ask the single adversarial question of each
51
+ requirement:
52
+
53
+ > *What could this feature silently become that the author would NOT want, but the spec
54
+ > does not forbid?*
55
+
56
+ This question is model-robust (17/17 holistic surfacing including smaller models in the N18
57
+ experiment). It deliberately over-produces: ~10 raw candidates per requirement, including
58
+ routine engineering items. That over-production is intentional — recall first.
59
+
60
+ **Stage 2 — Precision (one-pass classifier).** Filter the raw Stage-1 list in a single pass.
61
+ The rule is a drop/keep split:
62
+
63
+ - **DROP routine-engineering items** — anything that is a normal correctness or hygiene
64
+ concern rather than an intent constraint: "must not mutate its input", "must not throw on
65
+ empty list", "must return a primitive not an object", "must not leak a file handle". These
66
+ belong to the edge-probe or to ordinary code review, not here.
67
+ - **KEEP values / safety / ethics items** — anything that, if violated, makes the feature do
68
+ something the author would object to on product, fairness, privacy, transparency, or
69
+ safety grounds: "must not use shaming framing", "must not proxy on protected attributes",
70
+ "must not store raw PII in plaintext".
71
+
72
+ This collapses the raw ~10 to ~2–3 genuine prohibitions (GT 5/5, 0 false positives on the
73
+ N18 eight-spec battery). A requirement that yields zero kept prohibitions emits an empty
74
+ list — that is the correct precision outcome for a pure utility, not a failure.
75
+
76
+ ## Canon-referral (do not mint canon items)
77
+
78
+ Some kept candidates are not bespoke at all — they are **canon** security/compliance
79
+ constraints that a dedicated tool already owns. Do NOT mint a prohibition for them. Instead
80
+ emit a one-line breadcrumb and stop:
81
+
82
+ - OWASP / prototype-pollution / path-traversal / injection → breadcrumb to `/gsd:secure-phase`
83
+ and `eslint` (security plugins), not a minted prohibition.
84
+ - GDPR / data-retention / consent → breadcrumb to `/gsd:secure-phase`.
85
+ - Generic fairness/bias canon → breadcrumb to `/gsd:secure-phase`.
86
+
87
+ The breadcrumb reads like: *"prototype-pollution is canon — covered by /gsd:secure-phase +
88
+ eslint; not minted here."* This keeps the surfaced list to the ~2–3 **bespoke** items that
89
+ no other tool would catch — the manipulative-framing prohibition, the product-specific
90
+ fairness constraint — which is the whole value of the probe. Minting canon items both
91
+ duplicates other tooling and drowns the bespoke signal (ADR-550 Decision 6).
92
+
93
+ ## Resolution states
94
+
95
+ Each surfaced prohibition carries two orthogonal axes — a resolution **lifecycle** and, when
96
+ resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model; the
97
+ lifecycle is identical to the edge-probe, the verification tiers differ):
98
+
99
+ - **status** — `resolved | dismissed | unresolved` (IDENTICAL to the edge-probe):
100
+ - **resolved** — the prohibition is addressed; *how* it is addressed is the verification tier.
101
+ - **dismissed** — not a genuine prohibition for this feature, accompanied by a required,
102
+ non-empty reason string. "N/A — this utility has no user-facing surface, no values
103
+ constraint applies" is valid; silence is not. The reason string is the audit trail.
104
+ - **unresolved** — carried forward and flagged; the author chose not to resolve it yet.
105
+ - **verification** (only when `status` is `resolved`; `null` otherwise) — `test | judgment`
106
+ (this REPLACES the edge-probe's `explicit | backstop`):
107
+ - **test** — the prohibition can be mechanically checked (a negative test, a lint rule, an
108
+ assertion that the audit log contains no raw SSN). A checkable assertion exists.
109
+ - **judgment** — the prohibition is real but cannot be reduced to a mechanical test (a
110
+ human/LLM judgment that the framing is not manipulative). It records intent and routes
111
+ to a judgment-based review rather than a green/red test.
112
+
113
+ Splitting these axes keeps the lifecycle enum free of a verification fact and lets the
114
+ prohibition adapter declare `test | judgment` without forking the shared lifecycle enum that
115
+ the edge-probe's `explicit | backstop` also uses.
116
+
117
+ ## Output schema
118
+
119
+ The probe emits, per kept prohibition, an item of the form:
120
+
121
+ ```
122
+ { requirement_id, category, status, verification, resolution, reason, statement }
123
+ ```
124
+
125
+ where `statement` is the must-NOT sentence and `category` is the values/safety/ethics class
126
+ (`values`, `fairness`, `privacy`, `transparency`, `safety`, …), plus a coverage summary:
127
+
128
+ ```
129
+ coverage: { applicable, resolved, unresolved, byVerification: { test, judgment } }
130
+ ```
131
+
132
+ `applicable` is the number of kept prohibitions, `resolved` = closed (`resolved` +
133
+ `dismissed`) status items, `unresolved` is the remainder, and `byVerification` breaks the
134
+ `resolved`-status items down by tier (`{ test, judgment }`). This JSON is the stable contract
135
+ both the reference implementation and any third-party port emit.
136
+
137
+ ## Generic mapping (requirements → checks → verifier)
138
+
139
+ | Host structure | "requirement" | a `resolved`/`test` prohibition becomes | a `resolved`/`judgment` prohibition becomes |
140
+ |----------------|---------------|------------------------------------------|----------------------------------------------|
141
+ | GSD SPEC | a SPEC Requirement | a SPEC acceptance criterion (marked prohibition) that `plan-phase` lifts into `must_haves.prohibitions` | a `must_haves.prohibitions` item routed to judgment review |
142
+ | Gherkin feature | a Scenario | a negative `Then` assertion / tagged negative scenario | a tagged scenario routed to manual review |
143
+ | OpenAPI operation | an operation | a contract test asserting the forbidden behavior never occurs | a documented constraint flagged for review |
144
+ | Docstring contract | a documented behavior | a negative assertion in the contract test | a documented must-NOT for reviewers |
145
+
146
+ The portable invariant: a `resolved`/`test` prohibition produces **a checkable negative the
147
+ verifier iterates over** (GSD: a `must_haves.prohibitions` item with a test); a
148
+ `resolved`/`judgment` prohibition produces a recorded intent routed to judgment review. An
149
+ `unresolved` prohibition is an explicit assumption the downstream planner must surface, not
150
+ silently drop.
151
+
152
+ ## Worked example (streak-reminder)
153
+
154
+ A single requirement to send a daily habit reminder. The edge-probe sees a `stateful`
155
+ requirement and asks about idempotency; the prohibition-probe asks the adversarial question
156
+ and surfaces what the reminder must never *become*. Stage 1 over-produces ("must not spam",
157
+ "must not throw on a deleted habit", "must not use shaming framing"); Stage 2 drops the
158
+ routine-engineering items and keeps the one genuine values prohibition:
159
+
160
+ ```json prohibition-probe:01-streak-reminder/expected.json
161
+ {
162
+ "items": [
163
+ {
164
+ "requirement_id": "R1",
165
+ "category": "values",
166
+ "status": "resolved",
167
+ "verification": "judgment",
168
+ "resolution": null,
169
+ "reason": null,
170
+ "statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing"
171
+ }
172
+ ],
173
+ "coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } }
174
+ }
175
+ ```
176
+
177
+ The kept prohibition is `judgment`-tier: "manipulative framing" cannot be reduced to a
178
+ mechanical test, so it records intent and routes to judgment review — but it is now an
179
+ explicit acceptance criterion the spec must clear, not an unwritten assumption.
180
+
181
+ ## Worked example (clean-utility)
182
+
183
+ A pure utility requirement — "deduplicate a list of integers" — has no user-facing surface,
184
+ no values/safety/ethics dimension. Stage 1 still over-produces ("must not mutate the input",
185
+ "must not change order"), but every candidate is routine engineering that Stage 2 drops (and
186
+ the edge-probe already owns). The correct precision outcome is an empty prohibition list — a
187
+ zero, not a false positive:
188
+
189
+ ```json prohibition-probe:02-clean-utility/expected.json
190
+ {
191
+ "items": [],
192
+ "coverage": { "applicable": 0, "resolved": 0, "unresolved": 0, "byVerification": { "test": 0, "judgment": 0 } }
193
+ }
194
+ ```
195
+
196
+ This is the precision discipline that keeps the probe from crying wolf: a utility with no
197
+ intent surface produces zero prohibitions, so a non-empty list always carries signal.
198
+
199
+ ## Worked example (multi-prohibition)
200
+
201
+ A loan-decision requirement is the high-stakes case: it surfaces several distinct
202
+ prohibitions across categories. Stage 1 produces a long list including canon items
203
+ (prototype-pollution, generic GDPR retention) that canon-referral breadcrumbs out; Stage 2
204
+ keeps the three bespoke values/safety items — a `fairness` constraint, a `privacy` constraint
205
+ (`test`-tier, mechanically checkable against the audit log), and a `transparency` constraint:
206
+
207
+ ```json prohibition-probe:03-multi-prohibition/expected.json
208
+ {
209
+ "items": [
210
+ {
211
+ "requirement_id": "R1",
212
+ "category": "fairness",
213
+ "status": "resolved",
214
+ "verification": "judgment",
215
+ "resolution": null,
216
+ "reason": null,
217
+ "statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate"
218
+ },
219
+ {
220
+ "requirement_id": "R1",
221
+ "category": "privacy",
222
+ "status": "resolved",
223
+ "verification": "test",
224
+ "resolution": null,
225
+ "reason": null,
226
+ "statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log"
227
+ },
228
+ {
229
+ "requirement_id": "R1",
230
+ "category": "transparency",
231
+ "status": "resolved",
232
+ "verification": "judgment",
233
+ "resolution": null,
234
+ "reason": null,
235
+ "statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)"
236
+ }
237
+ ],
238
+ "coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } }
239
+ }
240
+ ```
241
+
242
+ The `privacy` row is `test`-tier — "no raw SSN in the audit log" is a mechanical assertion —
243
+ while `fairness` and `transparency` are `judgment`-tier. The byVerification rollup
244
+ `{ test: 1, judgment: 2 }` is the count-preserved breakdown of the three `resolved`-status
245
+ items. Each worked-example block above is kept byte-for-byte (parsed-JSON) identical to its
246
+ fixture under `gsd-core/references/prohibition-probe-fixtures/` by
247
+ `tests/prohibition-probe.docs-fixtures.test.cjs`, so the doc and the reference data cannot
248
+ silently drift.
@@ -80,6 +80,20 @@ No "should feel good", "looks reasonable", or "generally works" — those are no
80
80
  Acceptance Criteria above; `backstop` rows must be carried into plan-phase `must_haves`.
81
81
  `⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.]
82
82
 
83
+ ## Prohibitions (must-NOT)
84
+
85
+ **Coverage:** [resolved]/[applicable] applicable prohibitions resolved · [unresolved] unresolved
86
+
87
+ | Prohibition (must-NOT statement) | Requirement | Status | Verification / Reason |
88
+ |----------------------------------|-------------|--------|------------------------|
89
+ | [MUST NOT … must-NOT statement] | [Rn] | [resolved / dismissed / ⚠ UNRESOLVED] | [verification: test \| judgment, or dismissal reason] |
90
+
91
+ [Generated by the prohibition probe (Step 5.6). `resolved` prohibitions become NEGATIVE
92
+ acceptance criteria; a `resolved`/`test` row is a checkable negative the verifier iterates
93
+ over, a `resolved`/`judgment` row routes to judgment review. Resolved prohibitions are lifted
94
+ into `must_haves.prohibitions` by plan-phase. `dismissed` rows carry a required non-empty
95
+ reason. `⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.]
96
+
83
97
  ## Ambiguity Report
84
98
 
85
99
  | Dimension | Score | Min | Status | Notes |
@@ -612,11 +612,7 @@ Extract `branching_strategy`, `phase_branch_template`, `milestone_branch_templat
612
612
 
613
613
  Detect base branch:
614
614
  ```bash
615
- BASE_BRANCH=$(gsd_run query config-get git.base_branch 2>/dev/null || echo "")
616
- if [ -z "$BASE_BRANCH" ] || [ "$BASE_BRANCH" = "null" ]; then
617
- BASE_BRANCH=$(git symbolic-ref refs/remotes/origin/HEAD 2>/dev/null | sed 's|^refs/remotes/origin/||')
618
- BASE_BRANCH="${BASE_BRANCH:-main}"
619
- fi
615
+ BASE_BRANCH=$(gsd_run query git.base-branch)
620
616
  ```
621
617
 
622
618
  **If "none":** Skip to git_tag.
@@ -189,7 +189,7 @@ CURRENT_PLAN_ID="{phase_number}-{plan_padded}"
189
189
  SUMMARY_PATH="{phase_dir}/{plan_padded}-SUMMARY.md"
190
190
  PLAN_COMMITS=$(git log --oneline --grep="${CURRENT_PLAN_ID}" -30)
191
191
  ```
192
- If production commits exist and `SUMMARY.md is missing`, stop before spawning a
192
+ If production commits exist and `SUMMARY.md is missing` (no `.planning/async-jobs/*.json` manifest matches it: a match is a legal `external_job_waiting` deferral - reconcile per `docs/reference/planning-artifacts.md`, never re-dispatch), stop before spawning a
193
193
  new executor; continuing risks duplicate work and stale `STATE.md`/ROADMAP progress.
194
194
  Offer these recovery options:
195
195
  - `close out manually` — inspect commits, write SUMMARY.md, then update STATE/ROADMAP.
@@ -296,8 +296,9 @@ Check `branching_strategy` from init:
296
296
  Fork the new phase branch off `origin/HEAD` (the project's default branch), not the current HEAD — otherwise consecutive phases compound and stay unpushed (#2916). If `$BRANCH_NAME` already exists locally, reuse it as-is.
297
297
 
298
298
  ```bash
299
- DEFAULT_BRANCH=$(git symbolic-ref --quiet --short refs/remotes/origin/HEAD 2>/dev/null | sed 's|^origin/||')
300
- DEFAULT_BRANCH=${DEFAULT_BRANCH:-main}
299
+ DEFAULT_BRANCH=$(gsd_run query git.base-branch 2>/dev/null \
300
+ || git symbolic-ref --quiet --short refs/remotes/origin/HEAD 2>/dev/null | sed 's|^origin/||' \
301
+ || echo main)
301
302
 
302
303
  if git show-ref --verify --quiet "refs/heads/$BRANCH_NAME"; then
303
304
  git switch "$BRANCH_NAME" || { echo "ERROR: Could not switch to existing branch '$BRANCH_NAME'." >&2; exit 1; }
@@ -13,10 +13,23 @@ Read config.json for planning behavior settings.
13
13
  For each executed plan, the only complete close-out order is:
14
14
  `production-code commit(s) -> SUMMARY commit -> STATE/ROADMAP update`.
15
15
 
16
- The only legal half-state is mid-production-commits while the executor is still
17
- actively working. Once production commits for a plan exist, returning without a
18
- committed SUMMARY.md is an illegal partial-plan state. The next execute-phase
19
- resume must detect that condition before dispatching another executor.
16
+ For a synchronous executor, the only legal half-state is mid-production-commits
17
+ while the executor is still actively working. Once production commits for a plan
18
+ exist, returning without a committed SUMMARY.md is an illegal partial-plan state.
19
+ The next execute-phase resume must detect that condition before dispatching
20
+ another executor.
21
+
22
+ **Async exception — `external_job_waiting`.** When an executor dispatches an
23
+ async external job (long-running compute) it commits an async-job manifest at
24
+ `.planning/async-jobs/<job>.json` and returns *without* SUMMARY.md. With a
25
+ manifest recording a non-terminal job for this plan, the SUMMARY-absent state is
26
+ a **legal deferred state** (`external_job_waiting`), not an illegal partial.
27
+ SUMMARY.md is deferred until the external job reaches a terminal state and its
28
+ output is verified. Resume reconciles against the manifest and must NOT
29
+ re-dispatch a fresh executor for a plan with a non-terminal manifest (that would
30
+ duplicate the external job). The manifest schema is the stability contract in
31
+ `docs/reference/planning-artifacts.md`; the scheduler adapter that *writes* it is
32
+ a capability (#1164), not core.
20
33
  </atomic_close_out_invariant>
21
34
 
22
35
  <available_agent_types>
@@ -47,7 +60,9 @@ If `.planning/` missing: error.
47
60
  (ls .planning/phases/XX-name/*-SUMMARY.md 2>/dev/null || true) | sort
48
61
  ```
49
62
 
50
- Find first PLAN without matching SUMMARY. Decimal phases supported (`01.1-hotfix/`):
63
+ Find first PLAN without matching SUMMARY. Decimal phases supported (`01.1-hotfix/`).
64
+
65
+ **Exclude `external_job_waiting` plans from selection.** When choosing the first PLAN that lacks a matching SUMMARY, skip any plan whose `plan_id` matches an async-job manifest in `.planning/async-jobs/` (any status) — that plan is `external_job_waiting` or awaiting reconciliation, never work to (re-)dispatch (re-dispatching would duplicate the external job). Reconcile via the manifest / safe_resume_gate instead.
51
66
 
52
67
  ```bash
53
68
  PHASE=$(echo "$PLAN_PATH" | grep -oE '[0-9]+(\.[0-9]+)?-[0-9]+')
@@ -504,7 +519,7 @@ If `USER_SETUP_CREATED=true`: display `⚠️ USER SETUP REQUIRED` with path + e
504
519
 
505
520
  | Condition | Route | Action |
506
521
  |-----------|-------|--------|
507
- | summaries < plans | **A: More plans** | Find next PLAN without SUMMARY. Yolo: auto-continue. Interactive: show next plan, suggest `/gsd:execute-phase {phase}` + `/gsd:verify-work`. STOP here. |
522
+ | summaries < plans | **A: More plans** | Find next PLAN without SUMMARY — skip any plan whose `plan_id` matches a non-terminal async-job manifest (`external_job_waiting`; see `identify_plan`). Yolo: auto-continue. Interactive: show next plan, suggest `/gsd:execute-phase {phase}` + `/gsd:verify-work`. STOP here. |
508
523
  | summaries = plans, current < highest phase | **B: Phase done** | Show completion, suggest `/gsd:plan-phase {Z+1}` + `/gsd:verify-work {Z}` + `/gsd:discuss-phase {Z+1}` |
509
524
  | summaries = plans, current = highest phase | **C: Milestone done** | Show banner, suggest `/gsd:complete-milestone` + `/gsd:verify-work` + `/gsd-add-phase` |
510
525
 
@@ -256,11 +256,15 @@ Check project status and intelligently route to next action.
256
256
  Modes:
257
257
  - **default** — progress report + intelligent routing
258
258
  - **`--next`** — auto-advance to the next logical step (use `--next --force` to bypass safety gates)
259
+ - **`--next --auto`** — like `--next`, but chains steps automatically until milestone completion or a blocking decision
260
+ - **`--next --converge`** — when the next action is planning, route it through `/gsd:plan-review-convergence` instead of `/gsd:plan-phase`; requires `workflow.plan_review_convergence=true`. `--cross-ai` is an alias. Reviewer flags (`--codex`, `--gemini`, `--claude`, `--opencode`, `--ollama`, `--lm-studio`, `--llama-cpp`, `--all`) and `--max-cycles N` forward to the convergence loop.
259
261
  - **`--forensic`** — append a 6-check integrity audit after the progress report
260
262
  - **`--do "<text>"`** — smart router: dispatch freeform intent to the matching `/gsd-*` command (see *Smart Router* above)
261
263
 
262
264
  Usage: `/gsd:progress`
263
265
  Usage: `/gsd:progress --next`
266
+ Usage: `/gsd:progress --next --auto`
267
+ Usage: `/gsd:progress --next --auto --converge`
264
268
  Usage: `/gsd:progress --forensic`
265
269
 
266
270
  ### Session Management
@@ -230,7 +230,7 @@ If the current phase directory exists but has neither CONTEXT.md nor RESEARCH.md
230
230
 
231
231
  **Route 3: Phase has context but no plans → plan**
232
232
  If the current phase has CONTEXT.md (or RESEARCH.md) but no PLAN.md files:
233
- → Next action: `/gsd:plan-phase <current-phase>`
233
+ → Next action: `/gsd:plan-phase <current-phase>` (or `/gsd:plan-review-convergence <current-phase>` when `PLAN_STRATEGY=converge`)
234
234
 
235
235
  **Route 4: Phase has plans but incomplete summaries → execute**
236
236
  If plans exist but not all have matching summaries:
@@ -254,6 +254,47 @@ If STATE.md shows paused_at:
254
254
  </step>
255
255
 
256
256
  <step name="show_and_execute">
257
+ Parse the arguments passed to this workflow to detect the plan strategy and build convergence pass-through args:
258
+
259
+ ```bash
260
+ PLAN_STRATEGY="local"
261
+ if echo "$ARGUMENTS" | grep -qE '(^|[[:space:]])\-\-(converge|cross-ai)([[:space:]]|$)'; then
262
+ PLAN_STRATEGY="converge"
263
+ fi
264
+
265
+ CONVERGENCE_ARGS=""
266
+ for REVIEW_FLAG in --codex --gemini --claude --opencode --ollama --lm-studio --llama-cpp --all --text; do
267
+ if echo "$ARGUMENTS" | grep -qE "(^|[[:space:]])${REVIEW_FLAG}([[:space:]]|$)"; then
268
+ CONVERGENCE_ARGS="${CONVERGENCE_ARGS} ${REVIEW_FLAG}"
269
+ fi
270
+ done
271
+
272
+ MAX_CYCLES_ARG=""
273
+ if echo "$ARGUMENTS" | grep -qE '\-\-max-cycles\s+[0-9]+'; then
274
+ MAX_CYCLES_ARG=$(echo "$ARGUMENTS" | grep -oE '\-\-max-cycles\s+[0-9]+' | awk '{print $2}')
275
+ CONVERGENCE_ARGS="${CONVERGENCE_ARGS} --max-cycles ${MAX_CYCLES_ARG}"
276
+ fi
277
+ ```
278
+
279
+ If `PLAN_STRATEGY` is `converge`, fail fast unless the convergence feature gate is enabled:
280
+
281
+ ```bash
282
+ if [ "$PLAN_STRATEGY" = "converge" ]; then
283
+ CONVERGENCE_ENABLED=$(gsd_run query config-get workflow.plan_review_convergence 2>/dev/null || echo "false")
284
+ if [ "$CONVERGENCE_ENABLED" != "true" ]; then
285
+ printf '%s\n' \
286
+ '/gsd:progress --next --converge is disabled (workflow.plan_review_convergence=false).' \
287
+ '' \
288
+ 'Enable plan convergence with:' \
289
+ '' \
290
+ ' gsd config-set workflow.plan_review_convergence true' \
291
+ '' \
292
+ 'Then re-run with --converge.'
293
+ exit 1
294
+ fi
295
+ fi
296
+ ```
297
+
257
298
  Display the determination:
258
299
 
259
300
  ```
@@ -269,7 +310,9 @@ Display the determination:
269
310
  Then immediately invoke the determined command via SlashCommand.
270
311
  Do not ask for confirmation — the whole point of `/gsd:progress --next` is zero-friction advancement.
271
312
 
272
- **If `--auto` was passed:** after the determined command completes, automatically re-invoke `/gsd:progress --next --auto` to continue chaining to the next step. Repeat until one of:
313
+ **Route 3 convergence override:** When the routing decision is Route 3 (plan) and `PLAN_STRATEGY=converge`, invoke `/gsd:plan-review-convergence <current-phase> ${CONVERGENCE_ARGS}` instead of `/gsd:plan-phase <current-phase>`.
314
+
315
+ **If `--auto` was passed:** after the determined command completes, automatically re-invoke `/gsd:progress --next --auto` (forwarding `--converge`/`--cross-ai` and any reviewer flags if they were originally passed) to continue chaining to the next step. Repeat until one of:
273
316
  - A milestone completes (`/gsd:complete-milestone` is reached)
274
317
  - A blocking decision is required (safety gate triggers, prior-phase completeness prompt, user input needed)
275
318
  - An error or paused state is detected
@@ -296,4 +339,9 @@ Resume with: `/gsd:progress --next --auto` once resolved.
296
339
  - [ ] Next action correctly determined from routing rules
297
340
  - [ ] Command invoked immediately without user confirmation
298
341
  - [ ] Clear status shown before invoking
342
+ - [ ] `--converge` routes Route 3 planning through `gsd-plan-review-convergence`
343
+ - [ ] `--cross-ai` is accepted as an alias for `--converge`
344
+ - [ ] `--converge` fails fast with enable instructions when `workflow.plan_review_convergence=false`
345
+ - [ ] `--converge` forwards reviewer selector flags and `--max-cycles N`
346
+ - [ ] Default planning remains `gsd-plan-phase` when convergence is not requested
299
347
  </success_criteria>
@@ -48,7 +48,8 @@ If phase is detected, proceed with phase handoff path. Otherwise use the first m
48
48
  6. **Human actions pending**: Things that need manual intervention (MCP setup, API keys, approvals, manual testing)
49
49
  7. **Background processes**: Any running servers/watchers that were part of the workflow
50
50
  8. **Files modified**: What's changed but not committed
51
- 9. **Blocking constraints**: Anti-patterns or methodological failures encountered during this session that a resuming agent MUST be aware of before proceeding. Only include items discovered through actual failure — not warnings or predictions. Assign each constraint a `severity`:
51
+ 9. **Outstanding async external jobs**: any `.planning/async-jobs/*.json` manifests for non-terminal jobs — record job id, backend, status, expected artifacts, verification + resume commands, and any watcher/daemon state. Do NOT cancel the external job; it keeps running across the pause.
52
+ 10. **Blocking constraints**: Anti-patterns or methodological failures encountered during this session that a resuming agent MUST be aware of before proceeding. Only include items discovered through actual failure — not warnings or predictions. Assign each constraint a `severity`:
52
53
  - `blocking` — The resuming agent MUST demonstrate understanding before proceeding. The discuss-phase and execute-phase workflows will enforce a mandatory understanding check.
53
54
  - `advisory` — Important context but does not gate resumption.
54
55
 
@@ -93,6 +94,9 @@ timestamp=$(gsd_run query current-timestamp full --raw)
93
94
  "blockers": [
94
95
  {"description": "{blocker}", "type": "technical|human_action|external", "workaround": "{if any}"}
95
96
  ],
97
+ "async_jobs": [
98
+ {"manifest": ".planning/async-jobs/{job}.json", "job_id": "{id}", "backend": "{backend}", "status": "running", "submit_command": "{cmd}", "submitted_at": "{iso8601}", "expected_artifacts": ["..."], "verification_command": "{cmd}", "resume_command": "{cmd}"}
99
+ ],
96
100
  "human_actions_pending": [
97
101
  {"action": "{what needs to be done}", "context": "{why}", "blocking": true}
98
102
  ],
@@ -104,6 +108,8 @@ timestamp=$(gsd_run query current-timestamp full --raw)
104
108
  "context_notes": "{mental state, approach, what you were thinking}"
105
109
  }
106
110
  ```
111
+
112
+ Any recorded `async_jobs` entries are the primary resume context on the next session — check them first before treating a PLAN-without-SUMMARY as incomplete work.
107
113
  </step>
108
114
 
109
115
  <step name="write">
@@ -892,6 +892,7 @@ Output consumed by /gsd:execute-phase. Plans need:
892
892
  - Verification criteria
893
893
  - must_haves for goal-backward verification
894
894
  - If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths`, and every `backstop` edge into `must_haves.truths` as a non-inferable check (note it needs a held-out/property-based test). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them.
895
+ - If the SPEC has a `## Prohibitions` section, lift every resolved prohibition into the `must_haves.prohibitions:` sibling block (NOT `truths` — ADR-550 D3) carrying `statement` + `status` + `verification`; unresolved prohibitions are explicit assumptions — surface them in the plan, do not silently drop them. A prohibition is a must-NOT (negative) check that belongs in its own `must_haves.prohibitions` block. Never place a must-NOT under `must_haves.truths` — that block keeps positive-observable semantics only.
895
896
  - **"Artifacts this phase produces" section (MANDATORY)** — list every symbol this phase creates: decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths. The plan-review-convergence source-grounding pass reads this section to exclude newly-created symbols from drift verification; omitting it causes new symbols to be flagged for acknowledgement.
896
897
  </downstream_consumer>
897
898
 
@@ -938,6 +939,7 @@ Every task MUST include these fields — they are NOT optional:
938
939
  - [ ] must_haves derived from phase goal
939
940
  - [ ] Every PLAN.md includes an "Artifacts this phase produces" section listing symbols created by this phase (decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths)
940
941
  - [ ] Every SPEC ## Edge Coverage covered/backstop edge is represented in a plan's must_haves (no silent drops)
942
+ - [ ] Every SPEC ## Prohibitions resolved item is represented in a plan's must_haves.prohibitions (no silent drops)
941
943
  </quality_gate>
942
944
  ```
943
945
 
@@ -185,13 +185,23 @@ Run this pass unless `plan_review.source_grounding` is `false`. It verifies ever
185
185
 
186
186
  1. **Enumerate cited symbols.** List every referenced symbol by kind, quoting the plan line for each (coverage must be auditable): decorators (`@name`), classes/methods (`Class.method`), functions (`module.function`), CLI flags (`--name`), file paths, dataclass/struct fields.
187
187
  2. **Exclude new artifacts.** Do NOT verify symbols the plan declares under its "Artifacts this phase produces" section — those are created by this phase, not references to existing code.
188
- 3. **Resolve each remaining symbol** using the adapter named by `plan_review.source_grounding_authority` (default `grep`):
188
+ 3. **Resolve each remaining symbol** using the effective authority adapter (resolved deterministically — see step 4a):
189
189
  - `grep` — ripgrep / Read the source; confirm the name appears as a real declaration.
190
190
  - `intel` — consult `.planning/intel/API-SURFACE.md` / `api-map.json` (only when `intel.enabled`).
191
191
  Record one verdict per symbol: **VERIFIED** (quote `file:line`), **MISSING** (adapter can check this language/kind and the symbol is absent), **AMBIGUOUS** (multiple candidates), or **UNCHECKABLE** (adapter cannot analyze this language/kind — e.g. non-JS under `intel`, or any signature under `grep`). Never treat UNCHECKABLE as verified or missing.
192
- 4. **Severity & gating:**
193
- - **MISSING** at authority `grep`/`intel` → `needs-acknowledgement`: the plan proceeds only if the author confirms the symbol is genuinely new or dynamically resolved, and that acknowledgement is recorded. A hard block is reserved for higher-authority adapters (LSP/SCIP) that can prove absence.
194
- - **AMBIGUOUS** → MEDIUM. **UNCHECKABLE** → INFO.
192
+ 4a. **Resolve effective authority** (deterministic — replaces manual `intel.enabled` reasoning):
193
+ ```bash
194
+ EFFECTIVE_AUTHORITY=$(gsd_run drift-guard authority --raw)
195
+ ```
196
+ 4. **Severity & gating** — classify each symbol's verdict using the seam (do not apply the table manually):
197
+ ```bash
198
+ # For each symbol, e.g.:
199
+ RESULT=$(gsd_run drift-guard severity --status <verdict> --authority "$EFFECTIVE_AUTHORITY")
200
+ # $RESULT is JSON: {"severity":"…","hardBlock":true|false}
201
+ ```
202
+ - `hardBlock: true` (HIGH at authority `lsp`/`scip`) — stops the review cycle immediately; do not proceed until the plan author resolves the missing symbol.
203
+ - `hardBlock: false`, severity `needs-acknowledgement` — plan proceeds only if the author confirms the symbol is genuinely new or dynamically resolved, and that acknowledgement is recorded.
204
+ - `AMBIGUOUS` → MEDIUM. `UNCHECKABLE` → INFO.
195
205
  - Signature mismatches cannot be asserted under `grep`/`intel`; report the signature as UNCHECKABLE.
196
206
  5. **Coverage block.** Append a "Verification coverage" section to `REVIEWS.md` listing every UNCHECKABLE/skipped symbol and why — a clean review must never silently mean "nothing was checked."
197
207