@opengsd/gsd-core 1.5.0 → 1.6.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/agents/gsd-plan-checker.md +34 -0
  3. package/agents/gsd-planner.md +2 -0
  4. package/agents/gsd-roadmapper.md +6 -0
  5. package/bin/install.js +199 -365
  6. package/commands/gsd/capture.md +5 -1
  7. package/gemini-extension.json +1 -1
  8. package/gsd-core/bin/gsd-tools.cjs +695 -5
  9. package/gsd-core/bin/lib/adr-parser.cjs +45 -23
  10. package/gsd-core/bin/lib/audit.cjs +2 -2
  11. package/gsd-core/bin/lib/capability-consent.cjs +763 -0
  12. package/gsd-core/bin/lib/capability-ledger.cjs +831 -0
  13. package/gsd-core/bin/lib/capability-lifecycle.cjs +1551 -0
  14. package/gsd-core/bin/lib/capability-loader.cjs +764 -0
  15. package/gsd-core/bin/lib/capability-lock.cjs +553 -0
  16. package/gsd-core/bin/lib/capability-registry.cjs +198 -4
  17. package/gsd-core/bin/lib/capability-source.cjs +1242 -0
  18. package/gsd-core/bin/lib/capability-state.cjs +9 -6
  19. package/gsd-core/bin/lib/capability-trust.cjs +550 -0
  20. package/gsd-core/bin/lib/capability-validator.cjs +2066 -0
  21. package/gsd-core/bin/lib/capability-writer.cjs +14 -5
  22. package/gsd-core/bin/lib/check-command-router.cjs +69 -18
  23. package/gsd-core/bin/lib/command-aliases.cjs +8 -0
  24. package/gsd-core/bin/lib/commands.cjs +247 -0
  25. package/gsd-core/bin/lib/config-loader.cjs +98 -84
  26. package/gsd-core/bin/lib/config-schema.cjs +26 -7
  27. package/gsd-core/bin/lib/config.cjs +7 -1
  28. package/gsd-core/bin/lib/decisions.cjs +149 -60
  29. package/gsd-core/bin/lib/frontmatter.cjs +7 -3
  30. package/gsd-core/bin/lib/gap-checker.cjs +126 -11
  31. package/gsd-core/bin/lib/init.cjs +91 -22
  32. package/gsd-core/bin/lib/legacy-cleanup.cjs +96 -0
  33. package/gsd-core/bin/lib/loop-resolver.cjs +26 -2
  34. package/gsd-core/bin/lib/markdown-sectionizer.cjs +471 -0
  35. package/gsd-core/bin/lib/milestone.cjs +41 -2
  36. package/gsd-core/bin/lib/phase-command-router.cjs +5 -0
  37. package/gsd-core/bin/lib/phase-id.cjs +25 -11
  38. package/gsd-core/bin/lib/phase-lifecycle.cjs +14 -5
  39. package/gsd-core/bin/lib/phase.cjs +33 -4
  40. package/gsd-core/bin/lib/probe-core.cjs +7 -0
  41. package/gsd-core/bin/lib/prohibition-enforcement.cjs +59 -26
  42. package/gsd-core/bin/lib/project-root.cjs +89 -2
  43. package/gsd-core/bin/lib/resolution.cjs +26 -0
  44. package/gsd-core/bin/lib/roadmap-command-router.cjs +16 -3
  45. package/gsd-core/bin/lib/roadmap-parser.cjs +73 -106
  46. package/gsd-core/bin/lib/roadmap-upgrade.cjs +47 -17
  47. package/gsd-core/bin/lib/roadmap.cjs +5 -2
  48. package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +423 -3
  49. package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +77 -0
  50. package/gsd-core/bin/lib/runtime-artifact-layout.cjs +1 -28
  51. package/gsd-core/bin/lib/runtime-homes.cjs +53 -1
  52. package/gsd-core/bin/lib/runtime-name-policy.cjs +44 -0
  53. package/gsd-core/bin/lib/semver-compare.cjs +127 -0
  54. package/gsd-core/bin/lib/shell-command-projection.cjs +55 -1
  55. package/gsd-core/bin/lib/state-document.cjs +4 -2
  56. package/gsd-core/bin/lib/state.cjs +317 -161
  57. package/gsd-core/bin/lib/surface.cjs +12 -19
  58. package/gsd-core/bin/lib/uat-predicate.cjs +7 -47
  59. package/gsd-core/bin/lib/uat.cjs +39 -26
  60. package/gsd-core/bin/lib/validate.cjs +5 -2
  61. package/gsd-core/bin/lib/verify.cjs +40 -15
  62. package/gsd-core/bin/lib/worktree-safety.cjs +202 -0
  63. package/gsd-core/bin/shared/config-defaults.manifest.json +6 -1
  64. package/gsd-core/bin/shared/config-schema.manifest.json +5 -1
  65. package/gsd-core/references/context-budget.md +8 -8
  66. package/gsd-core/references/execute-phase-between-wave-reset.md +43 -0
  67. package/gsd-core/references/execute-phase-context-guard.md +16 -0
  68. package/gsd-core/references/execute-phase-wave-guard.md +33 -0
  69. package/gsd-core/references/planner-antipatterns.md +48 -0
  70. package/gsd-core/references/planning-config.md +4 -0
  71. package/gsd-core/references/prohibition-probe.md +15 -9
  72. package/gsd-core/references/scout-codebase.md +2 -2
  73. package/gsd-core/workflows/autonomous.md +33 -33
  74. package/gsd-core/workflows/diagnose-issues.md +6 -1
  75. package/gsd-core/workflows/discuss-phase/templates/context.md +1 -1
  76. package/gsd-core/workflows/discuss-phase.md +1 -2
  77. package/gsd-core/workflows/execute-phase.md +12 -12
  78. package/gsd-core/workflows/help/modes/full.md +10 -0
  79. package/gsd-core/workflows/list-seeds.md +63 -0
  80. package/gsd-core/workflows/manager.md +37 -37
  81. package/gsd-core/workflows/pr-branch.md +156 -0
  82. package/gsd-core/workflows/quick.md +6 -1
  83. package/gsd-core/workflows/review.md +10 -2
  84. package/gsd-core/workflows/spec-phase.md +8 -3
  85. package/gsd-core/workflows/verify-phase.md +2 -2
  86. package/package.json +6 -3
  87. package/scripts/gen-capability-matrix.cjs +284 -0
  88. package/scripts/gen-capability-registry.cjs +96 -1853
  89. package/scripts/lint-regression-test-names.allowlist.json +1 -0
  90. package/scripts/lint-resolution-provenance.allowlist.json +1 -0
  91. package/scripts/lint-resolution-provenance.cjs +192 -0
  92. package/scripts/lint-test-file-count.allowlist.json +9 -0
  93. package/scripts/prompt-injection-scan.sh +1 -0
  94. package/scripts/run-tests.cjs +14 -0
  95. package/scripts/sync-manifest-versions.cjs +77 -5
@@ -1,6 +1,6 @@
1
1
  <purpose>
2
2
 
3
- Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and plan/execute as background agents, and loops back to the dashboard after each action. Enables parallel phase work from one terminal.
3
+ Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and runs plan/execute inline (backgrounded only on Codex), and loops back to the dashboard after each action. Enables parallel phase work from one terminal.
4
4
 
5
5
  </purpose>
6
6
 
@@ -45,7 +45,7 @@ Display startup banner:
45
45
  {milestone_version} — {milestone_name}
46
46
  {phase_count} phases · {completed_count} complete
47
47
 
48
- ✓ Discuss → inline ◆ Plan/Execute → background
48
+ ✓ Discuss → inline ◆ Plan/Execute → inline (background on Codex)
49
49
  Dashboard auto-refreshes when background work is active.
50
50
  ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
51
51
  ```
@@ -221,8 +221,8 @@ Go to exit step.
221
221
 
222
222
  When the user selects a compound option, behavior depends on the runtime — the Plan Phase N / Execute Phase N handlers below resolve it via `gsd_run query config-get runtime`:
223
223
 
224
- - **On Claude Code:** a backgrounded agent cannot nest the pipeline's subagents, so run the chosen plan/execute step(s) **inline** via their handlers below (in order), then run the inline discuss. There is no overlap.
225
- - **On other runtimes:** **Spawn all background agents first** (plan/execute) — dispatch them in parallel using the Plan Phase N / Execute Phase N handlers below — then run the inline discuss; the background agents continue while you discuss.
224
+ - **On Codex:** **Spawn all background agents first** (plan/execute) — dispatch them in parallel using the Plan Phase N / Execute Phase N handlers below — then run the inline discuss; the background agents continue while you discuss.
225
+ - **Otherwise (Claude Code or any other non-Codex runtime):** a backgrounded agent cannot reliably nest the pipeline's subagents, so run the chosen plan/execute step(s) **inline** via their handlers below (in order), then run the inline discuss. There is no overlap.
226
226
 
227
227
  Inline discuss:
228
228
 
@@ -244,27 +244,13 @@ After discuss completes, loop back to dashboard step.
244
244
 
245
245
  ### Plan Phase N
246
246
 
247
- Planning runs autonomously. **First resolve the runtime.** On Claude Code a backgrounded agent has no `Agent`/`Task` tool, so it cannot spawn the plan-checker the pipeline relies on — backgrounding it there silently turns `workflow.plan_check` into a self-check. So run plan **inline** on Claude Code, and **background** it only on runtimes where a backgrounded agent can still nest subagents.
247
+ Planning runs autonomously. **First resolve the runtime.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background.
248
248
 
249
249
  ```bash
250
- RUNTIME=$(gsd_run query config-get runtime --default claude 2>/dev/null || echo "claude")
250
+ RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
251
251
  ```
252
252
 
253
- **If `RUNTIME` is `claude` (Claude Code):** Run plan inline so the plan-checker and quality gates actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
254
-
255
- ```
256
- Skill(skill="gsd-plan-phase", args="{N} --auto {manager_flags.plan}")
257
- ```
258
-
259
- Display while it runs:
260
-
261
- ```
262
- ◆ Planning Phase {N}: {phase_name}... (runs inline so the plan-checker runs — the dashboard resumes when it returns, ~1–5 min; expected, not a freeze)
263
- ```
264
-
265
- Then loop back to dashboard step.
266
-
267
- **If `RUNTIME` is not `claude` (e.g. Codex):** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
253
+ **If `RUNTIME` is `codex`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
268
254
 
269
255
  ```
270
256
  Agent(
@@ -286,7 +272,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
286
272
  )
287
273
  ```
288
274
 
289
- > **ORCHESTRATOR RULE — NON-CLAUDE RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any planning work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume planning-related work when the subagent result is available.
275
+ > **ORCHESTRATOR RULE — CODEX RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any planning work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume planning-related work when the subagent result is available.
290
276
 
291
277
  Display:
292
278
 
@@ -296,29 +282,29 @@ Display:
296
282
 
297
283
  Loop back to dashboard step.
298
284
 
299
- ### Execute Phase N
300
-
301
- Execution runs autonomously. **First resolve the runtime.** On Claude Code a backgrounded agent has no `Agent`/`Task` tool, so it cannot spawn the per-plan worktree-isolated executors or the verifier — backgrounding it there silently disables `workflow.use_worktrees` isolation and `workflow.verifier`. So run execute **inline** on Claude Code, and **background** it only on runtimes where a backgrounded agent can still nest subagents.
302
-
303
- ```bash
304
- RUNTIME=$(gsd_run query config-get runtime --default claude 2>/dev/null || echo "claude")
305
- ```
306
-
307
- **If `RUNTIME` is `claude` (Claude Code):** Run execute inline so worktree isolation and the verifier actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
285
+ **Otherwise (Claude Code or any other non-Codex runtime):** Run plan inline so the plan-checker and quality gates actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
308
286
 
309
287
  ```
310
- Skill(skill="gsd-execute-phase", args="{N} {manager_flags.execute}")
288
+ Skill(skill="gsd-plan-phase", args="{N} --auto {manager_flags.plan}")
311
289
  ```
312
290
 
313
291
  Display while it runs:
314
292
 
315
293
  ```
316
- ◆ Executing Phase {N}: {phase_name}... (runs inline so worktree isolation and verification run — the dashboard resumes when it returns; expected, not a freeze)
294
+ ◆ Planning Phase {N}: {phase_name}... (runs inline so the plan-checker runs — the dashboard resumes when it returns, ~1–5 min; expected, not a freeze)
317
295
  ```
318
296
 
319
297
  Then loop back to dashboard step.
320
298
 
321
- **If `RUNTIME` is not `claude` (e.g. Codex):** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
299
+ ### Execute Phase N
300
+
301
+ Execution runs autonomously. **First resolve the runtime.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background.
302
+
303
+ ```bash
304
+ RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
305
+ ```
306
+
307
+ **If `RUNTIME` is `codex`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
322
308
 
323
309
  ```
324
310
  Agent(
@@ -340,7 +326,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
340
326
  )
341
327
  ```
342
328
 
343
- > **ORCHESTRATOR RULE — NON-CLAUDE RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any execution work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume execution-related work when the subagent result is available.
329
+ > **ORCHESTRATOR RULE — CODEX RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any execution work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume execution-related work when the subagent result is available.
344
330
 
345
331
  Display:
346
332
 
@@ -350,6 +336,20 @@ Display:
350
336
 
351
337
  Loop back to dashboard step.
352
338
 
339
+ **Otherwise (Claude Code or any other non-Codex runtime):** Run execute inline so worktree isolation and the verifier actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
340
+
341
+ ```
342
+ Skill(skill="gsd-execute-phase", args="{N} {manager_flags.execute}")
343
+ ```
344
+
345
+ Display while it runs:
346
+
347
+ ```
348
+ ◆ Executing Phase {N}: {phase_name}... (runs inline so worktree isolation and verification run — the dashboard resumes when it returns; expected, not a freeze)
349
+ ```
350
+
351
+ Then loop back to dashboard step.
352
+
353
353
  </step>
354
354
 
355
355
  <step name="background_completion">
@@ -422,8 +422,8 @@ Display final status with progress bar:
422
422
  - [ ] Dependency resolution: blocked phases show which deps are missing
423
423
  - [ ] Recommendations prioritize: execute > plan > discuss
424
424
  - [ ] Discuss phases run inline via Skill() — interactive questions work
425
- - [ ] Plan phases spawn background Task agents — return to dashboard immediately
426
- - [ ] Execute phases spawn background Task agents — return to dashboard immediately
425
+ - [ ] Plan phases run inline (or as background Task agents on Codex) — dashboard resumes when complete
426
+ - [ ] Execute phases run inline (or as background Task agents on Codex) — dashboard resumes when complete
427
427
  - [ ] Dashboard refreshes pick up changes from background agents via disk state
428
428
  - [ ] Background agent completion triggers notification and dashboard refresh
429
429
  - [ ] Background agent errors present retry/skip options
@@ -43,6 +43,162 @@ Commits: {AHEAD} ahead
43
43
  ```
44
44
  </step>
45
45
 
46
+ <step name="handle_sub_repos">
47
+ Read the sub-repo list from config using the canonical key path — `planning.sub_repos`.
48
+ A non-zero exit code means the key is absent; treat that as "no sub-repos configured".
49
+
50
+ ```bash
51
+ SUB_REPOS_JSON=$(gsd_run query config-get planning.sub_repos 2>/dev/null)
52
+ if [ $? -ne 0 ] || [ -z "$SUB_REPOS_JSON" ] || [ "$SUB_REPOS_JSON" = "null" ] || [ "$SUB_REPOS_JSON" = "[]" ]; then
53
+ : # Not configured or empty — skip to analyze_commits
54
+ fi
55
+ ```
56
+
57
+ Scan each sub-repo for uncommitted changes using node (always available — avoids undeclared
58
+ jq dependency). Write dirty repo names to a temp file so the list survives across
59
+ subsequent command executions:
60
+
61
+ ```bash
62
+ ROOT=$(git rev-parse --show-toplevel)
63
+ DIRTY_FILE=$(mktemp)
64
+
65
+ node -e "
66
+ const repos = JSON.parse(process.argv[1]);
67
+ const { execFileSync } = require('child_process');
68
+ const path = require('path');
69
+ const fs = require('fs');
70
+ const root = process.argv[2];
71
+ // realpath parity with the pr-subrepo seam's validatePath: resolve $ROOT through
72
+ // symlinks once so the containment check below compares real paths, not text.
73
+ let realRoot;
74
+ try { realRoot = fs.realpathSync(root); } catch (_) { realRoot = path.resolve(root); }
75
+ const out = [];
76
+ for (const r of repos) {
77
+ // Reject before any git invocation: this scan runs on raw config values,
78
+ // ahead of the pr-subrepo seam's own validatePath guard. A traversal,
79
+ // embedded-newline, or symlink entry here would run git outside the
80
+ // workspace, or inject a spurious record into the dirty-file output.
81
+ if (typeof r !== 'string' || !/^[A-Za-z0-9._\/-]+$/.test(r)) continue;
82
+ // realpathSync follows symlinks — path.resolve only normalizes '..' textually,
83
+ // so an in-tree symlink pointing outside root would otherwise smuggle git out.
84
+ let resolved;
85
+ try { resolved = fs.realpathSync(path.resolve(realRoot, r)); } catch (_) { continue; }
86
+ if (resolved !== realRoot && !resolved.startsWith(realRoot + path.sep)) continue;
87
+ try {
88
+ const res = execFileSync('git', ['-C', resolved, 'status', '--porcelain'],
89
+ { encoding: 'utf8', timeout: 10_000 });
90
+ // Exclude untracked-only repos: seam filters ?? lines, so detection must match.
91
+ const tracked = res.split('\n').filter(l => l.length > 0 && !l.startsWith('??'));
92
+ if (tracked.length > 0) out.push(r);
93
+ } catch (_) {}
94
+ }
95
+ fs.writeFileSync(process.argv[3], out.join('\n'));
96
+ " "$SUB_REPOS_JSON" "$ROOT" "$DIRTY_FILE"
97
+
98
+ DIRTY_REPOS=$(cat "$DIRTY_FILE")
99
+ ```
100
+
101
+ If `$DIRTY_REPOS` is empty, remove the temp file and continue to `analyze_commits`.
102
+
103
+ Display dirty repos and prompt the user:
104
+
105
+ ```
106
+ Sub-repos with uncommitted changes:
107
+ backend
108
+ frontend
109
+
110
+ How should sub-repo changes be handled?
111
+ 1. all — branch, commit (explicit files only), push -u, open companion PR per repo
112
+ 2. select — choose which sub-repos to process
113
+ 3. skip — ignore sub-repos, continue with root repo only
114
+ ```
115
+
116
+ If the user chooses **skip**, remove the temp file and continue to `analyze_commits`.
117
+
118
+ For each selected sub-repo `$REPO_REL`, delegate all git work to the `pr-subrepo` query
119
+ seam — it stages explicit changed files (never `git add -A`), creates the branch,
120
+ commits, and pushes with `--set-upstream`. Branch names include the repo slug to avoid
121
+ colliding with the root `PR_BRANCH` that `create_pr_branch` creates later:
122
+
123
+ ```bash
124
+ # Replace path separators to make the name safe as a branch component
125
+ REPO_SAFE="${REPO_REL//\//-}"
126
+ SUB_BRANCH="${CURRENT_BRANCH}-${REPO_SAFE}-pr"
127
+ COMMIT_MSG="fix(${REPO_REL}): sync uncommitted changes for PR"
128
+
129
+ RESULT=$(gsd_run query pr-subrepo "$COMMIT_MSG" \
130
+ --repo "$REPO_REL" \
131
+ --branch "$SUB_BRANCH")
132
+ SUBREPO_EXIT=$?
133
+ ```
134
+
135
+ If the seam exited non-zero (stage/commit/push failure), report its error and move on to
136
+ the next selected sub-repo. **Do not run the companion-PR step below for this repo** —
137
+ the seam's stderr already explains the failure, and the "branch pushed" path would
138
+ otherwise contradict it:
139
+
140
+ ```bash
141
+ if [ "$SUBREPO_EXIT" -ne 0 ]; then
142
+ echo "pr-subrepo failed for $REPO_REL — see error above; skipping companion PR." >&2
143
+ fi
144
+ ```
145
+
146
+ Only when `$SUBREPO_EXIT` is `0`, parse the structured result with node and open the
147
+ companion PR. If `remote_slug` is null (non-GitHub remote), skip `gh pr create` and show
148
+ the push URL instead:
149
+
150
+ ```bash
151
+ REMOTE_SLUG=$(node -e "
152
+ try { console.log(JSON.parse(process.argv[1]).remote_slug || ''); } catch(_) {}
153
+ " "$RESULT")
154
+
155
+ if [ -n "$REMOTE_SLUG" ]; then
156
+ # Defense-in-depth: $REPO_REL was already validated by the dirty-scan filter and
157
+ # the pr-subrepo seam's validatePath, but these are separate, independent git -C
158
+ # invocations on the same value. Resolve it through symlinks with the SAME realpath
159
+ # containment the seam uses (path.resolve alone would not catch a symlink escape),
160
+ # and run git against the validated absolute path rather than re-concatenating.
161
+ SUB_REPO_DIR=$(node -e "
162
+ const fs = require('fs'), path = require('path');
163
+ try {
164
+ const realRoot = fs.realpathSync(process.argv[1]);
165
+ const resolved = fs.realpathSync(path.resolve(realRoot, process.argv[2]));
166
+ if (resolved !== realRoot && !resolved.startsWith(realRoot + path.sep)) process.exit(1);
167
+ process.stdout.write(resolved);
168
+ } catch (_) { process.exit(1); }
169
+ " "$ROOT" "$REPO_REL" 2>/dev/null)
170
+
171
+ if [ -z "$SUB_REPO_DIR" ]; then
172
+ echo "Refusing unsafe sub-repo path: $REPO_REL" >&2
173
+ SUB_TARGET="$TARGET"
174
+ else
175
+ # Resolve base branch: use $TARGET if it exists in sub-repo, else fall back to
176
+ # the sub-repo's own default branch
177
+ if git -C "$SUB_REPO_DIR" ls-remote --exit-code --heads origin "$TARGET" \
178
+ > /dev/null 2>&1; then
179
+ SUB_TARGET="$TARGET"
180
+ else
181
+ SUB_TARGET=$(git -C "$SUB_REPO_DIR" remote show origin 2>/dev/null \
182
+ | awk '/HEAD branch/ {print $NF}')
183
+ SUB_TARGET="${SUB_TARGET:-main}"
184
+ fi
185
+ fi
186
+
187
+ gh pr create \
188
+ --repo "$REMOTE_SLUG" \
189
+ --base "$SUB_TARGET" \
190
+ --head "$SUB_BRANCH" \
191
+ --title "$COMMIT_MSG" \
192
+ --body "Companion PR for root repo branch \`$CURRENT_BRANCH\`."
193
+ else
194
+ echo "No GitHub remote detected for $REPO_REL — branch pushed, open PR manually."
195
+ fi
196
+ ```
197
+
198
+ After processing all selected sub-repos, remove the temp file and continue to
199
+ `analyze_commits` for the root repo.
200
+ </step>
201
+
46
202
  <step name="analyze_commits">
47
203
  Classify commits:
48
204
 
@@ -137,7 +137,12 @@ AGENT_SKILLS_VERIFIER=$(gsd_run query agent-skills gsd-verifier)
137
137
  Parse JSON for: `planner_model`, `executor_model`, `checker_model`, `verifier_model`, `commit_docs`, `branch_name`, `quick_id`, `slug`, `date`, `timestamp`, `quick_dir`, `task_dir`, `roadmap_exists`, `planning_exists`.
138
138
 
139
139
  ```bash
140
- USE_WORKTREES=$(gsd_run query config-get workflow.use_worktrees 2>/dev/null || echo "true")
140
+ USE_WORKTREES=$(gsd_run query config-get workflow.use_worktrees --raw 2>/dev/null || echo "true")
141
+ RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
142
+ if [ "$RUNTIME" != "claude" ] && [ "$USE_WORKTREES" != "false" ]; then
143
+ echo "FATAL: git worktree isolation (isolation=\"worktree\") is unsupported on runtime '$RUNTIME' — it would run executor agents unisolated against the main checkout. Set workflow.use_worktrees=false." >&2
144
+ exit 1
145
+ fi
141
146
  ```
142
147
 
143
148
  If `USE_WORKTREES` is not `"false"`, run a startup orphan sweep before spawning any executors. This reaps locked worktrees whose lock-owner process is dead, whose branch is merged into the default branch, and whose lock file mtime is older than 5 minutes. Running it at startup prevents accumulation of orphaned worktrees from prior sessions that exited without cleanup (#3707).
@@ -157,6 +157,14 @@ Provide structured feedback on plan quality, completeness, and risks.
157
157
 
158
158
  ## Review Instructions
159
159
 
160
+ **Verify against source — do not review the plan text in isolation.** You are running inside the project's git working tree (the current directory). The plans reference real files, migrations, routes, and tests that exist in this repo now.
161
+ 1. Open the referenced files and check each claim against the actual code.
162
+ 2. For every strength or concern, cite concrete `path/to/file:line` evidence plus the mechanism.
163
+ 3. When a plan asserts a mechanism works (a guard, a query filter, a test that exercises a path), trace whether it actually does what is claimed — do not take the plan's word for it.
164
+ 4. If you cannot read the repo (no file access), say so and downgrade that finding to an open question rather than asserting it.
165
+
166
+ Findings citing `file:line` evidence are weighted far more heavily than impressionistic ones; a review that only restates the plan's own claims has low value.
167
+
160
168
  Analyze each plan and provide:
161
169
 
162
170
  1. **Summary** — One-paragraph assessment
@@ -273,7 +281,7 @@ fi
273
281
 
274
282
  **CodeRabbit:**
275
283
 
276
- Note: CodeRabbit reviews the current git diff/working tree — it does not accept a prompt or model flag. It may take up to 5 minutes. Use `timeout: 360000` on the Bash tool call.
284
+ Note: CodeRabbit reviews the current git diff/working tree — it does not accept a prompt or model flag. It may take up to 5 minutes. Use `timeout: 360000` on the Bash tool call. The source-grounding requirement in the build_prompt Review Instructions applies only to the prompt-fed reviewers above; CodeRabbit is a diff-only reviewer and never receives it. Treat its output as a diff observation, not a grounded plan-level verdict.
277
285
 
278
286
  ```bash
279
287
  coderabbit review --prompt-only 2>/dev/null > /tmp/gsd-review-coderabbit-{phase}.md
@@ -714,7 +722,7 @@ trimmed_reviewers: # only present if at least one reviewer was trimmed
714
722
 
715
723
  ## Consensus Summary
716
724
 
717
- {synthesize common concerns across all reviewers}
725
+ {synthesize common concerns across all reviewers. CodeRabbit is a diff-only reviewer (it never received the source-grounding prompt), so do not weight its verdict as a grounded plan review — fold in its diff findings, but base plan-level consensus on the prompt-fed reviewers.}
718
726
 
719
727
  ### Agreed Strengths
720
728
  {strengths mentioned by 2+ reviewers}
@@ -365,10 +365,15 @@ For each Requirement gathered so far, run the two-stage recall→precision pass:
365
365
  - `check_target` — the negative-test file path (for `node-test`), or the path to lint
366
366
  (for `lint-rule`).
367
367
  - `check_rule` — the eslint rule id (e.g. `local/no-source-grep`); `lint-rule` only.
368
- - `check_violation_fixture` (#1346) — path to a KNOWN-BAD subject the wired check is run
368
+ - `check_violation_fixture` (#1279) — path to a KNOWN-BAD subject the wired check is run
369
369
  against to **machine-prove fail-first**; rides BOTH kinds. Capture it to let the item green
370
370
  end-to-end with zero hand-authoring at verify time; for `node-test` the negative test should
371
371
  read its subject from the `GSD_PROHIB_SUBJECT` env var so the prover can inject this fixture.
372
+ - `check_clean_fixture` (#1346) — **optional** path to a KNOWN-CLEAN control subject. When
373
+ captured, the `node-test` prover also runs the check against it and requires GREEN — proving
374
+ the violation's RED is caused by the subject's *content*, not by `GSD_PROHIB_SUBJECT` merely
375
+ being set. Capture it for a stronger guarantee; omit it and the check still proves fail-first
376
+ on the violation alone (the content-causation residual stays documented for that case).
372
377
  This is a **SOFT capture (CHK-04): a `test`-tier prohibition WITHOUT a descriptor is still
373
378
  allowed** — if the author cannot yet name the wired check, leave the descriptor empty and
374
379
  proceed. It is NOT a hard authoring block; the item simply stays fail-closed/flagged
@@ -395,7 +400,7 @@ For each Requirement gathered so far, run the two-stage recall→precision pass:
395
400
  written (test or judgment tier); otherwise leave `unresolved`. **`--auto` NEVER auto-dismisses
396
401
  a prohibition** — a wrong dismissal is the exact silent failure this probe eliminates (PROB-06,
397
402
  the load-bearing safety property). On a `test`-tier auto-resolution, capture the `check_kind` /
398
- `check_target` / `check_rule` / `check_violation_fixture` descriptor **only when a wired check is unambiguous**; otherwise
403
+ `check_target` / `check_rule` / `check_violation_fixture` / `check_clean_fixture` descriptor **only when a wired check is unambiguous**; otherwise
399
404
  leave it empty — `--auto` NEVER fabricates a check path or fixture (a wrong locate is re-validated and
400
405
  fails closed at the producer, but a fabricated path is still noise to avoid). Log:
401
406
  `[auto] prohibitions: R resolved, U unresolved`.
@@ -408,7 +413,7 @@ Populate the `## Prohibitions` section of SPEC.md from the resolved prohibitions
408
413
  `resolved`/`test` row is a checkable negative acceptance criterion; `resolved`/`judgment`
409
414
  rows route to judgment review; `⚠ UNRESOLVED` rows are flagged as assumptions). A
410
415
  `resolved`/`test` row ALSO carries its captured `check_kind` / `check_target` / `check_rule` /
411
- `check_violation_fixture` descriptor when present (so the projection feeds `verify-phase`'s deterministic locate + machine-proof, #1278 + #1346);
416
+ `check_violation_fixture` / `check_clean_fixture` descriptor when present (so the projection feeds `verify-phase`'s deterministic locate + machine-proof + causation control, #1278 + #1279 + #1346);
412
417
  a `test` row with no captured descriptor is still valid — it stays fail-closed/flagged
413
418
  downstream rather than blocking authoring.
414
419
 
@@ -76,11 +76,11 @@ Aggregate all must_haves across plans for phase-level verification.
76
76
  gsd_run check prohibition-enforcement <request.json>
77
77
  ```
78
78
 
79
- where `<request.json>` carries `{ prohibition, check, mode }` — `check` being the wired mechanical-check descriptor `{ kind: 'node-test' | 'lint-rule', target, rule?, violationFixture, failFirst? }`, with `kind`/`target`/`rule`/`violationFixture` now sourced from the projected `check_*` scalars (not author/verifier invention — #1278 + #1346). For `node-test`, `target` (from `check_target`) is the negative-test file path; for `lint-rule`, `target` is the PATH to lint and `rule` (from `check_rule`) is the eslint rule id (e.g. `local/no-source-grep`) — both required (a lint-rule without `rule` is not a valid wired check). `violationFixture` (from `check_violation_fixture`) is the path to a KNOWN-BAD subject the producer runs the check against to **machine-prove fail-first** (for `node-test`, injected via the `GSD_PROHIB_SUBJECT` env convention — #1279); `failFirst` is a DEMOTED, non-authoritative hint kept only for backward route-JSON shape (no path greens on it alone — FF-08). The producer LOCATES the wired check from the projection, **machine-proves it is fail-first** by running it against the violation and confirming it goes RED, RUNS it for a genuine non-vacuous pass, builds `enforcementEvidence`, and emits the `dispositionForProhibition()` verdict (#1259 + #1278 + #1279, ADR-550 D5d). Fail-first is **machine-proven, not caller-attested** — absent a provable violation the producer fails closed, never falling back to attestation. Route the result by its typed fields:
79
+ where `<request.json>` carries `{ prohibition, check, mode }` — `check` being the wired mechanical-check descriptor `{ kind: 'node-test' | 'lint-rule', target, rule?, violationFixture, cleanFixture?, failFirst? }`, with `kind`/`target`/`rule`/`violationFixture`/`cleanFixture` now sourced from the projected `check_*` scalars (not author/verifier invention — #1278 + #1279 + #1346). For `node-test`, `target` (from `check_target`) is the negative-test file path; for `lint-rule`, `target` is the PATH to lint and `rule` (from `check_rule`) is the eslint rule id (e.g. `local/no-source-grep`) — both required (a lint-rule without `rule` is not a valid wired check). `violationFixture` (from `check_violation_fixture`) is the path to a KNOWN-BAD subject the producer runs the check against to **machine-prove fail-first** (for `node-test`, injected via the `GSD_PROHIB_SUBJECT` env convention — #1279); the optional `cleanFixture` (from `check_clean_fixture`) is a KNOWN-CLEAN control subject the `node-test` prover ALSO requires to stay GREEN, proving the RED is content-caused (#1346); `failFirst` is a DEMOTED, non-authoritative hint kept only for backward route-JSON shape (no path greens on it alone — FF-08). The producer LOCATES the wired check from the projection, **machine-proves it is fail-first** by running it against the violation and confirming it goes RED, RUNS it for a genuine non-vacuous pass, builds `enforcementEvidence`, and emits the `dispositionForProhibition()` verdict (#1259 + #1278 + #1279, ADR-550 D5d). Fail-first is **machine-proven, not caller-attested** — absent a provable violation the producer fails closed, never falling back to attestation. Route the result by its typed fields:
80
80
  - **`status: 'green'`, `flagged: false`** (a genuinely-passing wired negative test / lint rule, `located: true`, non-empty `evidence`) → the item is satisfiable → it can reach **passed**.
81
81
  - **missing, non-attested, or genuinely-non-passing check** (`located: false` OR `status: 'unverified'`, `flagged: true`) → **hard-gate**: disposes flagged-unverified, NEVER green, routing to `gaps_found` in BOTH interactive and autonomous modes (a failing mechanical check blocks even AFK; ADR-550 D4 / D3). The deterministic fail-closed default backing every miss/fail is `dispositionForProhibition()` in probe-core (`status: 'unverified'`, `flagged: true` on empty `enforcementEvidence`).
82
82
 
83
- > **Descriptor source — deterministic locate + machine-proof compose (#1278 + #1346, DELIVERED).** The `check` descriptor's `{ kind, target, rule, violationFixture }` is now sourced **deterministically from the projected `check_kind` / `check_target` / `check_rule` / `check_violation_fixture` scalars** on the `must_haves.prohibitions` item (authored at `/gsd:spec-phase`, projected by `projectProhibitions`, read back via the `descriptorFromProjection` adapter). So both halves close with **zero manual descriptor authoring** — the verifier neither invents the locate (#1278) nor hand-supplies the violation fixture (#1346): a prohibition authored with all four scalars machine-proves fail-first and greens end-to-end through the projection alone (removing the spoofable invent-at-verify-time surface; ADR-857 §147 exogenous grading). **Fail-closed is preserved:** an item with NO projected descriptor, a PARTIAL one (e.g. a `lint-rule` missing `check_rule`), OR a descriptor with **no `check_violation_fixture`** makes `descriptorFromProjection` return `null` / an under-specified or fixture-less descriptor, which falls through to the producer's fail-closed paths (`located: false`, or located-but-unprovable) → flagged-unverified, NEVER green, in BOTH modes. `failFirst` is demoted and greens nothing on its own (#1279, FF-08). Residual (tracked **#1346**): the node-test proof confirms the fixture exists and the check goes RED, but cannot generically prove the red was *caused by* the subject's content vs the env merely being set.
83
+ > **Descriptor source — deterministic locate + machine-proof compose (#1278 + #1346, DELIVERED).** The `check` descriptor's `{ kind, target, rule, violationFixture }` is now sourced **deterministically from the projected `check_kind` / `check_target` / `check_rule` / `check_violation_fixture` scalars** on the `must_haves.prohibitions` item (authored at `/gsd:spec-phase`, projected by `projectProhibitions`, read back via the `descriptorFromProjection` adapter). So both halves close with **zero manual descriptor authoring** — the verifier neither invents the locate (#1278) nor hand-supplies the violation fixture (#1346): a prohibition authored with all four scalars machine-proves fail-first and greens end-to-end through the projection alone (removing the spoofable invent-at-verify-time surface; ADR-857 §147 exogenous grading). **Fail-closed is preserved:** an item with NO projected descriptor, a PARTIAL one (e.g. a `lint-rule` missing `check_rule`), OR a descriptor with **no `check_violation_fixture`** makes `descriptorFromProjection` return `null` / an under-specified or fixture-less descriptor, which falls through to the producer's fail-closed paths (`located: false`, or located-but-unprovable) → flagged-unverified, NEVER green, in BOTH modes. `failFirst` is demoted and greens nothing on its own (#1279, FF-08). Causation (**#1346**): supplying `check_clean_fixture` adds an opt-in control — the `node-test` prover also requires GREEN on a known-clean subject, proving the RED is content-caused; with no clean fixture that one residual case (a deceptive test reding merely because the env var is set) stays a documented constraint, an author opting into the stronger proof by wiring a clean control.
84
84
 
85
85
  **Option B: Use Success Criteria from ROADMAP.md**
86
86
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@opengsd/gsd-core",
3
- "version": "1.5.0",
3
+ "version": "1.6.0-rc.2",
4
4
  "description": "GSD Core is a meta-prompting, context engineering, and spec-driven development system for AI coding agents.",
5
5
  "bin": {
6
6
  "gsd-core": "bin/install.js",
@@ -86,13 +86,13 @@
86
86
  "gen:capability-registry": "node scripts/gen-capability-registry.cjs --write",
87
87
  "prepack": "npm run build:lib",
88
88
  "prepare": "npm run build:lib",
89
- "version": "node scripts/sync-manifest-versions.cjs --stage",
89
+ "version": "node scripts/sync-manifest-versions.cjs --stage && node scripts/gen-capability-registry.cjs --write && git add gsd-core/bin/lib/capability-registry.cjs",
90
90
  "prepublishOnly": "npm run build:lib && npm run build:hooks",
91
91
  "pretest": "npm run build:lib && npm run lint:skill-deps",
92
92
  "pretest:coverage": "npm run build:lib && npm run lint:skill-deps",
93
93
  "lint": "eslint . --cache --cache-location node_modules/.cache/eslint/",
94
94
  "lint:fix": "eslint . --fix",
95
- "lint:ci": "npm run lint && npm run lint:skill-deps && node scripts/lint-test-file-count.cjs && node scripts/lint-command-contract.cjs && node scripts/lint-pr-check-project-dir.cjs && npm run lint:legacy-name && node scripts/lint-regression-test-names.cjs && node scripts/lint-windows-test-portability.cjs && node scripts/lint-allow-test-rule-refs.cjs",
95
+ "lint:ci": "npm run lint && npm run lint:skill-deps && node scripts/lint-test-file-count.cjs && node scripts/lint-command-contract.cjs && node scripts/lint-pr-check-project-dir.cjs && npm run lint:legacy-name && node scripts/lint-regression-test-names.cjs && node scripts/lint-windows-test-portability.cjs && node scripts/lint-allow-test-rule-refs.cjs && node scripts/lint-resolution-provenance.cjs",
96
96
  "lint:allow-test-rule-refs": "node scripts/lint-allow-test-rule-refs.cjs",
97
97
  "lint:windows-test-portability": "node scripts/lint-windows-test-portability.cjs",
98
98
  "lint:regression-names": "node scripts/lint-regression-test-names.cjs",
@@ -120,5 +120,8 @@
120
120
  "test:coverage:all": "npm run test:coverage",
121
121
  "test:mutation": "stryker run",
122
122
  "test:mutation:since": "stryker run --incremental --since origin/next"
123
+ },
124
+ "allowScripts": {
125
+ "fallow@2.70.0": true
123
126
  }
124
127
  }