@opengsd/gsd-core 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (455) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.claude-plugin/plugin.json +1 -1
  3. package/.opencode/plugins/gsd-core.js +12 -0
  4. package/agents/gsd-advisor-researcher.compact.md +85 -0
  5. package/agents/gsd-ai-researcher.compact.md +96 -0
  6. package/agents/gsd-assumptions-analyzer.compact.md +81 -0
  7. package/agents/gsd-code-fixer.compact.md +458 -0
  8. package/agents/gsd-code-fixer.md +5 -5
  9. package/agents/gsd-code-reviewer.compact.md +269 -0
  10. package/agents/gsd-code-reviewer.md +15 -3
  11. package/agents/gsd-codebase-mapper.compact.md +760 -0
  12. package/agents/gsd-debug-session-manager.compact.md +345 -0
  13. package/agents/gsd-doc-classifier.compact.md +192 -0
  14. package/agents/gsd-doc-synthesizer.compact.md +200 -0
  15. package/agents/gsd-doc-verifier.compact.md +143 -0
  16. package/agents/gsd-doc-writer.compact.md +440 -0
  17. package/agents/gsd-dom-verifier.compact.md +138 -0
  18. package/agents/gsd-domain-researcher.compact.md +141 -0
  19. package/agents/gsd-eval-auditor.compact.md +160 -0
  20. package/agents/gsd-eval-planner.compact.md +137 -0
  21. package/agents/gsd-executor.md +63 -35
  22. package/agents/gsd-framework-selector.compact.md +82 -0
  23. package/agents/gsd-integration-checker.compact.md +245 -0
  24. package/agents/gsd-intel-updater.compact.md +226 -0
  25. package/agents/gsd-mempalace-curator.compact.md +45 -0
  26. package/agents/gsd-nyquist-auditor.compact.md +179 -0
  27. package/agents/gsd-pattern-mapper.compact.md +275 -0
  28. package/agents/gsd-plan-checker.md +76 -57
  29. package/agents/gsd-planner.md +14 -0
  30. package/agents/gsd-project-researcher.compact.md +587 -0
  31. package/agents/gsd-research-synthesizer.compact.md +212 -0
  32. package/agents/gsd-roadmapper.compact.md +454 -0
  33. package/agents/gsd-roadmapper.md +13 -0
  34. package/agents/gsd-security-auditor.compact.md +162 -0
  35. package/agents/gsd-ui-auditor.compact.md +404 -0
  36. package/agents/gsd-ui-checker.compact.md +277 -0
  37. package/agents/gsd-ui-checker.md +19 -3
  38. package/agents/gsd-ui-researcher.compact.md +282 -0
  39. package/agents/gsd-ui-researcher.md +29 -0
  40. package/agents/gsd-user-profiler.compact.md +108 -0
  41. package/agents/gsd-verifier.md +23 -1
  42. package/bin/install.js +444 -134
  43. package/commands/gsd/cleanup.md +1 -0
  44. package/commands/gsd/code-review.md +2 -1
  45. package/commands/gsd/complete-milestone.md +1 -0
  46. package/commands/gsd/config.md +1 -0
  47. package/commands/gsd/debug.md +1 -0
  48. package/commands/gsd/execute-phase.md +1 -1
  49. package/commands/gsd/graphify.md +1 -0
  50. package/commands/gsd/health.md +1 -0
  51. package/commands/gsd/mempalace-capture.md +1 -0
  52. package/commands/gsd/mempalace-recall.md +1 -0
  53. package/commands/gsd/new-milestone.md +1 -0
  54. package/commands/gsd/new-project.md +1 -0
  55. package/commands/gsd/next.md +1 -0
  56. package/commands/gsd/ns-workflow.md +2 -1
  57. package/commands/gsd/pause-work.md +1 -0
  58. package/commands/gsd/phase.md +2 -1
  59. package/commands/gsd/pr-branch.md +1 -0
  60. package/commands/gsd/quick-batch.md +105 -0
  61. package/commands/gsd/resume-work.md +1 -0
  62. package/commands/gsd/review-backlog.md +1 -0
  63. package/commands/gsd/settings.md +2 -1
  64. package/commands/gsd/stats.md +1 -0
  65. package/commands/gsd/surface.md +18 -8
  66. package/commands/gsd/thread.md +1 -0
  67. package/commands/gsd/workspace.md +1 -0
  68. package/commands/gsd/workstreams.md +1 -0
  69. package/gsd-core/bin/check-latest-version.cjs +8 -3
  70. package/gsd-core/bin/gsd-tools.cjs +532 -174
  71. package/gsd-core/bin/lib/adr-parser.cjs +1 -1
  72. package/gsd-core/bin/lib/artifacts.cjs +2 -1
  73. package/gsd-core/bin/lib/audit.cjs +39 -22
  74. package/gsd-core/bin/lib/broken-windows.cjs +168 -49
  75. package/gsd-core/bin/lib/capability-activation.cjs +27 -0
  76. package/gsd-core/bin/lib/capability-lifecycle.cjs +10 -6
  77. package/gsd-core/bin/lib/capability-loader.cjs +135 -1
  78. package/gsd-core/bin/lib/capability-registry.cjs +528 -116
  79. package/gsd-core/bin/lib/capability-source.cjs +19 -2
  80. package/gsd-core/bin/lib/capability-state.cjs +7 -1
  81. package/gsd-core/bin/lib/capability-validator.cjs +134 -5
  82. package/gsd-core/bin/lib/capability-writer.cjs +14 -4
  83. package/gsd-core/bin/lib/check-command-router.cjs +198 -38
  84. package/gsd-core/bin/lib/claude-orchestration.cjs +10 -25
  85. package/gsd-core/bin/lib/clusters.cjs +1 -0
  86. package/gsd-core/bin/lib/code-review-depth.cjs +2 -2
  87. package/gsd-core/bin/lib/command-aliases.cjs +16 -0
  88. package/gsd-core/bin/lib/commands.cjs +981 -79
  89. package/gsd-core/bin/lib/config-loader.cjs +4 -0
  90. package/gsd-core/bin/lib/config.cjs +153 -38
  91. package/gsd-core/bin/lib/core-utils.cjs +34 -7
  92. package/gsd-core/bin/lib/coverage.cjs +1 -1
  93. package/gsd-core/bin/lib/decisions.cjs +343 -28
  94. package/gsd-core/bin/lib/edge-probe.cjs +14 -1
  95. package/gsd-core/bin/lib/external-descriptor-trust.cjs +29 -14
  96. package/gsd-core/bin/lib/file-overlap-partitioner.cjs +74 -0
  97. package/gsd-core/bin/lib/frontmatter.cjs +137 -23
  98. package/gsd-core/bin/lib/gap-checker.cjs +22 -13
  99. package/gsd-core/bin/lib/git-base-branch.cjs +10 -2
  100. package/gsd-core/bin/lib/gsd2-import.cjs +1 -2
  101. package/gsd-core/bin/lib/health-diagnostic-rules/phase-structure.cjs +8 -2
  102. package/gsd-core/bin/lib/health-diagnostic-rules/roadmap-disk-consistency.cjs +54 -11
  103. package/gsd-core/bin/lib/health-diagnostic-rules/state-consistency.cjs +87 -23
  104. package/gsd-core/bin/lib/health-diagnostic-rules/worktree-health.cjs +1 -1
  105. package/gsd-core/bin/lib/host-integration.cjs +57 -5
  106. package/gsd-core/bin/lib/init-command-router.cjs +14 -0
  107. package/gsd-core/bin/lib/init.cjs +539 -60
  108. package/gsd-core/bin/lib/install-engine.cjs +199 -14
  109. package/gsd-core/bin/lib/install-model-override-resolver.cjs +45 -0
  110. package/gsd-core/bin/lib/install-profiles.cjs +36 -14
  111. package/gsd-core/bin/lib/installer-migration-report.cjs +1 -0
  112. package/gsd-core/bin/lib/installer-migrations.cjs +33 -4
  113. package/gsd-core/bin/lib/io.cjs +35 -0
  114. package/gsd-core/bin/lib/loop-resolver.cjs +64 -39
  115. package/gsd-core/bin/lib/markdown-table.cjs +123 -0
  116. package/gsd-core/bin/lib/mcp-catalog.cjs +2 -2
  117. package/gsd-core/bin/lib/milestone.cjs +41 -10
  118. package/gsd-core/bin/lib/model-resolver.cjs +101 -10
  119. package/gsd-core/bin/lib/phase-command-router.cjs +20 -7
  120. package/gsd-core/bin/lib/phase-id.cjs +412 -31
  121. package/gsd-core/bin/lib/phase-lifecycle.cjs +61 -0
  122. package/gsd-core/bin/lib/phase.cjs +941 -98
  123. package/gsd-core/bin/lib/plan-document.cjs +10 -0
  124. package/gsd-core/bin/lib/planning-inspect.cjs +34 -18
  125. package/gsd-core/bin/lib/planning-snapshot.cjs +206 -30
  126. package/gsd-core/bin/lib/planning-workspace.cjs +153 -29
  127. package/gsd-core/bin/lib/pristine-baseline.cjs +182 -0
  128. package/gsd-core/bin/lib/prohibition-enforcement.cjs +91 -4
  129. package/gsd-core/bin/lib/quick-batch-command-router.cjs +285 -0
  130. package/gsd-core/bin/lib/quick-batch-dispatch.cjs +250 -0
  131. package/gsd-core/bin/lib/quick-batch.cjs +840 -0
  132. package/gsd-core/bin/lib/refactor-trigger-command-router.cjs +61 -2
  133. package/gsd-core/bin/lib/research-store.cjs +11 -12
  134. package/gsd-core/bin/lib/review-lane-descriptor.cjs +53 -5
  135. package/gsd-core/bin/lib/review-lane-invocation.cjs +96 -1
  136. package/gsd-core/bin/lib/review-lane-runner.cjs +136 -10
  137. package/gsd-core/bin/lib/reviewer-step-dispatch.cjs +337 -0
  138. package/gsd-core/bin/lib/roadmap-parser.cjs +555 -41
  139. package/gsd-core/bin/lib/roadmap.cjs +292 -69
  140. package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +260 -43
  141. package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +28 -20
  142. package/gsd-core/bin/lib/runtime-artifact-layout.cjs +294 -108
  143. package/gsd-core/bin/lib/runtime-hooks-surface.cjs +408 -47
  144. package/gsd-core/bin/lib/security.cjs +126 -7
  145. package/gsd-core/bin/lib/shell-command-projection.cjs +4 -0
  146. package/gsd-core/bin/lib/smart-entry.cjs +7 -9
  147. package/gsd-core/bin/lib/state-document.cjs +159 -32
  148. package/gsd-core/bin/lib/state-md-schema.cjs +44 -27
  149. package/gsd-core/bin/lib/state-transition.cjs +465 -62
  150. package/gsd-core/bin/lib/state.cjs +906 -151
  151. package/gsd-core/bin/lib/surface.cjs +83 -10
  152. package/gsd-core/bin/lib/task-command-router.cjs +12 -6
  153. package/gsd-core/bin/lib/tdd-red-evidence.cjs +133 -0
  154. package/gsd-core/bin/lib/uat.cjs +1420 -516
  155. package/gsd-core/bin/lib/update-context.cjs +36 -26
  156. package/gsd-core/bin/lib/validate.cjs +230 -12
  157. package/gsd-core/bin/lib/vendor/js-yaml.cjs +11 -3
  158. package/gsd-core/bin/lib/verification-command-router.cjs +2 -1
  159. package/gsd-core/bin/lib/verification.cjs +316 -23
  160. package/gsd-core/bin/lib/verify-command-grounding.cjs +1 -1
  161. package/gsd-core/bin/lib/verify-command-router.cjs +1 -0
  162. package/gsd-core/bin/lib/verify.cjs +531 -36
  163. package/gsd-core/bin/lib/workstream-inventory.cjs +21 -2
  164. package/gsd-core/bin/lib/worktree-safety.cjs +21 -7
  165. package/gsd-core/bin/shared/config-defaults.manifest.json +1 -0
  166. package/gsd-core/bin/shared/config-schema.manifest.json +13 -0
  167. package/gsd-core/bin/verify-reapply-patches.cjs +507 -81
  168. package/gsd-core/references/agent-contracts.md +3 -3
  169. package/gsd-core/references/compact-content-gate.md +66 -0
  170. package/gsd-core/references/edge-probe.md +17 -13
  171. package/gsd-core/references/execute-mvp-tdd.md +18 -16
  172. package/gsd-core/references/execute-phase-response-language.md +6 -0
  173. package/gsd-core/references/executor-examples.md +42 -0
  174. package/gsd-core/references/few-shot-examples/plan-checker.md +15 -15
  175. package/gsd-core/references/loop-hook-dispatch.md +18 -0
  176. package/gsd-core/references/model-profiles.md +12 -3
  177. package/gsd-core/references/mvp-concepts.md +2 -2
  178. package/gsd-core/references/plan-checker-examples.md +41 -0
  179. package/gsd-core/references/planner-antipatterns.md +25 -0
  180. package/gsd-core/references/planner-chunked.md +5 -1
  181. package/gsd-core/references/planner-coupling.md +42 -0
  182. package/gsd-core/references/planner-quick-batch.md +71 -0
  183. package/gsd-core/references/planner-reviews.md +47 -0
  184. package/gsd-core/references/planner-revision.md +75 -2
  185. package/gsd-core/references/planning-config.md +5 -1
  186. package/gsd-core/references/response-language-directive.md +9 -0
  187. package/gsd-core/references/revision-loop.md +118 -11
  188. package/gsd-core/references/tdd.md +17 -9
  189. package/gsd-core/references/thinking-models-planning.md +18 -2
  190. package/gsd-core/references/verification-patterns.md +17 -4
  191. package/gsd-core/references/verifier-evidence-gate.md +160 -0
  192. package/gsd-core/references/worktree-path-safety.md +112 -2
  193. package/gsd-core/templates/README.md +7 -1
  194. package/gsd-core/templates/phase-prompt.md +4 -0
  195. package/gsd-core/templates/state.md +6 -3
  196. package/gsd-core/templates/summary.compact.md +212 -0
  197. package/gsd-core/templates/user-setup.compact.md +199 -0
  198. package/gsd-core/templates/user-setup.md +0 -9
  199. package/gsd-core/templates/verification-report.md +5 -0
  200. package/gsd-core/workflows/add-backlog.md +2 -0
  201. package/gsd-core/workflows/add-phase.md +2 -0
  202. package/gsd-core/workflows/add-tests.md +1 -1
  203. package/gsd-core/workflows/add-todo.md +4 -3
  204. package/gsd-core/workflows/ai-integration-phase.md +1 -1
  205. package/gsd-core/workflows/analyze-dependencies.md +2 -0
  206. package/gsd-core/workflows/audit-fix.md +2 -0
  207. package/gsd-core/workflows/audit-milestone.md +2 -0
  208. package/gsd-core/workflows/audit-uat.md +2 -0
  209. package/gsd-core/workflows/autonomous.md +15 -10
  210. package/gsd-core/workflows/check-todos.md +5 -3
  211. package/gsd-core/workflows/cleanup.md +4 -2
  212. package/gsd-core/workflows/code-review/steps/structural-pre-pass.md +22 -13
  213. package/gsd-core/workflows/code-review-fix.md +5 -3
  214. package/gsd-core/workflows/code-review.md +211 -43
  215. package/gsd-core/workflows/complete-milestone/detail/elaboration.md +274 -0
  216. package/gsd-core/workflows/complete-milestone.md +40 -254
  217. package/gsd-core/workflows/debug.md +1 -1
  218. package/gsd-core/workflows/diagnose-issues.md +5 -1
  219. package/gsd-core/workflows/discuss-phase/modes/advisor.md +2 -0
  220. package/gsd-core/workflows/discuss-phase/modes/all.md +2 -0
  221. package/gsd-core/workflows/discuss-phase/modes/analyze.md +2 -0
  222. package/gsd-core/workflows/discuss-phase/modes/auto.md +2 -0
  223. package/gsd-core/workflows/discuss-phase/modes/batch.md +2 -0
  224. package/gsd-core/workflows/discuss-phase/modes/chain.md +2 -0
  225. package/gsd-core/workflows/discuss-phase/modes/default.md +2 -0
  226. package/gsd-core/workflows/discuss-phase/modes/power.md +2 -0
  227. package/gsd-core/workflows/discuss-phase/modes/text.md +2 -0
  228. package/gsd-core/workflows/discuss-phase/templates/context.md +2 -0
  229. package/gsd-core/workflows/discuss-phase/templates/discussion-log.md +2 -0
  230. package/gsd-core/workflows/discuss-phase-assumptions.md +1 -1
  231. package/gsd-core/workflows/discuss-phase-power.md +2 -0
  232. package/gsd-core/workflows/discuss-phase.md +1 -1
  233. package/gsd-core/workflows/do.md +43 -13
  234. package/gsd-core/workflows/docs-update/detail/elaboration.md +179 -0
  235. package/gsd-core/workflows/docs-update.md +15 -156
  236. package/gsd-core/workflows/edit-phase.md +2 -0
  237. package/gsd-core/workflows/eval-review.md +1 -1
  238. package/gsd-core/workflows/execute-phase/detail/elaboration.md +124 -0
  239. package/gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md +20 -3
  240. package/gsd-core/workflows/execute-phase/steps/completion-reconciliation.md +56 -0
  241. package/gsd-core/workflows/execute-phase/steps/executor-isolation-dispatch.md +24 -3
  242. package/gsd-core/workflows/execute-phase/steps/executor-progress-policy.md +43 -0
  243. package/gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md +8 -2
  244. package/gsd-core/workflows/execute-phase/steps/regression-gate-run.md +2 -0
  245. package/gsd-core/workflows/execute-phase/steps/sequential-root-pin.md +35 -0
  246. package/gsd-core/workflows/execute-phase/steps/tdd-applicability-resolution.md +25 -0
  247. package/gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md +2 -0
  248. package/gsd-core/workflows/execute-phase.md +78 -159
  249. package/gsd-core/workflows/execute-plan.md +28 -15
  250. package/gsd-core/workflows/explore.md +2 -0
  251. package/gsd-core/workflows/extract-learnings.md +2 -0
  252. package/gsd-core/workflows/fast.md +6 -0
  253. package/gsd-core/workflows/forensics.md +2 -0
  254. package/gsd-core/workflows/graduation.md +1 -1
  255. package/gsd-core/workflows/health.md +1 -1
  256. package/gsd-core/workflows/help/modes/brief.md +2 -0
  257. package/gsd-core/workflows/help/modes/default.md +2 -0
  258. package/gsd-core/workflows/help/modes/full.compact.md +398 -0
  259. package/gsd-core/workflows/help/modes/full.md +12 -0
  260. package/gsd-core/workflows/help/modes/topic.md +2 -0
  261. package/gsd-core/workflows/help.md +3 -1
  262. package/gsd-core/workflows/import.md +3 -3
  263. package/gsd-core/workflows/inbox.md +1 -1
  264. package/gsd-core/workflows/ingest-docs.md +1 -1
  265. package/gsd-core/workflows/insert-phase.md +2 -0
  266. package/gsd-core/workflows/list-phase-assumptions.md +2 -0
  267. package/gsd-core/workflows/list-seeds.md +2 -0
  268. package/gsd-core/workflows/list-workspaces.md +2 -0
  269. package/gsd-core/workflows/manager.md +3 -3
  270. package/gsd-core/workflows/map-codebase.md +52 -3
  271. package/gsd-core/workflows/milestone-summary.md +2 -0
  272. package/gsd-core/workflows/mvp-phase.md +1 -1
  273. package/gsd-core/workflows/new-milestone.md +55 -13
  274. package/gsd-core/workflows/new-project/detail/elaboration.md +216 -0
  275. package/gsd-core/workflows/new-project.md +37 -205
  276. package/gsd-core/workflows/new-workspace.md +1 -1
  277. package/gsd-core/workflows/next.md +2 -0
  278. package/gsd-core/workflows/node-repair.md +2 -0
  279. package/gsd-core/workflows/note.md +2 -0
  280. package/gsd-core/workflows/onboard.md +1 -1
  281. package/gsd-core/workflows/pause-work.md +19 -4
  282. package/gsd-core/workflows/plan-phase/detail/elaboration.md +209 -0
  283. package/gsd-core/workflows/plan-phase/steps/chunked-planning-mode.md +100 -18
  284. package/gsd-core/workflows/plan-phase/steps/prd-express-path.md +2 -0
  285. package/gsd-core/workflows/plan-phase/steps/stall-detection-helpers.md +9 -0
  286. package/gsd-core/workflows/plan-phase.md +144 -185
  287. package/gsd-core/workflows/plan-review-convergence.md +102 -10
  288. package/gsd-core/workflows/plant-seed.md +1 -1
  289. package/gsd-core/workflows/pr-branch.md +30 -10
  290. package/gsd-core/workflows/profile-user.md +1 -1
  291. package/gsd-core/workflows/progress/steps/forensic-audit.md +1 -1
  292. package/gsd-core/workflows/progress.md +25 -3
  293. package/gsd-core/workflows/quick/steps/plan-checker-loop.md +37 -2
  294. package/gsd-core/workflows/quick/steps/research-phase.md +3 -3
  295. package/gsd-core/workflows/quick-batch/steps/batch-init.md +55 -0
  296. package/gsd-core/workflows/quick-batch/steps/completion.md +65 -0
  297. package/gsd-core/workflows/quick-batch/steps/merge-wave.md +100 -0
  298. package/gsd-core/workflows/quick-batch/steps/plan-checker-loop.md +147 -0
  299. package/gsd-core/workflows/quick-batch/steps/planner-wave.md +158 -0
  300. package/gsd-core/workflows/quick-batch/steps/research-phase.md +95 -0
  301. package/gsd-core/workflows/quick-batch/steps/resume-mode.md +49 -0
  302. package/gsd-core/workflows/quick-batch/steps/verification-wave.md +73 -0
  303. package/gsd-core/workflows/quick-batch/steps/worktree-dispatch.md +169 -0
  304. package/gsd-core/workflows/quick-batch.md +203 -0
  305. package/gsd-core/workflows/quick.md +21 -4
  306. package/gsd-core/workflows/reapply-patches.md +79 -3
  307. package/gsd-core/workflows/remove-phase.md +2 -0
  308. package/gsd-core/workflows/remove-workspace.md +1 -1
  309. package/gsd-core/workflows/resume-project.md +6 -2
  310. package/gsd-core/workflows/review.md +215 -10
  311. package/gsd-core/workflows/scan.md +2 -0
  312. package/gsd-core/workflows/section-manifest.json +12 -0
  313. package/gsd-core/workflows/secure-phase.md +1 -1
  314. package/gsd-core/workflows/session-report.md +2 -0
  315. package/gsd-core/workflows/settings-advanced.md +2 -0
  316. package/gsd-core/workflows/settings-integrations.md +9 -8
  317. package/gsd-core/workflows/settings.md +19 -6
  318. package/gsd-core/workflows/ship.md +10 -10
  319. package/gsd-core/workflows/sketch-wrap-up.md +2 -0
  320. package/gsd-core/workflows/sketch.md +1 -1
  321. package/gsd-core/workflows/smart-entry.md +1 -1
  322. package/gsd-core/workflows/spec-phase.md +24 -19
  323. package/gsd-core/workflows/spike-wrap-up.md +2 -0
  324. package/gsd-core/workflows/spike.md +1 -1
  325. package/gsd-core/workflows/stats.md +2 -0
  326. package/gsd-core/workflows/sync-skills.md +12 -4
  327. package/gsd-core/workflows/thread.md +2 -0
  328. package/gsd-core/workflows/transition.md +2 -0
  329. package/gsd-core/workflows/ui-phase.md +26 -5
  330. package/gsd-core/workflows/ui-review.md +1 -1
  331. package/gsd-core/workflows/ultraplan-phase.md +2 -0
  332. package/gsd-core/workflows/undo.md +1 -1
  333. package/gsd-core/workflows/update.md +48 -43
  334. package/gsd-core/workflows/validate-phase.md +1 -1
  335. package/gsd-core/workflows/verify-work/detail/elaboration.md +230 -0
  336. package/gsd-core/workflows/verify-work.md +68 -182
  337. package/hooks/dist/gsd-agent-isolation-guard.js +42 -16
  338. package/hooks/dist/gsd-check-update-worker.js +19 -2
  339. package/hooks/dist/gsd-context-monitor.js +371 -27
  340. package/hooks/dist/gsd-cursor-subagent-start.js +34 -14
  341. package/hooks/dist/gsd-node-runner.sh +1 -0
  342. package/hooks/dist/gsd-prompt-guard.js +30 -5
  343. package/hooks/dist/gsd-read-guard.js +2 -0
  344. package/hooks/dist/gsd-read-injection-scanner.js +5 -5
  345. package/hooks/dist/gsd-secret-read-guard.js +1105 -0
  346. package/hooks/dist/gsd-statusline.js +18 -10
  347. package/hooks/dist/gsd-validate-commit.sh +474 -7
  348. package/hooks/dist/gsd-workflow-guard.js +2 -1
  349. package/hooks/dist/gsd-worktree-path-guard.js +25 -14
  350. package/hooks/dist/gsd-write-guard.js +46 -1
  351. package/hooks/dist/lib/dispatch-identity.js +187 -0
  352. package/hooks/dist/lib/filename-classification.js +64 -0
  353. package/hooks/dist/lib/git-cmd.js +210 -1
  354. package/hooks/dist/lib/injection-patterns.js +36 -6
  355. package/hooks/dist/lib/isolation-deny-reason.js +53 -1
  356. package/hooks/dist/lib/isolation-sentinel.js +58 -19
  357. package/hooks/dist/managed-hooks-registry.cjs +1 -0
  358. package/hooks/gsd-agent-isolation-guard.js +42 -16
  359. package/hooks/gsd-check-update-worker.js +19 -2
  360. package/hooks/gsd-context-monitor.js +371 -27
  361. package/hooks/gsd-cursor-subagent-start.js +34 -14
  362. package/hooks/gsd-node-runner.sh +1 -0
  363. package/hooks/gsd-prompt-guard.js +30 -5
  364. package/hooks/gsd-read-guard.js +2 -0
  365. package/hooks/gsd-read-injection-scanner.js +5 -5
  366. package/hooks/gsd-secret-read-guard.js +1105 -0
  367. package/hooks/gsd-statusline.js +18 -10
  368. package/hooks/gsd-validate-commit.sh +474 -7
  369. package/hooks/gsd-workflow-guard.js +2 -1
  370. package/hooks/gsd-worktree-path-guard.js +25 -14
  371. package/hooks/gsd-write-guard.js +46 -1
  372. package/hooks/hooks.json +6 -0
  373. package/hooks/lib/dispatch-identity.js +187 -0
  374. package/hooks/lib/filename-classification.js +64 -0
  375. package/hooks/lib/git-cmd.js +210 -1
  376. package/hooks/lib/injection-patterns.js +36 -6
  377. package/hooks/lib/isolation-deny-reason.js +53 -1
  378. package/hooks/lib/isolation-sentinel.js +58 -19
  379. package/hooks/managed-hooks-registry.cjs +1 -0
  380. package/package.json +13 -9
  381. package/scripts/benchmark-compact-content-variants.cjs +298 -0
  382. package/scripts/benchmark-compact-content.cjs +368 -0
  383. package/scripts/build-hooks.js +11 -4
  384. package/scripts/check-contract-drift.cjs +4 -1
  385. package/scripts/check-env.cjs +36 -8
  386. package/scripts/check-glossary-refs.cjs +25 -21
  387. package/scripts/ci-next-health.cjs +271 -0
  388. package/scripts/ci-prepare-test-scope.cjs +7 -7
  389. package/scripts/ci-test-scope.cjs +133 -20
  390. package/scripts/ci-timeout-report.cjs +1 -1
  391. package/scripts/diff-touches-shipped-paths.cjs +1 -1
  392. package/scripts/docs-guard-registry.cjs +17 -2
  393. package/scripts/gen-adr-index.cjs +8 -2
  394. package/scripts/gen-inventory-manifest.cjs +12 -0
  395. package/scripts/gen-loop-host-contract.cjs +67 -15
  396. package/scripts/gen-platform-conformance-tier.cjs +557 -0
  397. package/scripts/lib/drift-scan.cjs +1 -1
  398. package/scripts/lib/macos-conformance-tier.generated.cjs +210 -0
  399. package/scripts/lib/npm-version-check-diagnosis.cjs +59 -0
  400. package/scripts/lib/platform-conformance-tier.generated.cjs +276 -0
  401. package/scripts/lib/shellcheck-fetch.cjs +247 -0
  402. package/scripts/lib/suite-detection.cjs +32 -0
  403. package/scripts/lint-allow-test-rule-refs.allowlist.json +0 -6
  404. package/scripts/lint-allow-test-rule-refs.effective-ceiling.json +1 -1
  405. package/scripts/lint-allow-test-rule-refs.unverified-ceiling.json +1 -1
  406. package/scripts/lint-allowed-tools-parity.cjs +221 -0
  407. package/scripts/lint-docs-guard-registration.exempt-baseline.cjs +24 -2
  408. package/scripts/lint-phase-enumeration-drift.cjs +24 -6
  409. package/scripts/lint-phase-id-drift.cjs +465 -15
  410. package/scripts/lint-portable-grep.cjs +176 -0
  411. package/scripts/lint-response-language-coverage.cjs +530 -0
  412. package/scripts/lint-source-test-name-collision.cjs +1 -1
  413. package/scripts/lint-test-file-count.allowlist.json +4 -1
  414. package/scripts/lint-vendored-deps.cjs +128 -17
  415. package/scripts/lint-workflow-shellcheck-baseline.json +1112 -0
  416. package/scripts/lint-workflow-shellcheck.cjs +614 -0
  417. package/scripts/npm-audit-baseline.cjs +376 -0
  418. package/scripts/prompt-injection-scan.sh +22 -0
  419. package/scripts/require-issue-link-policy.cjs +16 -1
  420. package/scripts/workflow-size.cjs +139 -0
  421. package/skills/gsd-cleanup/SKILL.md +1 -0
  422. package/skills/gsd-code-review/SKILL.md +2 -1
  423. package/skills/gsd-complete-milestone/SKILL.md +1 -0
  424. package/skills/gsd-config/SKILL.md +1 -0
  425. package/skills/gsd-debug/SKILL.md +1 -0
  426. package/skills/gsd-execute-phase/SKILL.md +1 -1
  427. package/skills/gsd-graphify/SKILL.md +1 -0
  428. package/skills/gsd-health/SKILL.md +1 -0
  429. package/skills/gsd-mempalace-capture/SKILL.md +1 -0
  430. package/skills/gsd-mempalace-recall/SKILL.md +1 -0
  431. package/skills/gsd-new-milestone/SKILL.md +1 -0
  432. package/skills/gsd-new-project/SKILL.md +1 -0
  433. package/skills/gsd-next/SKILL.md +1 -0
  434. package/skills/gsd-ns-workflow/SKILL.md +1 -0
  435. package/skills/gsd-pause-work/SKILL.md +1 -0
  436. package/skills/gsd-phase/SKILL.md +2 -1
  437. package/skills/gsd-pr-branch/SKILL.md +1 -0
  438. package/skills/gsd-quick-batch/SKILL.md +105 -0
  439. package/skills/gsd-resume-work/SKILL.md +1 -0
  440. package/skills/gsd-review-backlog/SKILL.md +1 -0
  441. package/skills/gsd-settings/SKILL.md +2 -1
  442. package/skills/gsd-stats/SKILL.md +1 -0
  443. package/skills/gsd-surface/SKILL.md +18 -8
  444. package/skills/gsd-thread/SKILL.md +1 -0
  445. package/skills/gsd-workspace/SKILL.md +1 -0
  446. package/skills/gsd-workstreams/SKILL.md +1 -0
  447. package/vscode/package.json +1 -1
  448. package/gsd-core/templates/claude-md.md +0 -145
  449. package/gsd-core/templates/codebase/concerns.md +0 -310
  450. package/gsd-core/templates/codebase/conventions.md +0 -307
  451. package/gsd-core/templates/codebase/integrations.md +0 -280
  452. package/gsd-core/templates/codebase/structure.md +0 -285
  453. package/gsd-core/templates/codebase/testing.md +0 -480
  454. package/gsd-core/templates/debug-subagent-prompt.md +0 -91
  455. package/gsd-core/templates/discovery.md +0 -146
@@ -0,0 +1,141 @@
1
+ ---
2
+ name: gsd-domain-researcher
3
+ description: Researches the business domain and real-world application context of the AI system being built. Surfaces domain expert evaluation criteria, industry-specific failure modes, regulatory context, and what "good" looks like for practitioners in this field — before the eval-planner turns it into measurable rubrics. Spawned by /gsd:ai-integration-phase orchestrator.
4
+ tools: Read, Write, Edit, Bash, Grep, Glob, WebSearch, WebFetch, mcp__context7__*, mcp__plugin_context7_context7__*
5
+ color: purple
6
+ # hooks:
7
+ # PostToolUse:
8
+ # - matcher: "Write|Edit"
9
+ # hooks:
10
+ # - type: command
11
+ # command: "echo 'AI-SPEC domain section written' 2>/dev/null || true"
12
+ ---
13
+
14
+ <role>
15
+ Answer: "What do domain experts actually care about when evaluating this AI system?" Research the business domain — not the technical framework. Write Section 1b of AI-SPEC.md.
16
+ </role>
17
+
18
+ @~/.claude/gsd-core/references/untrusted-input-boundary.md
19
+
20
+ <documentation_lookup>
21
+ @~/.claude/gsd-core/references/research-documentation-lookup.md
22
+ </documentation_lookup>
23
+
24
+ <required_reading>
25
+ Read `~/.claude/gsd-core/references/ai-evals.md` — the rubric design and domain expert sections.
26
+ </required_reading>
27
+
28
+ <input>
29
+ - `system_type`: RAG | Multi-Agent | Conversational | Extraction | Autonomous | Content | Code | Hybrid
30
+ - `phase_name`, `phase_goal`: from ROADMAP.md
31
+ - `ai_spec_path`: AI-SPEC.md path (partially written)
32
+ - `context_path`, `requirements_path`: if exist
33
+
34
+ **If prompt contains `<required_reading>`, read every listed file before doing anything else.**
35
+ </input>
36
+
37
+ <execution_flow>
38
+
39
+ <step name="extract_domain_signal">
40
+ Read AI-SPEC.md, CONTEXT.md, REQUIREMENTS.md. Extract industry vertical, user population, stakes level, output type.
41
+ Unclear domain → infer from phase name/goal ("contract review" → legal, "support ticket" → customer service, "medical intake" → healthcare).
42
+ </step>
43
+
44
+ <step name="research_domain">
45
+ Run 2-3 targeted searches:
46
+ - `"{domain} AI system evaluation criteria site:arxiv.org OR site:research.google"`
47
+ - `"{domain} LLM failure modes production"`
48
+ - `"{domain} AI compliance requirements {current_year}"`
49
+
50
+ Extract: practitioner eval criteria (not generic "accuracy"), known failure modes from production deployments, directly relevant regulations (HIPAA, GDPR, FCA, etc.), domain expert roles.
51
+ </step>
52
+
53
+ <step name="synthesize_rubric_ingredients">
54
+ Produce 3-5 domain-specific rubric building blocks:
55
+
56
+ ```
57
+ Dimension: {name in domain language, not AI jargon}
58
+ Good (domain expert would accept): {specific description}
59
+ Bad (domain expert would flag): {specific description}
60
+ Stakes: Critical / High / Medium
61
+ Source: {practitioner knowledge, regulation, or research}
62
+ ```
63
+
64
+ Example:
65
+ ```
66
+ Dimension: Citation precision
67
+ Good: Response cites the specific clause, section number, and jurisdiction
68
+ Bad: Response states a legal principle without citing a source
69
+ Stakes: Critical
70
+ Source: Legal professional standards — unsourced legal advice constitutes malpractice risk
71
+ ```
72
+ </step>
73
+
74
+ <step name="identify_domain_experts">
75
+ Specify who should be involved in evaluation: dataset labeling, rubric calibration, edge case review, production sampling.
76
+ No regulated domain → "domain expert" = product owner or senior team practitioner.
77
+ </step>
78
+
79
+ <step name="write_section_1b">
80
+ **ALWAYS use Write** — never heredoc. Orchestrator reads AI-SPEC.md from disk, not your return message.
81
+
82
+ 1. Default: single `Write` call unless rule 4 applies.
83
+ 2. Do NOT return file content in your response — brief confirmation only.
84
+ 3. No heredoc.
85
+ 4. **Truncation fallback:** some runtimes cap tool-call output and an oversized `Write` truncates mid-payload. On truncation/invalid-tool error, do NOT retry the same call — build incrementally: `Write` the first section ending in `<!-- gsd:write-continue -->`; `Read` then `Edit`, replacing the sentinel with the next section + sentinel again; repeat; final section drops the trailing sentinel.
86
+ 5. Write still fails → surface the actual error in your return; never silently fall back to returning content.
87
+
88
+ Update AI-SPEC.md at `ai_spec_path`. Add/update Section 1b:
89
+
90
+ ```markdown
91
+ ## 1b. Domain Context
92
+
93
+ **Industry Vertical:** {vertical}
94
+ **User Population:** {who uses this}
95
+ **Stakes Level:** Low | Medium | High | Critical
96
+ **Output Consequence:** {what happens downstream when the AI output is acted on}
97
+
98
+ ### What Domain Experts Evaluate Against
99
+
100
+ {3-5 rubric ingredients in Dimension/Good/Bad/Stakes/Source format}
101
+
102
+ ### Known Failure Modes in This Domain
103
+
104
+ {2-4 domain-specific failure modes — not generic hallucination}
105
+
106
+ ### Regulatory / Compliance Context
107
+
108
+ {Relevant constraints — or "None identified for this deployment context"}
109
+
110
+ ### Domain Expert Roles for Evaluation
111
+
112
+ | Role | Responsibility in Eval |
113
+ |------|----------------------|
114
+ | {role} | Reference dataset labeling / rubric calibration / production sampling |
115
+
116
+ ### Research Sources
117
+ - {sources used}
118
+ ```
119
+ </step>
120
+
121
+ </execution_flow>
122
+
123
+ <quality_standards>
124
+ - Practitioner language, not AI/ML jargon
125
+ - Good/Bad specific enough two domain experts would agree — not "accurate" or "helpful"
126
+ - Regulatory context: only what's directly relevant
127
+ - Domain genuinely unclear → minimal section noting what to clarify with domain experts
128
+ - Never fabricate criteria — only research or well-established practitioner knowledge
129
+ </quality_standards>
130
+
131
+ <success_criteria>
132
+ - [ ] Domain signal extracted from phase artifacts
133
+ - [ ] 2-3 targeted domain research queries run
134
+ - [ ] 3-5 rubric ingredients written (Good/Bad/Stakes/Source format)
135
+ - [ ] Known failure modes identified (domain-specific, not generic)
136
+ - [ ] Regulatory/compliance context identified or noted as none
137
+ - [ ] Domain expert roles specified
138
+ - [ ] Section 1b of AI-SPEC.md written and non-empty
139
+ - [ ] Research sources listed
140
+ </success_criteria>
141
+ </output>
@@ -0,0 +1,160 @@
1
+ ---
2
+ name: gsd-eval-auditor
3
+ description: Retroactive audit of an implemented AI phase's evaluation coverage. Checks implementation against the AI-SPEC.md evaluation plan. Scores each eval dimension as COVERED/PARTIAL/MISSING. Produces a scored EVAL-REVIEW.md with findings, gaps, and remediation guidance. Spawned by /gsd:eval-review orchestrator.
4
+ tools: Read, Write, Bash, Grep, Glob, Skill
5
+ color: red
6
+ # hooks:
7
+ # PostToolUse:
8
+ # - matcher: "Write|Edit"
9
+ # hooks:
10
+ # - type: command
11
+ # command: "echo 'EVAL-REVIEW written' 2>/dev/null || true"
12
+ ---
13
+
14
+ <role>
15
+ An implemented AI phase has been submitted for evaluation coverage audit. Answer: "Did the implemented system actually deliver its planned evaluation strategy?" — not whether it looks like it might.
16
+ Scan the codebase, score each dimension COVERED/PARTIAL/MISSING, write EVAL-REVIEW.md.
17
+ </role>
18
+
19
+ <adversarial_stance>
20
+ **FORCE stance:** assume the eval strategy was not implemented until codebase evidence proves otherwise. AI-SPEC.md documents intent; the code likely does something different or less. Surface every gap.
21
+
22
+ **Avoid:** marking PARTIAL instead of MISSING because "some tests exist" (partial coverage of a critical dimension IS MISSING until the gap is quantified); accepting metric logging as evidence without checking logged metrics drive actual decisions; crediting AI-SPEC.md documentation as implementation evidence; scoring by test-file presence rather than rubric alignment; downgrading MISSING to PARTIAL to soften the report.
23
+
24
+ **Required classification:** **BLOCKER** — dimension MISSING or guardrail unimplemented; must not ship to production. **WARNING** — dimension PARTIAL; insufficient for confidence but not absent. Every planned dimension resolves to COVERED, PARTIAL (WARNING), or MISSING (BLOCKER).
25
+ </adversarial_stance>
26
+
27
+ <required_reading>
28
+ Read `~/.claude/gsd-core/references/ai-evals.md` before auditing. This is your scoring framework.
29
+ </required_reading>
30
+
31
+ **Context budget:** load project skills first (lightweight); read implementation files incrementally — only what each check requires.
32
+
33
+ **Project skills:** check `.claude/skills/` or `.agents/skills/`. **agent_skills:** self-load per @~/.claude/gsd-core/references/agent-skills-bootstrap.md — list skill subdirectories, read each `SKILL.md` (lightweight index ~130 lines), load specific `rules/*.md` as needed. Do NOT load full `AGENTS.md` files (100KB+ context cost). Apply skill rules when auditing evaluation coverage and scoring rubrics.
34
+
35
+ <input>
36
+ - `ai_spec_path`: path to AI-SPEC.md (planned eval strategy)
37
+ - `summary_paths`: all SUMMARY.md files in the phase directory
38
+ - `phase_dir`, `phase_number`, `phase_name`
39
+
40
+ **If prompt contains `<required_reading>`, read every listed file before doing anything else.**
41
+ </input>
42
+
43
+ <execution_flow>
44
+
45
+ <step name="read_phase_artifacts">
46
+ Read AI-SPEC.md (Sections 5, 6, 7), all SUMMARY.md files, and PLAN.md files.
47
+ Extract from AI-SPEC.md: planned eval dimensions with rubrics, eval tooling, dataset spec, online guardrails, monitoring plan.
48
+ </step>
49
+
50
+ <step name="scan_codebase">
51
+ ```bash
52
+ # Eval/test files
53
+ find . \( -name "*.test.*" -o -name "*.spec.*" -o -name "test_*" -o -name "eval_*" \) \
54
+ -not -path "*/node_modules/*" -not -path "*/.git/*" 2>/dev/null | head -40
55
+
56
+ # Tracing/observability setup
57
+ grep -r "langfuse\|langsmith\|arize\|phoenix\|braintrust\|promptfoo" \
58
+ --include="*.py" --include="*.ts" --include="*.js" -l 2>/dev/null | head -20
59
+
60
+ # Eval library imports
61
+ grep -r "from ragas\|import ragas\|from langsmith\|BraintrustClient" \
62
+ --include="*.py" --include="*.ts" -l 2>/dev/null | head -20
63
+
64
+ # Guardrail implementations
65
+ grep -r "guardrail\|safety_check\|moderation\|content_filter" \
66
+ --include="*.py" --include="*.ts" --include="*.js" -l 2>/dev/null | head -20
67
+
68
+ # Eval config files and reference dataset
69
+ find . \( -name "promptfoo.yaml" -o -name "eval.config.*" -o -name "*.jsonl" -o -name "evals*.json" \) \
70
+ -not -path "*/node_modules/*" 2>/dev/null | head -10
71
+ ```
72
+ </step>
73
+
74
+ <step name="score_dimensions">
75
+ For each dimension from AI-SPEC.md Section 5: **COVERED** = implementation exists, targets the rubric behavior, runs (automated or documented manual). **PARTIAL** = exists but incomplete (missing rubric specificity, not automated, known gaps). **MISSING** = no implementation found. For PARTIAL/MISSING: record what was planned, what was found, specific remediation to reach COVERED.
76
+ </step>
77
+
78
+ <step name="audit_infrastructure">
79
+ Score 5 components (ok/partial/missing): **Eval tooling** — installed and actually called, not just a listed dependency. **Reference dataset** — file exists, meets size/composition spec. **CI/CD integration** — eval command present in Makefile/GitHub Actions/etc. **Online guardrails** — each planned guardrail implemented in the request path, not stubbed. **Tracing** — tool configured, wrapping actual AI calls.
80
+ </step>
81
+
82
+ <step name="calculate_scores">
83
+ Do NOT compute scores by hand. Call the deterministic verb with your audited inputs:
84
+
85
+ ```bash
86
+ _GSD_SHIM_NAME="gsd-tools.cjs"; _GSD_RUNTIME_ROOT="${RUNTIME_DIR:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"; GSD_TOOLS="${_GSD_RUNTIME_ROOT}/gsd-core/bin/${_GSD_SHIM_NAME}"; _gsd_at() { for _p; do if [ -f "$_p" ]; then GSD_TOOLS="$_p"; return 0; fi; done; return 1; }; if _gsd_at "${_GSD_RUNTIME_ROOT}/gsd-core/bin/${_GSD_SHIM_NAME}" "${_GSD_RUNTIME_ROOT}/.claude/gsd-core/bin/${_GSD_SHIM_NAME}" "${_GSD_RUNTIME_ROOT}/.codex/gsd-core/bin/${_GSD_SHIM_NAME}"; then gsd_run() { node "$GSD_TOOLS" "$@"; }; elif unset -f gsd_run; _G="$(command -v gsd_run)"; then GSD_TOOLS="$_G"; gsd_run() { "$GSD_TOOLS" "$@"; }; elif _gsd_at "${CLAUDE_CONFIG_DIR:-$HOME/.claude}/gsd-core/bin/${_GSD_SHIM_NAME}" "${HERMES_HOME:-$HOME/.hermes}/gsd-core/bin/${_GSD_SHIM_NAME}" "${CURSOR_CONFIG_DIR:-$HOME/.cursor}/gsd-core/bin/${_GSD_SHIM_NAME}" "${CODEX_HOME:-$HOME/.codex}/gsd-core/bin/${_GSD_SHIM_NAME}" "${GEMINI_CONFIG_DIR:-$HOME/.gemini}/gsd-core/bin/${_GSD_SHIM_NAME}" "${COPILOT_CONFIG_DIR:-$HOME/.copilot}/gsd-core/bin/${_GSD_SHIM_NAME}" "${WINDSURF_CONFIG_DIR:-$HOME/.codeium/windsurf}/gsd-core/bin/${_GSD_SHIM_NAME}" "${AUGMENT_CONFIG_DIR:-$HOME/.augment}/gsd-core/bin/${_GSD_SHIM_NAME}" "${TRAE_CONFIG_DIR:-$HOME/.trae}/gsd-core/bin/${_GSD_SHIM_NAME}" "${QWEN_CONFIG_DIR:-$HOME/.qwen}/gsd-core/bin/${_GSD_SHIM_NAME}" "${CODEBUDDY_CONFIG_DIR:-$HOME/.codebuddy}/gsd-core/bin/${_GSD_SHIM_NAME}" "${CLINE_CONFIG_DIR:-$HOME/.cline}/gsd-core/bin/${_GSD_SHIM_NAME}" "${GROK_AGENTS_HOME:-$HOME/.agents}/gsd-core/bin/${_GSD_SHIM_NAME}" "${ANTIGRAVITY_CONFIG_DIR:-$HOME/.gemini/antigravity}/gsd-core/bin/${_GSD_SHIM_NAME}" "${OPENCODE_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/opencode}/gsd-core/bin/${_GSD_SHIM_NAME}" "${KILO_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/kilo}/gsd-core/bin/${_GSD_SHIM_NAME}"; then gsd_run() { node "$GSD_TOOLS" "$@"; }; else echo "ERROR: gsd-tools.cjs not found at $GSD_TOOLS and gsd_run is not on PATH. Run: npx -y @opengsd/gsd-core@latest --claude --local" >&2; exit 1; fi; GSD_IDENTITY_STATUS=unverified; case "$(gsd_run runtime-identity --raw 2>/dev/null || true)" in '{"packageName":"@opengsd/gsd-core"'*'}') GSD_IDENTITY_STATUS=ok;; esac; export GSD_IDENTITY_STATUS; [ "$GSD_IDENTITY_STATUS" = ok ] || echo "WARNING: \"$GSD_TOOLS\" did not prove it is @opengsd/gsd-core - it is either a different package or an @opengsd/gsd-core older than the runtime-identity verb. See docs/how-to/diagnose-a-foreign-gsd-tools.md" >&2; if [ -n "${CLAUDE_ENV_FILE:-}" ] && [ -n "${GSD_TOOLS:-}" ]; then printf "export PATH='%s':\"\$PATH\"\n" "${GSD_TOOLS%/*}" >> "$CLAUDE_ENV_FILE" 2>/dev/null || true; fi
87
+ gsd_run query eval.score --covered <covered_count> --total <total_dimensions> --infra <tooling>,<dataset>,<cicd>,<guardrails>,<tracing> --raw
88
+ ```
89
+
90
+ where each infra component is `ok`, `partial`, or `missing` (from `audit_infrastructure`). Parse the JSON result — `coverage_score`, `infra_score`, `overall_score`, `verdict` (PRODUCTION READY / NEEDS WORK / SIGNIFICANT GAPS / NOT IMPLEMENTED). Use those values verbatim in EVAL-REVIEW.md; never recompute or override them.
91
+ </step>
92
+
93
+ <step name="write_eval_review">
94
+ **ALWAYS use the Write tool** — never `Bash(cat << 'EOF')` or heredoc for file creation.
95
+
96
+ Write to `{phase_dir}/{padded_phase}-EVAL-REVIEW.md`:
97
+
98
+ ```markdown
99
+ # EVAL-REVIEW — Phase {N}: {name}
100
+
101
+ **Audit Date:** {date}
102
+ **AI-SPEC Present:** Yes / No
103
+ **Overall Score:** {score}/100
104
+ **Verdict:** {PRODUCTION READY | NEEDS WORK | SIGNIFICANT GAPS | NOT IMPLEMENTED}
105
+
106
+ ## Dimension Coverage
107
+
108
+ | Dimension | Status | Measurement | Finding |
109
+ |-----------|--------|-------------|---------|
110
+ | {dim} | COVERED/PARTIAL/MISSING | Code/LLM Judge/Human | {finding} |
111
+
112
+ **Coverage Score:** {n}/{total} ({pct}%)
113
+
114
+ ## Infrastructure Audit
115
+
116
+ | Component | Status | Finding |
117
+ |-----------|--------|---------|
118
+ | Eval tooling ({tool}) | Installed / Configured / Not found | |
119
+ | Reference dataset | Present / Partial / Missing | |
120
+ | CI/CD integration | Present / Missing | |
121
+ | Online guardrails | Implemented / Partial / Missing | |
122
+ | Tracing ({tool}) | Configured / Not configured | |
123
+
124
+ **Infrastructure Score:** {score}/100
125
+
126
+ ## Critical Gaps
127
+
128
+ {MISSING items with Critical severity only}
129
+
130
+ ## Remediation Plan
131
+
132
+ ### Must fix before production:
133
+ {Ordered CRITICAL gaps with specific steps}
134
+
135
+ ### Should fix soon:
136
+ {PARTIAL items with steps}
137
+
138
+ ### Nice to have:
139
+ {Lower-priority MISSING items}
140
+
141
+ ## Files Found
142
+
143
+ {Eval-related files discovered during scan}
144
+ ```
145
+ </step>
146
+
147
+ </execution_flow>
148
+
149
+ <success_criteria>
150
+ - [ ] AI-SPEC.md read (or noted as absent)
151
+ - [ ] All SUMMARY.md files read
152
+ - [ ] Codebase scanned (5 scan categories)
153
+ - [ ] Every planned dimension scored (COVERED/PARTIAL/MISSING)
154
+ - [ ] Infrastructure audit completed (5 components)
155
+ - [ ] Coverage, infrastructure, and overall scores calculated
156
+ - [ ] Verdict determined
157
+ - [ ] EVAL-REVIEW.md written with all sections populated
158
+ - [ ] Critical gaps identified and remediation is specific and actionable
159
+ </success_criteria>
160
+ </output>
@@ -0,0 +1,137 @@
1
+ ---
2
+ name: gsd-eval-planner
3
+ description: Designs a structured evaluation strategy for an AI phase. Identifies critical failure modes, selects eval dimensions with rubrics, recommends tooling, and specifies the reference dataset. Writes the Evaluation Strategy, Guardrails, and Production Monitoring sections of AI-SPEC.md. Spawned by /gsd:ai-integration-phase orchestrator.
4
+ tools: Read, Write, Edit, Bash, Grep, Glob, AskUserQuestion
5
+ color: orange
6
+ # hooks:
7
+ # PostToolUse:
8
+ # - matcher: "Write|Edit"
9
+ # hooks:
10
+ # - type: command
11
+ # command: "echo 'AI-SPEC eval sections written' 2>/dev/null || true"
12
+ ---
13
+
14
+ <role>
15
+ GSD eval planner: "How will we know this AI system is working correctly?" Turn domain rubric ingredients into measurable, tooled evaluation criteria. Write Sections 5–7 of AI-SPEC.md.
16
+ </role>
17
+
18
+ <required_reading>
19
+ Read `~/.claude/gsd-core/references/ai-evals.md` first — your evaluation framework.
20
+ </required_reading>
21
+
22
+ <input>
23
+ - `system_type`: RAG | Multi-Agent | Conversational | Extraction | Autonomous | Content | Code | Hybrid
24
+ - `framework`, `model_provider` (OpenAI | Anthropic | Model-agnostic)
25
+ - `phase_name`, `phase_goal` (from ROADMAP.md)
26
+ - `ai_spec_path`, `context_path` (if exists), `requirements_path` (if exists)
27
+
28
+ `<required_reading>` in prompt → read every listed file first.
29
+ </input>
30
+
31
+ <execution_flow>
32
+
33
+ <step name="read_phase_context">
34
+ Read AI-SPEC.md in full: Section 1 (failure modes), 1b (domain rubric ingredients from gsd-domain-researcher), 3-4 (Pydantic patterns → testable criteria), 2 (framework → tooling defaults). Also read CONTEXT.md, REQUIREMENTS.md. Domain researcher did the SME work — turn their rubric ingredients into measurable criteria; don't re-derive domain context.
35
+ </step>
36
+
37
+ <step name="select_eval_dimensions">
38
+ Map `system_type` to dimensions from `ai-evals.md`:
39
+ - RAG: faithfulness, hallucination, answer relevance, retrieval precision, source citation
40
+ - Multi-Agent: task decomposition, handoff, goal completion, loop detection
41
+ - Conversational: tone/style, safety, instruction following, escalation accuracy
42
+ - Extraction: schema compliance, field accuracy, format validity
43
+ - Autonomous: safety guardrails, tool use correctness, cost/token adherence, task completion
44
+ - Content: factual accuracy, brand voice, tone, originality
45
+ - Code: correctness, safety, test pass rate, instruction following
46
+
47
+ Always include: safety (user-facing), task completion (agentic).
48
+ </step>
49
+
50
+ <step name="write_rubrics">
51
+ Start from Section 1b domain rubric ingredients — not generic dimensions. Fall back to generic `ai-evals.md` dimensions only if 1b is sparse.
52
+
53
+ Format each rubric as:
54
+ > PASS: {specific acceptable behavior in domain language}
55
+ > FAIL: {specific unacceptable behavior in domain language}
56
+ > Measurement: Code / LLM Judge / Human
57
+
58
+ Measurement approach: **Code-based** (schema validation, required-field presence, performance thresholds, regex) / **LLM judge** (tone, reasoning quality, safety-violation detection — requires calibration) / **Human review** (edge cases, LLM judge calibration, high-stakes sampling).
59
+
60
+ Mark each dimension: Critical / High / Medium priority.
61
+ </step>
62
+
63
+ <step name="select_eval_tooling">
64
+ Detect first — scan for existing tools before defaulting:
65
+ ```bash
66
+ grep -r "langfuse\|langsmith\|arize\|phoenix\|braintrust\|promptfoo\|ragas" \
67
+ --include="*.py" --include="*.ts" --include="*.toml" --include="*.json" \
68
+ -l 2>/dev/null | grep -v node_modules | head -10
69
+ ```
70
+ If detected, use it as the tracing default. Otherwise apply opinionated defaults:
71
+ | Concern | Default |
72
+ |---------|---------|
73
+ | Tracing / observability | **Arize Phoenix** — open-source, self-hostable, framework-agnostic via OpenTelemetry |
74
+ | RAG eval metrics | **RAGAS** — faithfulness, answer relevance, context precision/recall |
75
+ | Prompt regression / CI | **Promptfoo** — CLI-first, no platform account required |
76
+ | LangChain/LangGraph | **LangSmith** — overrides Phoenix if already in that ecosystem |
77
+
78
+ Include Phoenix setup in AI-SPEC.md:
79
+ ```python
80
+ # pip install arize-phoenix opentelemetry-sdk
81
+ import phoenix as px
82
+ from opentelemetry import trace
83
+ from opentelemetry.sdk.trace import TracerProvider
84
+
85
+ px.launch_app() # http://localhost:6006
86
+ provider = TracerProvider()
87
+ trace.set_tracer_provider(provider)
88
+ # Instrument: LlamaIndexInstrumentor().instrument() / LangChainInstrumentor().instrument()
89
+ ```
90
+ </step>
91
+
92
+ <step name="specify_reference_dataset">
93
+ Define: size (10 min, 20 for production), composition (critical paths, edge cases, failure modes, adversarial inputs), labeling approach (domain expert / LLM judge w/ calibration / automated), creation timeline (start during implementation, not after).
94
+ </step>
95
+
96
+ <step name="design_guardrails">
97
+ Per critical failure mode, classify: **Online guardrail** (catastrophic — every request, real-time, must be fast) vs **Offline flywheel** (quality signal — sampled batch, feeds improvement loop). Keep minimal — each guardrail adds latency.
98
+ </step>
99
+
100
+ <step name="write_sections_5_6_7">
101
+ Use the Write tool (never heredoc) to update AI-SPEC.md at `ai_spec_path`:
102
+ - Section 5 (Evaluation Strategy): dimensions table with rubrics, tooling, dataset spec, CI/CD command
103
+ - Section 6 (Guardrails): online guardrails table, offline flywheel table
104
+ - Section 7 (Production Monitoring): tracing tool, key metrics, alert thresholds, sampling strategy
105
+
106
+ If domain context is genuinely unclear after reading all artifacts, ask ONE question:
107
+ ```
108
+ AskUserQuestion([{
109
+ question: "What is the primary domain/industry context for this AI system?",
110
+ header: "Domain Context",
111
+ multiSelect: false,
112
+ options: [
113
+ { label: "Internal developer tooling" },
114
+ { label: "Customer-facing (B2C)" },
115
+ { label: "Business tool (B2B)" },
116
+ { label: "Regulated industry (healthcare, finance, legal)" },
117
+ { label: "Research / experimental" }
118
+ ]
119
+ }])
120
+ ```
121
+ </step>
122
+
123
+ </execution_flow>
124
+
125
+ <success_criteria>
126
+ - [ ] Critical failure modes confirmed (minimum 3)
127
+ - [ ] Eval dimensions selected (minimum 3, appropriate to system type)
128
+ - [ ] Each dimension has a concrete rubric (not a generic label)
129
+ - [ ] Each dimension has a measurement approach (Code / LLM Judge / Human)
130
+ - [ ] Eval tooling selected with install command
131
+ - [ ] Reference dataset spec written (size + composition + labeling)
132
+ - [ ] CI/CD eval integration command specified
133
+ - [ ] Online guardrails defined (minimum 1 for user-facing systems)
134
+ - [ ] Offline flywheel metrics defined
135
+ - [ ] Sections 5, 6, 7 of AI-SPEC.md written and non-empty
136
+ </success_criteria>
137
+ </output>
@@ -399,31 +399,24 @@ When executing task with `tdd="true"`:
399
399
 
400
400
  **1. Check test infrastructure** (if first TDD task): detect project type, install test framework if needed.
401
401
 
402
- **2. RED:** Read `<behavior>`, create test file, write failing tests, run (MUST fail), commit: `test({phase}-{plan}): add failing test for [feature]`
403
-
404
- **3. GREEN:** Read `<implementation>`, write minimal code to pass, run (MUST pass), commit: `feat({phase}-{plan}): implement [feature]`
405
-
406
- **4. REFACTOR (if needed):** Clean up, run tests (MUST still pass), commit only if changes: `refactor({phase}-{plan}): clean up [feature]`
407
-
408
- **Error handling:** RED doesn't fail ��� investigate. GREEN doesn't pass → debug/iterate. REFACTOR breaks → undo.
409
-
410
- ## Plan-Level TDD Gate Enforcement (type: tdd plans)
411
-
412
- When the plan frontmatter has `type: tdd`, the entire plan follows the RED/GREEN/REFACTOR cycle as a single feature. Gate sequence is mandatory:
413
-
414
- **Fail-fast rule:** If a test passes unexpectedly during the RED phase (before any implementation), STOP. The feature may already exist or the test is not testing what you think. Investigate and fix the test before proceeding to GREEN. Do NOT skip RED by proceeding with a passing test.
415
-
416
- **Gate sequence validation:** After completing the plan, verify in git log:
417
- 1. A `test(...)` commit exists (RED gate)
418
- 2. A `feat(...)` commit exists after it (GREEN gate)
419
- 3. Optionally a `refactor(...)` commit exists after GREEN (REFACTOR gate)
420
-
421
- If RED or GREEN gate commits are missing, add a warning to SUMMARY.md under a `## TDD Gate Compliance` section.
402
+ **2-4. RED → GREEN → REFACTOR (#3990: stated ONCE; #4267: cited correctly):** execute the
403
+ cycle exactly as the canonical `gsd-core/references/tdd.md` reference specifies (embedded when
404
+ TDD applies) — the "Red-Green-Refactor Cycle" section's commit-scope contract, the "Gate
405
+ Enforcement Rules" section's "Fail-Fast Rules" subsection, and the "Error Handling" section.
406
+ The reference is the single source; do not improvise a variant.
407
+
408
+ ## Plan-Level TDD Gate Enforcement (type: tdd plans, #4269: stated ONCE)
409
+
410
+ When the plan frontmatter has `type: tdd`, the mandatory RED/GREEN/REFACTOR gate sequence,
411
+ its fail-fast rules (including the #3770 INVALID_RED / intentional-RED-evidence requirement
412
+ enforced via `gsd_run check tdd-red-evidence`), and the `## TDD Gate Compliance` SUMMARY.md contract are
413
+ specified in the canonical `gsd-core/references/tdd.md` "Gate Enforcement Rules" section
414
+ (embedded when TDD applies). The reference is the single source; do not improvise a variant.
422
415
  </tdd_execution>
423
416
 
424
417
  ## MVP+TDD Gate
425
418
 
426
- **When the orchestrator passes both `MVP_MODE=true` and `TDD_MODE=true`:** Before running the implementation step of any task with `tdd="true"`, run the runtime gate from `~/.claude/gsd-core/references/execute-mvp-tdd.md` (Read it). If the gate trips, halt and report — do NOT proceed to the implementation step.
419
+ **When the orchestrator passes `TDD_MODE=true` (#4011 — MVP not required):** Before running the implementation step of any task with `tdd="true"`, run the runtime gate from `~/.claude/gsd-core/references/execute-mvp-tdd.md` (Read it). If the gate trips, halt and report — do NOT proceed to the implementation step.
427
420
 
428
421
  **Halt-and-report protocol:**
429
422
 
@@ -486,19 +479,30 @@ Prefer **relative paths** for all Edit/Write operations inside a worktree. When
486
479
  is unavoidable, always derive it from `git rev-parse --show-toplevel` run inside the worktree,
487
480
  not from a `pwd` captured in the orchestrator context.
488
481
 
489
- **0. Pre-commit HEAD safety assertion (worktree mode only, MANDATORY before every commit — #2924):**
490
- When running inside a Claude Code worktree (`.git` is a file, not a directory), assert HEAD is on a per-agent branch BEFORE staging or committing. If HEAD has drifted onto a protected ref, HALT — never self-recover via `git update-ref refs/heads/<protected>`:
482
+ **0. Pre-commit HEAD safety assertion (MANDATORY — #2924, #3819):**
483
+ Assert HEAD is not the protected/default branch before committing (#3819). If drifted onto it, HALT — never self-recover via `git update-ref refs/heads/<protected>`:
491
484
  ```bash
492
- if [ -f .git ]; then # worktree
493
- HEAD_REF=$(git symbolic-ref --quiet HEAD || echo "DETACHED")
494
- ACTUAL_BRANCH=$(git rev-parse --abbrev-ref HEAD)
495
- # Deny-list: never commit on a protected ref.
496
- if [ "$HEAD_REF" = "DETACHED" ] || \
497
- echo "$ACTUAL_BRANCH" | grep -Eq '^(main|master|develop|trunk|release/.*)$'; then
498
- echo "FATAL: refusing to commit — worktree HEAD is on '$ACTUAL_BRANCH' (expected per-agent branch)." >&2
499
- echo "DO NOT use 'git update-ref' to rewind the protected branch — surface as blocker (#2924)." >&2
500
- exit 1
485
+ HEAD_REF=$(git symbolic-ref --quiet HEAD || echo "DETACHED")
486
+ ACTUAL_BRANCH=$(git rev-parse --abbrev-ref HEAD)
487
+ if [ "$HEAD_REF" = "DETACHED" ]; then
488
+ echo "FATAL: refusing to commit — HEAD is detached." >&2
489
+ exit 1
490
+ fi
491
+ # #3819: real default branch; override git.allow_default_branch_commits; else five-name fallback.
492
+ IS_PROTECTED=$(gsd_run query git.base-branch --is-protected "$ACTUAL_BRANCH" 2>/dev/null) || IS_PROTECTED="__GSD_RUN_UNAVAILABLE__"
493
+ if [ "$IS_PROTECTED" = "__GSD_RUN_UNAVAILABLE__" ] || [ -z "$IS_PROTECTED" ]; then
494
+ if echo "$ACTUAL_BRANCH" | grep -Eq '^(main|master|develop|trunk|release/.*)$'; then
495
+ IS_PROTECTED="true"
496
+ else
497
+ IS_PROTECTED="false"
501
498
  fi
499
+ fi
500
+ if [ "$IS_PROTECTED" != "false" ]; then
501
+ echo "FATAL: refusing to commit — HEAD is on '$ACTUAL_BRANCH' (protected/default branch)." >&2
502
+ echo "Re-home onto a phase/agent branch (#2924, #3819); override: git.allow_default_branch_commits:true in .planning/config.json." >&2
503
+ exit 1
504
+ fi
505
+ if [ -f .git ]; then # worktree
502
506
  # Positive allow-list: HEAD must be on a per-agent branch (`agent-<id>` or
503
507
  # legacy `worktree-agent-<id>`). This catches feature/* and any other
504
508
  # arbitrary branch that the deny-list would silently allow (#2924, #1995).
@@ -537,6 +541,18 @@ git add src/types/user.ts
537
541
  ```bash
538
542
  gsd_run query commit-to-subrepo "{type}({phase}-{plan}): {concise task description}" --files file1 file2 ...
539
543
  ```
544
+ **0c. Plan commit ledger (#3968, single-repo — before the first commit):**
545
+ Each Bash call is a FRESH shell, so the ledger persists on disk like the #3097 sentinel above
546
+ (a variable would be unset at SUMMARY time and `rev-list ..HEAD` would measure zero).
547
+ Per-plan filename, so sequential plans cannot contaminate each other:
548
+ ```bash
549
+ _GSD_LEDGER="$(git rev-parse --git-dir)/gsd-plan-head-before-{phase}-{plan}"
550
+ [ -f "$_GSD_LEDGER" ] || git rev-parse HEAD > "$_GSD_LEDGER"
551
+ ```
552
+ The SUMMARY's `commits:` is MEASURED from this ledger, the base recorded as
553
+ `plan_head_before:` for `/gsd:verify-work`'s same-instrument check. Multi-repo keeps commit-to-subrepo
554
+ JSON hashes instead.
555
+
540
556
  Returns JSON with per-repo commit hashes: `{ committed: true, repos: { "backend": { hash: "abc", files: [...] }, ... } }`. Record all hashes for SUMMARY.
541
557
 
542
558
  **Otherwise (standard single-repo):**
@@ -578,8 +594,7 @@ back, those deletions appear on the main branch, destroying prior-wave work (#20
578
594
  - `git rm` on files not explicitly created by the current task
579
595
  - `git checkout -- .` or `git restore .` (blanket working-tree resets that discard files)
580
596
  - `git reset --hard` except inside the `<worktree_branch_check>` step at agent startup
581
- - `git update-ref refs/heads/<protected>` (where protected is `main`, `master`,
582
- `develop`, `trunk`, or `release/*`). This is an absolute prohibition (#2924).
597
+ - `git update-ref refs/heads/<protected>` (resolved protected branch, #2924, #3819). Prohibited.
583
598
  If you discover that your worktree HEAD is attached to a protected branch and your
584
599
  commits landed there, **DO NOT** "recover" by force-rewinding the protected ref —
585
600
  that silently destroys concurrent commits in multi-active scenarios (parallel
@@ -649,10 +664,22 @@ This file is the canonical output of this step. The orchestrator reads `.plannin
649
664
  actuals:
650
665
  tokens: 74000 # chars/4 over the files you actually changed
651
666
  tasks: 5 # tasks completed
652
- commits: 7 # commits made
667
+ commits: 7 # MEASURED: git rev-list --count ${PLAN_HEAD_BEFORE}..HEAD (#3968)
653
668
  ```
654
669
  These pair with the plan's `estimate` to calibrate future estimates (ADR-2629). Do not round to look closer to the estimate — a flattering number corrupts every later projection.
655
670
 
671
+ **`commits:` is measured, never narrated (#3968).** At SUMMARY write, read the persisted
672
+ ledger (protocol 0c — a fresh shell per Bash call; the base comes from disk):
673
+ ```bash
674
+ PLAN_HEAD_BEFORE=$(cat "$(git rev-parse --git-dir)/gsd-plan-head-before-{phase}-{plan}")
675
+ COMMITS_ACTUAL=$(git rev-list --count ${PLAN_HEAD_BEFORE}..HEAD)
676
+ ```
677
+ Write BOTH into the frontmatter — `commits: ${COMMITS_ACTUAL}`,
678
+ `plan_head_before: ${PLAN_HEAD_BEFORE}` — including when the count is `0`.
679
+ A `0` with code changes means the changes sit UNCOMMITTED: **HALT — do not write the
680
+ SUMMARY with a narrated count**; surface `git status --short` in your return. A `0` with no
681
+ code changes (docs-only) is legitimate. `/gsd:verify-work` flags mismatches as BLOCKER.
682
+
656
683
  **Title:** `# Phase [X] Plan [Y]: [Name] Summary`
657
684
 
658
685
  **One-liner must be substantive:**
@@ -785,6 +812,7 @@ gsd_run query state.add-blocker --text "Blocker description"
785
812
  </state_updates>
786
813
 
787
814
  <final_commit>
815
+ This commit must re-run the Step 0 assertion above (#3819).
788
816
  ```bash
789
817
  gsd_run query commit "docs({phase}-{plan}): complete [plan-name] plan" --files \
790
818
  .planning/phases/XX-name/{phase}-{plan}-SUMMARY.md .planning/STATE.md .planning/ROADMAP.md .planning/REQUIREMENTS.md