@opengsd/gsd-core 1.4.4 → 1.5.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/README.md +3 -3
- package/agents/gsd-code-fixer.md +3 -2
- package/agents/gsd-debug-session-manager.md +2 -1
- package/agents/gsd-debugger.md +4 -3
- package/agents/gsd-executor.md +17 -16
- package/agents/gsd-intel-updater.md +38 -41
- package/agents/gsd-phase-researcher.md +8 -8
- package/agents/gsd-plan-checker.md +23 -13
- package/agents/gsd-planner.md +32 -188
- package/agents/gsd-project-researcher.md +5 -4
- package/agents/gsd-research-synthesizer.md +2 -1
- package/agents/gsd-ui-researcher.md +2 -1
- package/agents/gsd-verifier.md +12 -11
- package/bin/install.js +965 -1486
- package/commands/gsd/autonomous.md +5 -1
- package/commands/gsd/ns-manage.md +8 -1
- package/commands/gsd/ns-project.md +5 -0
- package/commands/gsd/ns-review.md +4 -1
- package/commands/gsd/ns-workflow.md +7 -1
- package/commands/gsd/plan-review-convergence.md +5 -4
- package/commands/gsd/surface.md +12 -5
- package/gemini-extension.json +1 -1
- package/gsd-core/bin/gsd-tools.cjs +198 -101
- package/gsd-core/bin/gsd_run +20 -0
- package/gsd-core/bin/lib/audit-command-router.cjs +61 -0
- package/gsd-core/bin/lib/capability-registry.cjs +2234 -0
- package/gsd-core/bin/lib/capability-state.cjs +336 -0
- package/gsd-core/bin/lib/check-command-router.cjs +133 -2
- package/gsd-core/bin/lib/cli-exit.cjs +22 -3
- package/gsd-core/bin/lib/config-loader.cjs +716 -0
- package/gsd-core/bin/lib/configuration.cjs +4 -34
- package/gsd-core/bin/lib/core-utils.cjs +198 -0
- package/gsd-core/bin/lib/core.cjs +107 -1817
- package/gsd-core/bin/lib/edge-probe.cjs +173 -0
- package/gsd-core/bin/lib/fallow-runner.cjs +63 -25
- package/gsd-core/bin/lib/federated-config.cjs +182 -0
- package/gsd-core/bin/lib/graphify-command-router.cjs +74 -0
- package/gsd-core/bin/lib/init.cjs +58 -12
- package/gsd-core/bin/lib/install-profiles.cjs +157 -3
- package/gsd-core/bin/lib/intel-command-router.cjs +116 -0
- package/gsd-core/bin/lib/intel.cjs +3 -3
- package/gsd-core/bin/lib/io.cjs +222 -0
- package/gsd-core/bin/lib/loop-host-contract.cjs +105 -0
- package/gsd-core/bin/lib/loop-resolver.cjs +460 -0
- package/gsd-core/bin/lib/model-resolver.cjs +426 -0
- package/gsd-core/bin/lib/phase-command-router.cjs +20 -0
- package/gsd-core/bin/lib/phase-id.cjs +215 -0
- package/gsd-core/bin/lib/phase-locator.cjs +148 -0
- package/gsd-core/bin/lib/phase.cjs +17 -0
- package/gsd-core/bin/lib/probe-core.cjs +257 -0
- package/gsd-core/bin/lib/profile-pipeline.cjs +2 -2
- package/gsd-core/bin/lib/roadmap-parser.cjs +443 -0
- package/gsd-core/bin/lib/roadmap.cjs +44 -1
- package/gsd-core/bin/lib/runtime-artifact-layout.cjs +92 -95
- package/gsd-core/bin/lib/runtime-config-adapter-registry.cjs +68 -29
- package/gsd-core/bin/lib/runtime-homes.cjs +163 -87
- package/gsd-core/bin/lib/runtime-hooks-surface.cjs +1439 -0
- package/gsd-core/bin/lib/runtime-name-policy.cjs +2 -1
- package/gsd-core/bin/lib/runtime-slash.cjs +7 -2
- package/gsd-core/bin/lib/shell-command-projection.cjs +13 -0
- package/gsd-core/bin/lib/state-document.cjs +8 -0
- package/gsd-core/bin/lib/state.cjs +114 -2
- package/gsd-core/bin/lib/surface.cjs +66 -14
- package/gsd-core/bin/lib/uat-predicate.cjs +329 -0
- package/gsd-core/bin/lib/update-context.cjs +4 -1
- package/gsd-core/bin/lib/verify.cjs +104 -1
- package/gsd-core/bin/lib/worktree-base-ref.cjs +33 -8
- package/gsd-core/bin/shared/model-catalog.json +11 -6
- package/gsd-core/bin/shared/runtime-aliases.manifest.json +5 -1
- package/gsd-core/references/edge-probe-fixtures/01-round-half-even/expected-coverage.json +7 -0
- package/gsd-core/references/edge-probe-fixtures/01-round-half-even/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/02-merge-intervals/expected-coverage.json +8 -0
- package/gsd-core/references/edge-probe-fixtures/02-merge-intervals/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/03-truncate-graphemes/expected-coverage.json +7 -0
- package/gsd-core/references/edge-probe-fixtures/03-truncate-graphemes/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/04-money-rounding/expected-coverage.json +7 -0
- package/gsd-core/references/edge-probe-fixtures/04-money-rounding/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/05-list-dedupe/expected-coverage.json +8 -0
- package/gsd-core/references/edge-probe-fixtures/05-list-dedupe/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/06-resolved-mixed/expected-coverage.json +8 -0
- package/gsd-core/references/edge-probe-fixtures/06-resolved-mixed/requirements.json +1 -0
- package/gsd-core/references/edge-probe-fixtures/06-resolved-mixed/resolutions.json +4 -0
- package/gsd-core/references/edge-probe.md +261 -0
- package/gsd-core/references/planner-antipatterns.md +41 -0
- package/gsd-core/references/planner-guidance.md +186 -0
- package/gsd-core/references/planner-reviews.md +5 -2
- package/gsd-core/templates/phase-prompt.md +7 -7
- package/gsd-core/templates/project.md +19 -2
- package/gsd-core/templates/spec.md +12 -0
- package/gsd-core/templates/summary-complex.md +1 -0
- package/gsd-core/templates/summary-minimal.md +1 -0
- package/gsd-core/templates/summary-standard.md +1 -0
- package/gsd-core/templates/summary.md +1 -0
- package/gsd-core/workflows/_runtime-launcher.snippet.sh +1 -1
- package/gsd-core/workflows/add-backlog.md +1 -1
- package/gsd-core/workflows/add-phase.md +1 -1
- package/gsd-core/workflows/add-tests.md +1 -1
- package/gsd-core/workflows/add-todo.md +1 -1
- package/gsd-core/workflows/ai-integration-phase.md +1 -1
- package/gsd-core/workflows/audit-fix.md +1 -1
- package/gsd-core/workflows/audit-milestone.md +1 -1
- package/gsd-core/workflows/audit-uat.md +1 -1
- package/gsd-core/workflows/autonomous.md +111 -51
- package/gsd-core/workflows/check-todos.md +1 -1
- package/gsd-core/workflows/cleanup.md +1 -1
- package/gsd-core/workflows/code-review-fix.md +6 -4
- package/gsd-core/workflows/code-review.md +53 -17
- package/gsd-core/workflows/complete-milestone.md +11 -5
- package/gsd-core/workflows/debug.md +1 -1
- package/gsd-core/workflows/diagnose-issues.md +1 -1
- package/gsd-core/workflows/discuss-phase/modes/advisor.md +1 -1
- package/gsd-core/workflows/discuss-phase/modes/auto.md +1 -1
- package/gsd-core/workflows/discuss-phase/modes/chain.md +1 -1
- package/gsd-core/workflows/discuss-phase-assumptions.md +1 -1
- package/gsd-core/workflows/discuss-phase.md +8 -1
- package/gsd-core/workflows/do.md +1 -1
- package/gsd-core/workflows/docs-update.md +1 -1
- package/gsd-core/workflows/edit-phase.md +1 -1
- package/gsd-core/workflows/eval-review.md +4 -1
- package/gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/post-merge-gate.md +1 -1
- package/gsd-core/workflows/execute-phase.md +8 -1
- package/gsd-core/workflows/execute-plan.md +1 -1
- package/gsd-core/workflows/explore.md +1 -1
- package/gsd-core/workflows/extract-learnings.md +1 -1
- package/gsd-core/workflows/forensics.md +1 -1
- package/gsd-core/workflows/graduation.md +1 -1
- package/gsd-core/workflows/health.md +1 -1
- package/gsd-core/workflows/help/modes/full.md +1 -1
- package/gsd-core/workflows/import.md +1 -1
- package/gsd-core/workflows/ingest-docs.md +1 -1
- package/gsd-core/workflows/insert-phase.md +1 -1
- package/gsd-core/workflows/list-workspaces.md +1 -1
- package/gsd-core/workflows/manager.md +1 -1
- package/gsd-core/workflows/map-codebase.md +1 -1
- package/gsd-core/workflows/milestone-summary.md +1 -1
- package/gsd-core/workflows/mvp-phase.md +1 -1
- package/gsd-core/workflows/new-milestone.md +9 -1
- package/gsd-core/workflows/new-project.md +9 -1
- package/gsd-core/workflows/new-workspace.md +1 -1
- package/gsd-core/workflows/next.md +1 -1
- package/gsd-core/workflows/pause-work.md +1 -1
- package/gsd-core/workflows/plan-milestone-gaps.md +1 -1
- package/gsd-core/workflows/plan-phase.md +65 -28
- package/gsd-core/workflows/plan-review-convergence.md +60 -33
- package/gsd-core/workflows/plant-seed.md +1 -1
- package/gsd-core/workflows/profile-user.md +1 -1
- package/gsd-core/workflows/progress.md +1 -1
- package/gsd-core/workflows/quick.md +2 -2
- package/gsd-core/workflows/remove-phase.md +1 -1
- package/gsd-core/workflows/remove-workspace.md +1 -1
- package/gsd-core/workflows/resume-project.md +1 -1
- package/gsd-core/workflows/review.md +1 -1
- package/gsd-core/workflows/scan.md +1 -1
- package/gsd-core/workflows/secure-phase.md +1 -1
- package/gsd-core/workflows/settings-advanced.md +7 -7
- package/gsd-core/workflows/settings-integrations.md +1 -1
- package/gsd-core/workflows/settings.md +2 -2
- package/gsd-core/workflows/ship.md +8 -1
- package/gsd-core/workflows/sketch-wrap-up.md +1 -1
- package/gsd-core/workflows/sketch.md +1 -1
- package/gsd-core/workflows/spec-phase.md +130 -1
- package/gsd-core/workflows/spike-wrap-up.md +1 -1
- package/gsd-core/workflows/spike.md +1 -1
- package/gsd-core/workflows/stats.md +1 -1
- package/gsd-core/workflows/thread.md +1 -1
- package/gsd-core/workflows/transition.md +1 -1
- package/gsd-core/workflows/ui-phase.md +1 -1
- package/gsd-core/workflows/ui-review.md +1 -1
- package/gsd-core/workflows/ultraplan-phase.md +1 -1
- package/gsd-core/workflows/update.md +2 -2
- package/gsd-core/workflows/validate-phase.md +1 -1
- package/gsd-core/workflows/verify-phase.md +1 -1
- package/gsd-core/workflows/verify-work.md +8 -1
- package/package.json +11 -3
- package/scripts/base64-scan.sh +1 -1
- package/scripts/changeset/cli.cjs +8 -1
- package/scripts/changeset/lint.cjs +38 -2
- package/scripts/ci-test-scope.cjs +21 -10
- package/scripts/gen-capability-registry.cjs +2293 -0
- package/scripts/gen-loop-host-contract.cjs +471 -0
- package/scripts/lib/allowlist-ratchet.cjs +101 -1
- package/scripts/lint-regression-test-names.allowlist.json +269 -0
- package/scripts/lint-regression-test-names.cjs +117 -0
- package/scripts/lint-test-file-count.allowlist.json +25 -4
- package/scripts/lint-windows-test-portability.cjs +178 -0
- package/scripts/prompt-injection-scan.sh +4 -4
- package/scripts/research-profiles.cjs +10 -10
- package/scripts/run-tests.cjs +133 -29
- package/scripts/secret-scan.sh +3 -3
- package/scripts/sync-next-version.cjs +133 -0
- package/scripts/sync-runtime-launcher.cjs +21 -5
- package/scripts/update-size-baseline.cjs +68 -0
- package/scripts/workflow-policy.cjs +42 -9
- package/scripts/workflow-size.cjs +90 -0
- package/scripts/run-cross-platform-tests.cjs +0 -67
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
4
|
+
{ "requirement_id": "R1", "category": "encoding", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "Whose definition of length/equality applies — bytes, code points, grapheme clusters, or normalized form?" }
|
|
5
|
+
],
|
|
6
|
+
"coverage": { "applicable": 2, "resolved": 0, "unresolved": 2, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
7
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[{ "id": "R1", "text": "Truncate a display string to its first N characters" }]
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{ "requirement_id": "R1", "category": "boundary", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What happens exactly at each min/max/threshold — and one step either side?" },
|
|
4
|
+
{ "requirement_id": "R1", "category": "precision", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "Where can precision loss, overflow, or rounding/tie-breaking occur — and what is the exact contract (e.g. half-up vs half-to-even, ceil/floor/truncate)?" }
|
|
5
|
+
],
|
|
6
|
+
"coverage": { "applicable": 2, "resolved": 0, "unresolved": 2, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
7
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[{ "id": "R1", "text": "Compute the total price as an amount rounded to two decimals" }]
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When two things are exactly equal or just touch, do they merge, collide, or separate?" },
|
|
4
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
5
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When elements compare equal, is output order specified and stable?" }
|
|
6
|
+
],
|
|
7
|
+
"coverage": { "applicable": 3, "resolved": 0, "unresolved": 3, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
8
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[{ "id": "R1", "text": "Return all items in the list with duplicates removed" }]
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "resolved", "verification": "explicit", "resolution": "AC#6: touching intervals merge", "reason": null, "probe": "When two things are exactly equal or just touch, do they merge, collide, or separate?" },
|
|
4
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
5
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "dismissed", "verification": null, "resolution": null, "reason": "output is canonically sorted; no tie possible", "probe": "When elements compare equal, is output order specified and stable?" }
|
|
6
|
+
],
|
|
7
|
+
"coverage": { "applicable": 3, "resolved": 2, "unresolved": 1, "byVerification": { "explicit": 1, "backstop": 0 } }
|
|
8
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[{ "id": "R1", "text": "Merge a list of overlapping intervals into the minimal set" }]
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
[
|
|
2
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "resolved", "verification": "explicit", "resolution": "AC#6: touching intervals merge", "reason": null },
|
|
3
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "dismissed", "resolution": null, "reason": "output is canonically sorted; no tie possible" }
|
|
4
|
+
]
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
# Edge-Probe — Spec-Completeness Reference
|
|
2
|
+
|
|
3
|
+
Shared reference for the spec/requirements phase. Companion to
|
|
4
|
+
`@~/.claude/gsd-core/references/domain-probes.md`: `domain-probes` covers the
|
|
5
|
+
**technology axis** (auth, search, caching, deployment); this covers the
|
|
6
|
+
**data/behavior-shape axis** (boundaries, adjacency, encoding, ordering…). Walk each
|
|
7
|
+
requirement against the closed taxonomy below, propose a concrete candidate edge for each
|
|
8
|
+
applicable category, and resolve each to exactly one state. Adopt the established QA names
|
|
9
|
+
verbatim — this is decades-old black-box test technique, moved upstream to the spec layer.
|
|
10
|
+
|
|
11
|
+
This doc is written in generic `requirements → checks → verifier` terms with no
|
|
12
|
+
tool-specific vocabulary, so it is portable: copy it into any spec/requirements process.
|
|
13
|
+
A short mapping table at the end binds it to common host structures.
|
|
14
|
+
|
|
15
|
+
## Why front-of-pipeline
|
|
16
|
+
|
|
17
|
+
A goal-backward verifier only checks assertions that exist; an assertion only exists for a
|
|
18
|
+
requirement that was written down. A domain-boundary edge the author never surfaced is
|
|
19
|
+
invisible to the verifier — and worse, the verifier is *confidently wrong* about it
|
|
20
|
+
(measured: ~0.93 confidence while catching 0/12 omitted-edge defects, an expected
|
|
21
|
+
calibration error of 0.81 — worse than a coin flip — versus 100% / 0.03 on edges the spec
|
|
22
|
+
did state). The fix is not a better verifier or a confidence gate; confidence is
|
|
23
|
+
uninformative across these regimes. The fix is **spec completeness**: surface the omitted
|
|
24
|
+
edge into an explicit, checkable assertion *before* any code exists, after which the
|
|
25
|
+
verifier reliably catches it.
|
|
26
|
+
|
|
27
|
+
The underlying techniques are classic and should be named as such — **Boundary Value
|
|
28
|
+
Analysis**, **Equivalence Partitioning**, the **Category-Partition method**, and
|
|
29
|
+
**Metamorphic Relations**; property-based testing (PBT) is the academic name for the
|
|
30
|
+
held-out backstop. The literature applies these at the *test* layer (back of pipeline,
|
|
31
|
+
generating checks). The differentiated move here is **placement, not technique**: apply the
|
|
32
|
+
same edge taxonomy at the *spec* layer (front of pipeline, generating requirements), with
|
|
33
|
+
the explicit goal of extending a verifier's reach. Recent LLM spec work corroborates the
|
|
34
|
+
core finding that the spec layer is the measured weak point:
|
|
35
|
+
|
|
36
|
+
- SLD-Spec — program slicing + logical deletion (code→spec): arXiv 2509.09917
|
|
37
|
+
- Specine / AutoReSpec / SpecMind — spec alignment & postcondition inference
|
|
38
|
+
- CodeSpecBench (2604.12268), OSVBench (2504.20964), VERINA (2505.23135) — benchmarks
|
|
39
|
+
showing models solve tasks far better than they generate precise behavioral specs
|
|
40
|
+
- LLM property-based tests for edge cases: arXiv 2510.25297 (PBT+EBT ≈ 81% edge detection)
|
|
41
|
+
|
|
42
|
+
## Inputs
|
|
43
|
+
|
|
44
|
+
A list of requirements, each a `{ id, text, shapes? }` record where `text` is a testable
|
|
45
|
+
statement and `shapes` is an optional author-supplied override of the data/behavior shape.
|
|
46
|
+
The five shapes are: `numeric-range`, `collection`, `text`, `stateful`, `io`. When
|
|
47
|
+
`shapes` is absent, a heuristic classifier proposes them from the requirement prose
|
|
48
|
+
(propose-then-confirm) — the author may correct the shape.
|
|
49
|
+
|
|
50
|
+
## Taxonomy (8 categories)
|
|
51
|
+
|
|
52
|
+
Closed and small by design: a fixed eight the author must explicitly clear beats thirty
|
|
53
|
+
nobody finishes. The failure mode being eliminated was never "too few categories" — it was
|
|
54
|
+
that no taxonomy was *systematically applied at all*. Categories 1, 2, and 4 alone cover
|
|
55
|
+
the three canonical corpus defects (banker's-rounding ties, touching intervals, grapheme
|
|
56
|
+
truncation). Growth happens via optional domain packs, not by bloating the core.
|
|
57
|
+
|
|
58
|
+
| id | name (QA term) | applies to shapes | probe question |
|
|
59
|
+
|----|----------------|-------------------|----------------|
|
|
60
|
+
| boundary | Boundary values | numeric-range | What happens exactly at each min/max/threshold — and one step either side? |
|
|
61
|
+
| adjacency | Adjacency / touching | collection | When two things are exactly equal or just touch, do they merge, collide, or separate? |
|
|
62
|
+
| empty | Empty / degenerate | collection, text | What is the result for empty, single-element, or null input? |
|
|
63
|
+
| encoding | Encoding / representation | text | Whose definition of length/equality applies — bytes, code points, grapheme clusters, or normalized form? |
|
|
64
|
+
| ordering | Ordering / stability | collection | When elements compare equal, is output order specified and stable? |
|
|
65
|
+
| precision | Precision / overflow | numeric-range | Where can precision loss, overflow, or rounding/tie-breaking occur — and what is the exact contract (e.g. half-up vs half-to-even, ceil/floor/truncate)? |
|
|
66
|
+
| idempotency | Idempotency / repetition | stateful | What happens if this runs twice on the same input? |
|
|
67
|
+
| concurrency | Concurrency / effect ordering | stateful, io | If interrupted or run in parallel, what is guaranteed? |
|
|
68
|
+
|
|
69
|
+
## Relevance filter + resolution states
|
|
70
|
+
|
|
71
|
+
Two rules keep the probe honest and prevent an "everything is N/A" failure mode:
|
|
72
|
+
|
|
73
|
+
1. **Relevance filter first.** Classify each requirement's shape, then raise only the
|
|
74
|
+
categories whose `applies to shapes` intersect that requirement's shapes. A pure-text
|
|
75
|
+
requirement is never asked about overflow. This is what makes an unresolved edge
|
|
76
|
+
meaningful: it is an edge that *applies* and was not addressed.
|
|
77
|
+
2. **Dismissal requires a reason string.** "N/A — input is a bounded enum, no boundary
|
|
78
|
+
exists" is valid; silence is not. The reason string is the audit trail.
|
|
79
|
+
|
|
80
|
+
Each raised edge carries two orthogonal axes — a resolution **lifecycle** and, when
|
|
81
|
+
resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model):
|
|
82
|
+
|
|
83
|
+
- **status** — `resolved | dismissed | unresolved`:
|
|
84
|
+
- **resolved** — the edge is addressed; *how* it is addressed is the verification tier.
|
|
85
|
+
- **dismissed** — not applicable, accompanied by a required, non-empty reason string.
|
|
86
|
+
- **unresolved** — carried forward and flagged; the author chose not to resolve it yet.
|
|
87
|
+
- **verification** (only when `status` is `resolved`; `null` otherwise) — `explicit | backstop`:
|
|
88
|
+
- **explicit** — a checkable assertion for the edge is written (a SPEC acceptance criterion).
|
|
89
|
+
- **backstop** — a held-out / property-based test stands in for an edge the author knows
|
|
90
|
+
but cannot fully articulate in prose (records intent; the test body is authored later).
|
|
91
|
+
|
|
92
|
+
Splitting these axes keeps the lifecycle enum free of a verification fact (the old single
|
|
93
|
+
enum smuggled `covered`/`backstop` — both *resolved* — into one flat list) and lets sibling
|
|
94
|
+
probes add their own verification tiers (e.g. `test | judgment`) without a parallel enum.
|
|
95
|
+
|
|
96
|
+
`coverage.resolved` is the count of **closed** edges — `resolved` + `dismissed` (the
|
|
97
|
+
pre-re-cut "covered + dismissed + backstop" set, count-preserved). An `unresolved`
|
|
98
|
+
*applicable* edge is the precise signal a soft completeness gate raises.
|
|
99
|
+
|
|
100
|
+
## Output schema
|
|
101
|
+
|
|
102
|
+
The probe emits, per edge, an item of the form:
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
{ requirement_id, category, status, verification, resolution, reason, probe }
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
plus a coverage summary:
|
|
109
|
+
|
|
110
|
+
```
|
|
111
|
+
coverage: { applicable, resolved, unresolved, byVerification: { explicit, backstop } }
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
`applicable` is the number of raised edges, `resolved` = closed (`resolved` + `dismissed`)
|
|
115
|
+
status edges, `unresolved` is the remainder, and `byVerification` breaks the
|
|
116
|
+
`resolved`-status edges down by tier (probe-agnostic in core; the edge adapter declares
|
|
117
|
+
`{ explicit, backstop }`). This JSON is the stable contract both the reference
|
|
118
|
+
implementation and any third-party port emit.
|
|
119
|
+
|
|
120
|
+
## Generic mapping (requirements → checks → verifier)
|
|
121
|
+
|
|
122
|
+
| Host structure | "requirement" | a `resolved`/`explicit` edge becomes | a `resolved`/`backstop` edge becomes |
|
|
123
|
+
|----------------|---------------|--------------------------|---------------------------|
|
|
124
|
+
| GSD SPEC | a SPEC Requirement | an Acceptance Criterion that `plan-phase` lifts into `must_haves.truths` | a non-inferable check in `must_haves.truths` (needs a held-out/PBT test) |
|
|
125
|
+
| Gherkin feature | a Scenario | an additional `Then` assertion / Scenario Outline row | a tagged scenario routed to a property test |
|
|
126
|
+
| OpenAPI operation | an operation | a response/constraint example + schema rule | a contract/property test on the operation |
|
|
127
|
+
| Docstring contract | a documented behavior | an assertion in the contract test | a property-based test for the function |
|
|
128
|
+
|
|
129
|
+
The portable invariant: a `resolved`/`explicit` edge produces **the unit your verifier
|
|
130
|
+
iterates over** (GSD: a `must_haves.truth`); a `resolved`/`backstop` edge produces a test
|
|
131
|
+
added to that same set as a non-inferable check. An `unresolved` edge is an explicit
|
|
132
|
+
assumption the downstream planner must surface, not silently drop.
|
|
133
|
+
|
|
134
|
+
## Worked example (merge-intervals)
|
|
135
|
+
|
|
136
|
+
Given a single requirement with no resolutions yet:
|
|
137
|
+
|
|
138
|
+
```
|
|
139
|
+
[{ "id": "R1", "text": "Merge a list of overlapping intervals into the minimal set" }]
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
the requirement classifies as a `collection`, which raises `adjacency`, `empty`, and
|
|
143
|
+
`ordering` (but not `boundary`/`precision`/`encoding`/`idempotency`/`concurrency`). With no
|
|
144
|
+
resolutions supplied, every applicable edge is `unresolved`:
|
|
145
|
+
|
|
146
|
+
```json edge-probe:02-merge-intervals/expected-coverage.json
|
|
147
|
+
{
|
|
148
|
+
"items": [
|
|
149
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When two things are exactly equal or just touch, do they merge, collide, or separate?" },
|
|
150
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
151
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When elements compare equal, is output order specified and stable?" }
|
|
152
|
+
],
|
|
153
|
+
"coverage": { "applicable": 3, "resolved": 0, "unresolved": 3, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
154
|
+
}
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
The `adjacency` row is the one that catches the canonical defect: `[[1,2],[2,3]]` intervals
|
|
158
|
+
that only *touch* — does the spec say they merge? Resolving it `resolved`/`explicit` writes
|
|
159
|
+
that assertion, and the verifier can then enforce it. This worked-example block is kept
|
|
160
|
+
byte-for-byte (parsed-JSON) identical to its fixture by `edge-probe-docs-fixtures.test.cjs`,
|
|
161
|
+
so the doc and the reference implementation cannot silently drift.
|
|
162
|
+
|
|
163
|
+
## Worked example (round-half-even)
|
|
164
|
+
|
|
165
|
+
Given a requirement to round floating-point values using the banker's-rounding (half-even)
|
|
166
|
+
rule, the requirement classifies as `numeric-range`, which raises `boundary` and `precision`
|
|
167
|
+
(but not `adjacency`/`empty`/`ordering`/`encoding`/`idempotency`/`concurrency`):
|
|
168
|
+
|
|
169
|
+
```json edge-probe:01-round-half-even/expected-coverage.json
|
|
170
|
+
{
|
|
171
|
+
"items": [
|
|
172
|
+
{ "requirement_id": "R1", "category": "boundary", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What happens exactly at each min/max/threshold — and one step either side?" },
|
|
173
|
+
{ "requirement_id": "R1", "category": "precision", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "Where can precision loss, overflow, or rounding/tie-breaking occur — and what is the exact contract (e.g. half-up vs half-to-even, ceil/floor/truncate)?" }
|
|
174
|
+
],
|
|
175
|
+
"coverage": { "applicable": 2, "resolved": 0, "unresolved": 2, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
176
|
+
}
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
The `precision` row surfaces the canonical defect for this requirement: IEEE 754 floating-point
|
|
180
|
+
arithmetic rounds 2.5 to 2 (not 3) under half-even — the probe asks whether the spec
|
|
181
|
+
states that contract explicitly, so the verifier can enforce it.
|
|
182
|
+
|
|
183
|
+
## Worked example (truncate-graphemes)
|
|
184
|
+
|
|
185
|
+
Given a requirement to truncate a string to N grapheme clusters (not bytes or code points),
|
|
186
|
+
the requirement classifies as `text`, which raises `empty` and `encoding`:
|
|
187
|
+
|
|
188
|
+
```json edge-probe:03-truncate-graphemes/expected-coverage.json
|
|
189
|
+
{
|
|
190
|
+
"items": [
|
|
191
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
192
|
+
{ "requirement_id": "R1", "category": "encoding", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "Whose definition of length/equality applies — bytes, code points, grapheme clusters, or normalized form?" }
|
|
193
|
+
],
|
|
194
|
+
"coverage": { "applicable": 2, "resolved": 0, "unresolved": 2, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
195
|
+
}
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
The `encoding` row is the load-bearing one: a requirement that says "truncate to 10
|
|
199
|
+
characters" is ambiguous — the spec must state whether "character" means bytes, UTF-16
|
|
200
|
+
code units, Unicode code points, or grapheme clusters (emoji sequences are 1 grapheme,
|
|
201
|
+
multiple code points).
|
|
202
|
+
|
|
203
|
+
## Worked example (money-rounding)
|
|
204
|
+
|
|
205
|
+
Given a requirement to round monetary amounts to two decimal places, the requirement
|
|
206
|
+
classifies as `numeric-range`, which raises `boundary` and `precision`:
|
|
207
|
+
|
|
208
|
+
```json edge-probe:04-money-rounding/expected-coverage.json
|
|
209
|
+
{
|
|
210
|
+
"items": [
|
|
211
|
+
{ "requirement_id": "R1", "category": "boundary", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What happens exactly at each min/max/threshold — and one step either side?" },
|
|
212
|
+
{ "requirement_id": "R1", "category": "precision", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "Where can precision loss, overflow, or rounding/tie-breaking occur — and what is the exact contract (e.g. half-up vs half-to-even, ceil/floor/truncate)?" }
|
|
213
|
+
],
|
|
214
|
+
"coverage": { "applicable": 2, "resolved": 0, "unresolved": 2, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
215
|
+
}
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
The `boundary` row asks what the minimum and maximum representable values are (negative
|
|
219
|
+
amounts? fractional cents?). The `precision` row asks whether IEEE 754 binary rounding
|
|
220
|
+
can produce `$0.30000000000000004` — the spec must commit to a representation.
|
|
221
|
+
|
|
222
|
+
## Worked example (list-dedupe)
|
|
223
|
+
|
|
224
|
+
Given a requirement to deduplicate a list of items, the requirement classifies as
|
|
225
|
+
`collection`, which raises `adjacency`, `empty`, and `ordering`:
|
|
226
|
+
|
|
227
|
+
```json edge-probe:05-list-dedupe/expected-coverage.json
|
|
228
|
+
{
|
|
229
|
+
"items": [
|
|
230
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When two things are exactly equal or just touch, do they merge, collide, or separate?" },
|
|
231
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
232
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "When elements compare equal, is output order specified and stable?" }
|
|
233
|
+
],
|
|
234
|
+
"coverage": { "applicable": 3, "resolved": 0, "unresolved": 3, "byVerification": { "explicit": 0, "backstop": 0 } }
|
|
235
|
+
}
|
|
236
|
+
```
|
|
237
|
+
|
|
238
|
+
The `ordering` row is the non-obvious one: dedupe removes duplicates, but which copy is
|
|
239
|
+
kept — first occurrence, last, or implementation-defined? The spec must commit.
|
|
240
|
+
|
|
241
|
+
## Worked example (resolved-mixed)
|
|
242
|
+
|
|
243
|
+
The same merge-intervals requirement after a resolution session where `adjacency` was
|
|
244
|
+
resolved with an explicit acceptance criterion, `ordering` was dismissed, and `empty` was
|
|
245
|
+
left unresolved:
|
|
246
|
+
|
|
247
|
+
```json edge-probe:06-resolved-mixed/expected-coverage.json
|
|
248
|
+
{
|
|
249
|
+
"items": [
|
|
250
|
+
{ "requirement_id": "R1", "category": "adjacency", "status": "resolved", "verification": "explicit", "resolution": "AC#6: touching intervals merge", "reason": null, "probe": "When two things are exactly equal or just touch, do they merge, collide, or separate?" },
|
|
251
|
+
{ "requirement_id": "R1", "category": "empty", "status": "unresolved", "verification": null, "resolution": null, "reason": null, "probe": "What is the result for empty, single-element, or null input?" },
|
|
252
|
+
{ "requirement_id": "R1", "category": "ordering", "status": "dismissed", "verification": null, "resolution": null, "reason": "output is canonically sorted; no tie possible", "probe": "When elements compare equal, is output order specified and stable?" }
|
|
253
|
+
],
|
|
254
|
+
"coverage": { "applicable": 3, "resolved": 2, "unresolved": 1, "byVerification": { "explicit": 1, "backstop": 0 } }
|
|
255
|
+
}
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
`coverage.resolved` is 2 — the closed set (adjacency=`resolved`/`explicit` + ordering=`dismissed`);
|
|
259
|
+
`unresolved` is 1 (empty); `byVerification.explicit` is 1 (the single explicitly-verified
|
|
260
|
+
edge), `backstop` 0. The soft gate raises on this example because one applicable edge remains
|
|
261
|
+
unresolved — the author must either specify, dismiss, or backstop it before writing the SPEC.
|
|
@@ -87,3 +87,44 @@ A plan should not interleave multiple checkpoint types with implementation tasks
|
|
|
87
87
|
- "will be wired later", "dynamic in future phase", "skip for now"
|
|
88
88
|
|
|
89
89
|
If a decision from CONTEXT.md says "display cost calculated from billing table in impulses", the plan must deliver exactly that. Not "static label /min" as a "v1". If the phase is too complex, recommend a phase split instead of silently reducing scope.
|
|
90
|
+
|
|
91
|
+
## Comment-Text Discipline (HARD GATE)
|
|
92
|
+
|
|
93
|
+
> Enforced at plan-write time by `verify.plan-structure` (the `validate_plan` step). Issue #429.
|
|
94
|
+
|
|
95
|
+
When an `<acceptance_criteria>` or `<verify>` block uses a **negative grep** — `grep -c 'LITERAL' file == 0`, meaning "this literal must NOT appear in the file" — that same `LITERAL` must not appear verbatim anywhere in an `<action>` body. Verbatim code blocks, JSDoc samples, head-comment references, and "what NOT to do" illustrations get echoed into the file the executor writes, so the executor's commit-time gate fails on the *comment text*, not on a real code regression. The work is correct; the gate output is semantically wrong; the executor wastes cycles and learns to distrust the gate.
|
|
96
|
+
|
|
97
|
+
**The gate:** plan creation FAILS (error, `valid: false`) when a confidently-extracted (quoted) negative-grep literal also appears in an `<action>` block. When the grep literal is unquoted and cannot be extracted unambiguously, the gate WARNS instead of failing (so you still get the plan, with the risk surfaced).
|
|
98
|
+
|
|
99
|
+
### Bad — JSDoc sample echoes the forbidden literal
|
|
100
|
+
|
|
101
|
+
```xml
|
|
102
|
+
<task>
|
|
103
|
+
<action>
|
|
104
|
+
Add a `?from=` query param to the share link. Do NOT reintroduce the old
|
|
105
|
+
`?from=` referrer hack the JSDoc warned about. <!-- echoes ?from= -->
|
|
106
|
+
</action>
|
|
107
|
+
<verify><automated>grep -c '?from=' src/animal-detail.tsx == 0</automated></verify>
|
|
108
|
+
</task>
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Good — rephrase the comment by concept
|
|
112
|
+
|
|
113
|
+
```xml
|
|
114
|
+
<task>
|
|
115
|
+
<action>
|
|
116
|
+
Add the share-link query param. Do NOT reintroduce the legacy referrer hack.
|
|
117
|
+
</action>
|
|
118
|
+
<verify><automated>grep -c '?from=' src/animal-detail.tsx == 0</automated></verify>
|
|
119
|
+
</task>
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
### Allowlist escape hatch
|
|
123
|
+
|
|
124
|
+
When the literal MUST appear in the plan body verbatim — e.g. the plan documents the test file that exercises the gate itself, or the literal is part of the verification command's own grep regex — add a marker on its own line so the gate skips that literal:
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
<!-- planner-discipline-allow: ?from= -->
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
One marker per literal. The marker exempts only the exact literal it names.
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
# Planner Guidance: Philosophy, Task Calibration, and Output Formats
|
|
2
|
+
|
|
3
|
+
## Solo Developer + Claude Workflow
|
|
4
|
+
|
|
5
|
+
Planning for ONE person (the user) and ONE implementer (Claude).
|
|
6
|
+
- No teams, stakeholders, ceremonies, coordination overhead
|
|
7
|
+
- User = visionary/product owner, Claude = builder
|
|
8
|
+
- Estimate effort in context window cost, not time
|
|
9
|
+
|
|
10
|
+
## Plans Are Prompts
|
|
11
|
+
|
|
12
|
+
PLAN.md IS the prompt (not a document that becomes one). Contains:
|
|
13
|
+
- Objective (what and why)
|
|
14
|
+
- Context (@file references)
|
|
15
|
+
- Tasks (with verification criteria)
|
|
16
|
+
- Success criteria (measurable)
|
|
17
|
+
|
|
18
|
+
## Quality Degradation Curve
|
|
19
|
+
|
|
20
|
+
| Context Usage | Quality | Claude's State |
|
|
21
|
+
|---------------|---------|----------------|
|
|
22
|
+
| 0-30% | PEAK | Thorough, comprehensive |
|
|
23
|
+
| 30-50% | GOOD | Confident, solid work |
|
|
24
|
+
| 50-70% | DEGRADING | Efficiency mode begins |
|
|
25
|
+
| 70%+ | POOR | Rushed, minimal |
|
|
26
|
+
|
|
27
|
+
**Rule:** Plans should complete within ~50% context. More plans, smaller scope, consistent quality. Each plan: 2-3 tasks max.
|
|
28
|
+
|
|
29
|
+
## Ship Fast
|
|
30
|
+
|
|
31
|
+
Plan -> Execute -> Ship -> Learn -> Repeat
|
|
32
|
+
|
|
33
|
+
**Anti-enterprise patterns (delete if seen):** team structures, RACI matrices, sprint ceremonies, time estimates in human units, complexity/difficulty as scope justification, documentation for documentation's sake.
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## Task Types
|
|
38
|
+
|
|
39
|
+
| Type | Use For | Autonomy |
|
|
40
|
+
|------|---------|----------|
|
|
41
|
+
| `auto` | Everything Claude can do independently | Fully autonomous |
|
|
42
|
+
| `checkpoint:human-verify` | Visual/functional verification | Pauses for user |
|
|
43
|
+
| `checkpoint:decision` | Implementation choices | Pauses for user |
|
|
44
|
+
| `checkpoint:human-action` | Truly unavoidable manual steps (rare) | Pauses for user |
|
|
45
|
+
|
|
46
|
+
**Automation-first rule:** If Claude CAN do it via CLI/API, Claude MUST do it. Checkpoints verify AFTER automation, not replace it.
|
|
47
|
+
|
|
48
|
+
## Task Sizing
|
|
49
|
+
|
|
50
|
+
Each task targets **10–30% context consumption**.
|
|
51
|
+
|
|
52
|
+
| Context Cost | Action |
|
|
53
|
+
|--------------|--------|
|
|
54
|
+
| < 10% context | Too small — combine with a related task |
|
|
55
|
+
| 10-30% context | Right size — proceed |
|
|
56
|
+
| > 30% context | Too large — split into two tasks |
|
|
57
|
+
|
|
58
|
+
**Context cost signals (use these, not time estimates):**
|
|
59
|
+
- Files modified: 0-3 = ~10-15%, 4-6 = ~20-30%, 7+ = ~40%+ (split)
|
|
60
|
+
- New subsystem: ~25-35%
|
|
61
|
+
- Migration + data transform: ~30-40%
|
|
62
|
+
- Pure config/wiring: ~5-10%
|
|
63
|
+
|
|
64
|
+
**Too large signals:** Touches >3-5 files, multiple distinct chunks, action section >1 paragraph.
|
|
65
|
+
|
|
66
|
+
**Combine signals:** One task sets up for the next, separate tasks touch same file, neither meaningful alone.
|
|
67
|
+
|
|
68
|
+
## Interface-First Task Ordering
|
|
69
|
+
|
|
70
|
+
When a plan creates new interfaces consumed by subsequent tasks:
|
|
71
|
+
|
|
72
|
+
1. **First task: Define contracts** — Create type files, interfaces, exports
|
|
73
|
+
2. **Middle tasks: Implement** — Build against the defined contracts
|
|
74
|
+
3. **Last task: Wire** — Connect implementations to consumers
|
|
75
|
+
|
|
76
|
+
This prevents the "scavenger hunt" anti-pattern where executors explore the codebase to understand contracts. They receive the contracts in the plan itself.
|
|
77
|
+
|
|
78
|
+
## Specificity
|
|
79
|
+
|
|
80
|
+
**Test:** Could a different Claude instance execute without asking clarifying questions? If not, add specificity. See @~/.claude/gsd-core/references/planner-antipatterns.md for vague-vs-specific comparison table.
|
|
81
|
+
|
|
82
|
+
## User Setup Detection
|
|
83
|
+
|
|
84
|
+
For tasks involving external services, identify human-required configuration:
|
|
85
|
+
|
|
86
|
+
External service indicators: New SDK (`stripe`, `@sendgrid/mail`, `twilio`, `openai`), webhook handlers, OAuth integration, `process.env.SERVICE_*` patterns.
|
|
87
|
+
|
|
88
|
+
For each external service, determine:
|
|
89
|
+
1. **Env vars needed** — What secrets from dashboards?
|
|
90
|
+
2. **Account setup** — Does user need to create an account?
|
|
91
|
+
3. **Dashboard config** — What must be configured in external UI?
|
|
92
|
+
|
|
93
|
+
Record in `user_setup` frontmatter. Only include what Claude literally cannot do. Do NOT surface in planning output — execute-plan handles presentation.
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Building the Dependency Graph
|
|
98
|
+
|
|
99
|
+
**For each task, record:**
|
|
100
|
+
- `needs`: What must exist before this runs
|
|
101
|
+
- `creates`: What this produces
|
|
102
|
+
- `has_checkpoint`: Requires user interaction?
|
|
103
|
+
|
|
104
|
+
**Example:** A→C, B→D, C+D→E, E→F(checkpoint). Waves: {A,B} → {C,D} → {E} → {F}.
|
|
105
|
+
|
|
106
|
+
**Prefer vertical slices** (User feature: model+API+UI) over horizontal layers (all models → all APIs → all UIs). Vertical = parallel. Horizontal = sequential. Use horizontal only when shared foundation is required.
|
|
107
|
+
|
|
108
|
+
## File Ownership for Parallel Execution
|
|
109
|
+
|
|
110
|
+
Exclusive file ownership prevents conflicts:
|
|
111
|
+
|
|
112
|
+
```yaml
|
|
113
|
+
# Plan 01 frontmatter
|
|
114
|
+
files_modified: [src/models/user.ts, src/api/users.ts]
|
|
115
|
+
|
|
116
|
+
# Plan 02 frontmatter (no overlap = parallel)
|
|
117
|
+
files_modified: [src/models/product.ts, src/api/products.ts]
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
No overlap → can run parallel. File in multiple plans → later plan depends on earlier.
|
|
121
|
+
|
|
122
|
+
---
|
|
123
|
+
|
|
124
|
+
## Granularity Calibration
|
|
125
|
+
|
|
126
|
+
The resolved granularity is provided in the planning context as `**Granularity:** <value>`. Read that value and apply the corresponding row below. When no explicit value is present, default to Standard.
|
|
127
|
+
|
|
128
|
+
| Granularity | Typical Plans/Phase | Tasks/Plan |
|
|
129
|
+
|-------------|---------------------|------------|
|
|
130
|
+
| Coarse | 1-3 | 2-3 |
|
|
131
|
+
| Standard | 3-5 | 2-3 |
|
|
132
|
+
| Fine | 5-10 | 2-3 |
|
|
133
|
+
|
|
134
|
+
Derive plans from actual work. Granularity determines compression tolerance, not a target.
|
|
135
|
+
|
|
136
|
+
---
|
|
137
|
+
|
|
138
|
+
## Planning Complete Return Format
|
|
139
|
+
|
|
140
|
+
```markdown
|
|
141
|
+
## PLANNING COMPLETE
|
|
142
|
+
|
|
143
|
+
**Phase:** {phase-name}
|
|
144
|
+
**Plans:** {N} plan(s) in {M} wave(s)
|
|
145
|
+
|
|
146
|
+
### Wave Structure
|
|
147
|
+
|
|
148
|
+
| Wave | Plans | Autonomous |
|
|
149
|
+
|------|-------|------------|
|
|
150
|
+
| 1 | {plan-01}, {plan-02} | yes, yes |
|
|
151
|
+
| 2 | {plan-03} | no (has checkpoint) |
|
|
152
|
+
|
|
153
|
+
### Plans Created
|
|
154
|
+
|
|
155
|
+
| Plan | Objective | Tasks | Files |
|
|
156
|
+
|------|-----------|-------|-------|
|
|
157
|
+
| {phase}-01 | [brief] | 2 | [files] |
|
|
158
|
+
| {phase}-02 | [brief] | 3 | [files] |
|
|
159
|
+
|
|
160
|
+
### Next Steps
|
|
161
|
+
|
|
162
|
+
Run `/clear` first for a fresh context window, then execute: `/gsd:execute-phase {phase}`
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
## Gap Closure Plans Created Return Format
|
|
166
|
+
|
|
167
|
+
```markdown
|
|
168
|
+
## GAP CLOSURE PLANS CREATED
|
|
169
|
+
|
|
170
|
+
**Phase:** {phase-name}
|
|
171
|
+
**Closing:** {N} gaps from {VERIFICATION|UAT}.md
|
|
172
|
+
|
|
173
|
+
### Plans
|
|
174
|
+
|
|
175
|
+
| Plan | Gaps Addressed | Files |
|
|
176
|
+
|------|----------------|-------|
|
|
177
|
+
| {phase}-04 | [gap truths] | [files] |
|
|
178
|
+
|
|
179
|
+
### Next Steps
|
|
180
|
+
|
|
181
|
+
Execute: `/gsd:execute-phase {phase} --gaps-only`
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
## Checkpoint Reached / Revision Complete
|
|
185
|
+
|
|
186
|
+
Follow templates in checkpoints and revision_mode sections respectively.
|
|
@@ -4,6 +4,8 @@ Triggered when orchestrator sets Mode to `reviews`. Replanning from scratch with
|
|
|
4
4
|
|
|
5
5
|
**Mindset:** Fresh planner with review insights — not a surgeon making patches, but an architect who has read peer critiques.
|
|
6
6
|
|
|
7
|
+
**Execution contract:** REVIEWS.md is audit trail and feedback input, not a second execution contract. /gsd:execute-phase primarily consumes PLAN.md plus the normal phase context. Every current actionable review finding must therefore be incorporated into the relevant PLAN.md or explicitly deferred/rejected in that PLAN.md.
|
|
8
|
+
|
|
7
9
|
### Step 1: Load REVIEWS.md
|
|
8
10
|
Read the reviews file from `<files_to_read>`. Parse:
|
|
9
11
|
- Per-reviewer feedback (strengths, concerns, suggestions)
|
|
@@ -13,13 +15,14 @@ Read the reviews file from `<files_to_read>`. Parse:
|
|
|
13
15
|
### Step 2: Categorize Feedback
|
|
14
16
|
Group review feedback into:
|
|
15
17
|
- **Must address**: HIGH severity consensus concerns
|
|
16
|
-
- **
|
|
18
|
+
- **Must represent in PLAN.md**: actionable MEDIUM/LOW findings that require task, action, acceptance criteria, verify command, must_haves, threat-model, artifact, stale-path, or execution-contract changes
|
|
19
|
+
- **Should address**: MEDIUM severity concerns from 2+ reviewers that improve quality but do not change the executable contract
|
|
17
20
|
- **Consider**: Individual reviewer suggestions, LOW severity items
|
|
18
21
|
|
|
19
22
|
### Step 3: Plan Fresh with Review Context
|
|
20
23
|
Create new plans following the standard planning process, but with review feedback as additional constraints:
|
|
21
24
|
- Each HIGH severity consensus concern MUST have a task that addresses it
|
|
22
|
-
- MEDIUM
|
|
25
|
+
- Each current actionable MEDIUM/LOW finding MUST either appear in the relevant PLAN.md executable content or have a deferral/rejection rationale in that PLAN.md
|
|
23
26
|
- Note in task actions: "Addresses review concern: {concern}" for traceability
|
|
24
27
|
|
|
25
28
|
### Step 4: Return
|
|
@@ -568,12 +568,12 @@ must_haves:
|
|
|
568
568
|
contains: "model Message"
|
|
569
569
|
key_links:
|
|
570
570
|
- from: "src/components/Chat.tsx"
|
|
571
|
-
to: "/api/chat"
|
|
572
|
-
via: "fetch in useEffect"
|
|
571
|
+
to: "src/app/api/chat/route.ts"
|
|
572
|
+
via: "fetch in useEffect — calls /api/chat endpoint"
|
|
573
573
|
pattern: "fetch.*api/chat"
|
|
574
574
|
- from: "src/app/api/chat/route.ts"
|
|
575
|
-
to: "prisma.
|
|
576
|
-
via: "database query"
|
|
575
|
+
to: "prisma/schema.prisma"
|
|
576
|
+
via: "database query via prisma.message"
|
|
577
577
|
pattern: "prisma\\.message\\.(find|create)"
|
|
578
578
|
```
|
|
579
579
|
|
|
@@ -589,9 +589,9 @@ must_haves:
|
|
|
589
589
|
| `artifacts[].exports` | Optional. Expected exports to verify. |
|
|
590
590
|
| `artifacts[].contains` | Optional. Pattern that must exist in file. |
|
|
591
591
|
| `key_links` | Critical connections between artifacts. |
|
|
592
|
-
| `key_links[].from` | Source
|
|
593
|
-
| `key_links[].to` | Target
|
|
594
|
-
| `key_links[].via` | How they connect (
|
|
592
|
+
| `key_links[].from` | Source file (relative path from project root). Describe components or symbols in `via:`. |
|
|
593
|
+
| `key_links[].to` | Target file (relative path from project root). Describe endpoints, APIs, or modules in `via:`. |
|
|
594
|
+
| `key_links[].via` | How they connect, including any endpoint or symbol name (e.g. `fetch in useEffect — calls /api/chat`, `Prisma query via prisma.message`). |
|
|
595
595
|
| `key_links[].pattern` | Optional. Regex to verify connection exists. |
|
|
596
596
|
|
|
597
597
|
**Why this matters:**
|