@opengsd/gsd-core 1.5.0-rc.3 → 1.5.0-rc.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/agents/gsd-advisor-researcher.md +1 -1
- package/agents/gsd-assumptions-analyzer.md +1 -1
- package/agents/gsd-code-fixer.md +1 -1
- package/agents/gsd-code-reviewer.md +1 -1
- package/agents/gsd-codebase-mapper.md +1 -1
- package/agents/gsd-debugger.md +1 -1
- package/agents/gsd-doc-writer.md +1 -1
- package/agents/gsd-eval-auditor.md +1 -1
- package/agents/gsd-executor.md +1 -1
- package/agents/gsd-integration-checker.md +1 -1
- package/agents/gsd-nyquist-auditor.md +1 -0
- package/agents/gsd-phase-researcher.md +1 -1
- package/agents/gsd-plan-checker.md +1 -1
- package/agents/gsd-planner.md +1 -1
- package/agents/gsd-project-researcher.md +1 -1
- package/agents/gsd-research-synthesizer.md +1 -1
- package/agents/gsd-roadmapper.md +55 -2
- package/agents/gsd-security-auditor.md +1 -0
- package/agents/gsd-ui-auditor.md +1 -1
- package/agents/gsd-ui-checker.md +1 -1
- package/agents/gsd-ui-researcher.md +1 -1
- package/agents/gsd-verifier.md +13 -2
- package/bin/install.js +36 -57
- package/commands/gsd/progress.md +2 -1
- package/gemini-extension.json +1 -1
- package/gsd-core/bin/gsd-tools.cjs +167 -3
- package/gsd-core/bin/lib/active-workstream-store.cjs +6 -0
- package/gsd-core/bin/lib/capability-state.cjs +97 -3
- package/gsd-core/bin/lib/capability-writer.cjs +354 -0
- package/gsd-core/bin/lib/config.cjs +80 -24
- package/gsd-core/bin/lib/edge-probe.cjs +25 -2
- package/gsd-core/bin/lib/frontmatter.cjs +53 -1
- package/gsd-core/bin/lib/git-base-branch.cjs +194 -0
- package/gsd-core/bin/lib/init.cjs +28 -6
- package/gsd-core/bin/lib/install-profiles.cjs +55 -0
- package/gsd-core/bin/lib/installer-migration-report.cjs +1 -0
- package/gsd-core/bin/lib/phase.cjs +28 -12
- package/gsd-core/bin/lib/plan-drift-guard.cjs +117 -0
- package/gsd-core/bin/lib/probe-core.cjs +117 -1
- package/gsd-core/bin/lib/roadmap-parser.cjs +13 -3
- package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +246 -0
- package/gsd-core/bin/lib/runtime-artifact-layout.cjs +34 -2
- package/gsd-core/bin/lib/state.cjs +240 -59
- package/gsd-core/bin/lib/verify.cjs +73 -4
- package/gsd-core/bin/lib/worktree-safety.cjs +2 -1
- package/gsd-core/references/edge-probe.md +11 -0
- package/gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json +14 -0
- package/gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json +4 -0
- package/gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json +32 -0
- package/gsd-core/references/prohibition-probe.md +248 -0
- package/gsd-core/templates/spec.md +14 -0
- package/gsd-core/workflows/complete-milestone.md +1 -5
- package/gsd-core/workflows/execute-phase.md +4 -3
- package/gsd-core/workflows/execute-plan.md +21 -6
- package/gsd-core/workflows/help/modes/full.md +4 -0
- package/gsd-core/workflows/next.md +50 -2
- package/gsd-core/workflows/pause-work.md +7 -1
- package/gsd-core/workflows/plan-phase.md +2 -0
- package/gsd-core/workflows/plan-review-convergence.md +14 -4
- package/gsd-core/workflows/pr-branch.md +4 -2
- package/gsd-core/workflows/quick.md +3 -2
- package/gsd-core/workflows/resume-project.md +17 -1
- package/gsd-core/workflows/settings.md +27 -1
- package/gsd-core/workflows/ship.md +1 -5
- package/gsd-core/workflows/spec-phase.md +75 -0
- package/gsd-core/workflows/verify-phase.md +14 -4
- package/hooks/dist/gsd-ensure-canonical-path.js +305 -0
- package/hooks/dist/gsd-statusline.js +1 -1
- package/hooks/dist/managed-hooks-registry.cjs +1 -0
- package/hooks/gsd-ensure-canonical-path.js +305 -0
- package/hooks/gsd-statusline.js +1 -1
- package/hooks/hooks.json +1 -0
- package/hooks/managed-hooks-registry.cjs +1 -0
- package/package.json +3 -3
- package/scripts/build-hooks.js +7 -0
- package/scripts/changeset/new.cjs +17 -3
- package/scripts/fix-slash-commands.cjs +15 -3
- package/scripts/gen-capability-registry.cjs +14 -1
- package/scripts/lint-allow-test-rule-refs.allowlist.json +2 -1
- package/scripts/lint-test-file-count.allowlist.json +6 -0
- package/scripts/mutation-matrix.cjs +108 -7
- package/scripts/pr-target-policy.cjs +63 -0
- package/scripts/research-profiles.cjs +5 -5
- package/scripts/run-tests.cjs +107 -6
|
@@ -701,6 +701,7 @@ function reapOrphanWorktrees(repoRoot, deps = {}) {
|
|
|
701
701
|
const readFileSafe = deps.readFileSafe || defaultReadFileSafe;
|
|
702
702
|
const mtimeSafe = deps.mtimeSafe || defaultMtimeSafe;
|
|
703
703
|
const reapMtimeGuardMs = deps.reapMtimeGuardMs !== undefined ? deps.reapMtimeGuardMs : REAP_MTIME_GUARD_MS;
|
|
704
|
+
const nowMs = deps.nowMs ?? Date.now();
|
|
704
705
|
const results = [];
|
|
705
706
|
// 1. Discover the .git/worktrees/ admin directory.
|
|
706
707
|
const gitDir = execGit(['rev-parse', '--git-dir'], { cwd: repoRoot });
|
|
@@ -803,7 +804,7 @@ function reapOrphanWorktrees(repoRoot, deps = {}) {
|
|
|
803
804
|
}
|
|
804
805
|
// 4a. Stale-lock guard: skip if lock is too fresh (PID recycling / race).
|
|
805
806
|
const lockMtime = mtimeSafe(lockedFile);
|
|
806
|
-
if (!lockMtime ||
|
|
807
|
+
if (!lockMtime || nowMs - lockMtime.getTime() < reapMtimeGuardMs) {
|
|
807
808
|
results.push({ path: worktreePath, status: 'skipped', reason: 'lock_too_fresh' });
|
|
808
809
|
continue;
|
|
809
810
|
}
|
|
@@ -77,6 +77,17 @@ Two rules keep the probe honest and prevent an "everything is N/A" failure mode:
|
|
|
77
77
|
2. **Dismissal requires a reason string.** "N/A — input is a bounded enum, no boundary
|
|
78
78
|
exists" is valid; silence is not. The reason string is the audit trail.
|
|
79
79
|
|
|
80
|
+
**Zero-classification surfaces an `unclassified` candidate (#1110).** The relevance filter is
|
|
81
|
+
a heuristic over prose cues, so a requirement whose wording *is* edge-relevant but matches no
|
|
82
|
+
shape cue would otherwise classify to zero shapes → zero edges and vanish from coverage with
|
|
83
|
+
no signal — the same silent blind spot the probe exists to catch. Instead, a requirement with
|
|
84
|
+
non-empty prose, no authored `shapes`, and zero matched shapes surfaces exactly one soft
|
|
85
|
+
`unclassified — review manually` candidate (`category: "unclassified"`, `status: "unresolved"`).
|
|
86
|
+
It is a dismissible nudge — resolve it, or dismiss it with a reason (e.g. a genuinely edge-free
|
|
87
|
+
static-asset requirement) — never a hard block. `unclassified` is a review signal, **not** a
|
|
88
|
+
ninth taxonomy category: the closed eight above are unchanged, and an explicit `shapes: []`
|
|
89
|
+
opt-out stays silent (the author's deliberate "no edge surface").
|
|
90
|
+
|
|
80
91
|
Each raised edge carries two orthogonal axes — a resolution **lifecycle** and, when
|
|
81
92
|
resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model):
|
|
82
93
|
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{
|
|
4
|
+
"requirement_id": "R1",
|
|
5
|
+
"category": "values",
|
|
6
|
+
"status": "resolved",
|
|
7
|
+
"verification": "judgment",
|
|
8
|
+
"resolution": null,
|
|
9
|
+
"reason": null,
|
|
10
|
+
"statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing"
|
|
11
|
+
}
|
|
12
|
+
],
|
|
13
|
+
"coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } }
|
|
14
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
{
|
|
2
|
+
"items": [
|
|
3
|
+
{
|
|
4
|
+
"requirement_id": "R1",
|
|
5
|
+
"category": "fairness",
|
|
6
|
+
"status": "resolved",
|
|
7
|
+
"verification": "judgment",
|
|
8
|
+
"resolution": null,
|
|
9
|
+
"reason": null,
|
|
10
|
+
"statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate"
|
|
11
|
+
},
|
|
12
|
+
{
|
|
13
|
+
"requirement_id": "R1",
|
|
14
|
+
"category": "privacy",
|
|
15
|
+
"status": "resolved",
|
|
16
|
+
"verification": "test",
|
|
17
|
+
"resolution": null,
|
|
18
|
+
"reason": null,
|
|
19
|
+
"statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log"
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"requirement_id": "R1",
|
|
23
|
+
"category": "transparency",
|
|
24
|
+
"status": "resolved",
|
|
25
|
+
"verification": "judgment",
|
|
26
|
+
"resolution": null,
|
|
27
|
+
"reason": null,
|
|
28
|
+
"statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)"
|
|
29
|
+
}
|
|
30
|
+
],
|
|
31
|
+
"coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } }
|
|
32
|
+
}
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
# Prohibition-Probe — Spec-Completeness Must-NOT Reference
|
|
2
|
+
|
|
3
|
+
Shared reference for the spec/requirements phase. Companion to
|
|
4
|
+
`@~/.claude/gsd-core/references/edge-probe.md`: `edge-probe` reaches the
|
|
5
|
+
**data/behavior-shape axis** (boundaries, adjacency, encoding, ordering) — the things a
|
|
6
|
+
feature must *do*. This reference reaches the orthogonal **must-NOT axis** (product, values,
|
|
7
|
+
safety, ethics) — the things a feature must *never silently become*. The edge-probe caught
|
|
8
|
+
0/8 of these in controlled testing because it is the wrong instrument: a shape taxonomy
|
|
9
|
+
cannot surface "the reminder must not shame the user." Walk each requirement through the
|
|
10
|
+
two-stage recall→precision protocol below and resolve each surfaced prohibition to exactly
|
|
11
|
+
one state.
|
|
12
|
+
|
|
13
|
+
This doc is written in generic `requirements → checks → verifier` terms with no
|
|
14
|
+
tool-specific vocabulary, so it is portable: copy it into any spec/requirements process.
|
|
15
|
+
A short mapping table at the end binds it to common host structures.
|
|
16
|
+
|
|
17
|
+
## Why front-of-pipeline
|
|
18
|
+
|
|
19
|
+
A goal-backward verifier only checks assertions that exist; an assertion only exists for a
|
|
20
|
+
requirement that was written down. The class of constraint this probe targets — the
|
|
21
|
+
*"must-NOT"* the author assumed but never wrote — is invisible to the verifier in exactly
|
|
22
|
+
the same way an omitted edge is, but with a sharper failure mode: a `✅ done` that means
|
|
23
|
+
"the code matches the words in the spec" can still ship a feature that does what the author
|
|
24
|
+
explicitly would *not* want. The manipulative-streak reminder, the loan model that proxies
|
|
25
|
+
on zip code, the audit log that stores raw SSN — each one passes a literal spec while
|
|
26
|
+
violating the intent. The fix is not a better verifier; it is **spec completeness**: surface
|
|
27
|
+
the omitted prohibition into an explicit, checkable acceptance criterion *before* any code
|
|
28
|
+
exists, after which the verifier reliably enforces it.
|
|
29
|
+
|
|
30
|
+
The technique is adversarial elicitation, not deterministic computation. Unlike the edge
|
|
31
|
+
taxonomy (a closed eight categories a classifier can apply), the recall stage is inherently
|
|
32
|
+
model-driven: it asks an open question and reads prose. There is **no compiled
|
|
33
|
+
`prohibition-probe.cjs` engine** — the recall stage is an LLM prose pass, and only the
|
|
34
|
+
schema/projection layer is real code (ADR-550 Decision 7b). Building a deterministic
|
|
35
|
+
recall adapter would be the scope-creep the maintainer flags.
|
|
36
|
+
|
|
37
|
+
## Inputs
|
|
38
|
+
|
|
39
|
+
A list of requirements, each a `{ id, text }` record where `text` is a testable statement.
|
|
40
|
+
There is no shape override and no taxonomy classifier — the recall stage reads the prose
|
|
41
|
+
directly and the precision stage filters its raw output. The probe runs **after** the
|
|
42
|
+
edge-probe in the spec phase, over the same requirement list.
|
|
43
|
+
|
|
44
|
+
## Two-stage protocol (recall → precision)
|
|
45
|
+
|
|
46
|
+
The probe is a two-pass pipeline per requirement. Stage 1 maximizes recall (cast wide);
|
|
47
|
+
Stage 2 restores precision (drop the noise). Running them in this order — wide then narrow —
|
|
48
|
+
is what keeps the surfaced list both complete and short.
|
|
49
|
+
|
|
50
|
+
**Stage 1 — Recall (adversarial probe).** Ask the single adversarial question of each
|
|
51
|
+
requirement:
|
|
52
|
+
|
|
53
|
+
> *What could this feature silently become that the author would NOT want, but the spec
|
|
54
|
+
> does not forbid?*
|
|
55
|
+
|
|
56
|
+
This question is model-robust (17/17 holistic surfacing including smaller models in the N18
|
|
57
|
+
experiment). It deliberately over-produces: ~10 raw candidates per requirement, including
|
|
58
|
+
routine engineering items. That over-production is intentional — recall first.
|
|
59
|
+
|
|
60
|
+
**Stage 2 — Precision (one-pass classifier).** Filter the raw Stage-1 list in a single pass.
|
|
61
|
+
The rule is a drop/keep split:
|
|
62
|
+
|
|
63
|
+
- **DROP routine-engineering items** — anything that is a normal correctness or hygiene
|
|
64
|
+
concern rather than an intent constraint: "must not mutate its input", "must not throw on
|
|
65
|
+
empty list", "must return a primitive not an object", "must not leak a file handle". These
|
|
66
|
+
belong to the edge-probe or to ordinary code review, not here.
|
|
67
|
+
- **KEEP values / safety / ethics items** — anything that, if violated, makes the feature do
|
|
68
|
+
something the author would object to on product, fairness, privacy, transparency, or
|
|
69
|
+
safety grounds: "must not use shaming framing", "must not proxy on protected attributes",
|
|
70
|
+
"must not store raw PII in plaintext".
|
|
71
|
+
|
|
72
|
+
This collapses the raw ~10 to ~2–3 genuine prohibitions (GT 5/5, 0 false positives on the
|
|
73
|
+
N18 eight-spec battery). A requirement that yields zero kept prohibitions emits an empty
|
|
74
|
+
list — that is the correct precision outcome for a pure utility, not a failure.
|
|
75
|
+
|
|
76
|
+
## Canon-referral (do not mint canon items)
|
|
77
|
+
|
|
78
|
+
Some kept candidates are not bespoke at all — they are **canon** security/compliance
|
|
79
|
+
constraints that a dedicated tool already owns. Do NOT mint a prohibition for them. Instead
|
|
80
|
+
emit a one-line breadcrumb and stop:
|
|
81
|
+
|
|
82
|
+
- OWASP / prototype-pollution / path-traversal / injection → breadcrumb to `/gsd:secure-phase`
|
|
83
|
+
and `eslint` (security plugins), not a minted prohibition.
|
|
84
|
+
- GDPR / data-retention / consent → breadcrumb to `/gsd:secure-phase`.
|
|
85
|
+
- Generic fairness/bias canon → breadcrumb to `/gsd:secure-phase`.
|
|
86
|
+
|
|
87
|
+
The breadcrumb reads like: *"prototype-pollution is canon — covered by /gsd:secure-phase +
|
|
88
|
+
eslint; not minted here."* This keeps the surfaced list to the ~2–3 **bespoke** items that
|
|
89
|
+
no other tool would catch — the manipulative-framing prohibition, the product-specific
|
|
90
|
+
fairness constraint — which is the whole value of the probe. Minting canon items both
|
|
91
|
+
duplicates other tooling and drowns the bespoke signal (ADR-550 Decision 6).
|
|
92
|
+
|
|
93
|
+
## Resolution states
|
|
94
|
+
|
|
95
|
+
Each surfaced prohibition carries two orthogonal axes — a resolution **lifecycle** and, when
|
|
96
|
+
resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model; the
|
|
97
|
+
lifecycle is identical to the edge-probe, the verification tiers differ):
|
|
98
|
+
|
|
99
|
+
- **status** — `resolved | dismissed | unresolved` (IDENTICAL to the edge-probe):
|
|
100
|
+
- **resolved** — the prohibition is addressed; *how* it is addressed is the verification tier.
|
|
101
|
+
- **dismissed** — not a genuine prohibition for this feature, accompanied by a required,
|
|
102
|
+
non-empty reason string. "N/A — this utility has no user-facing surface, no values
|
|
103
|
+
constraint applies" is valid; silence is not. The reason string is the audit trail.
|
|
104
|
+
- **unresolved** — carried forward and flagged; the author chose not to resolve it yet.
|
|
105
|
+
- **verification** (only when `status` is `resolved`; `null` otherwise) — `test | judgment`
|
|
106
|
+
(this REPLACES the edge-probe's `explicit | backstop`):
|
|
107
|
+
- **test** — the prohibition can be mechanically checked (a negative test, a lint rule, an
|
|
108
|
+
assertion that the audit log contains no raw SSN). A checkable assertion exists.
|
|
109
|
+
- **judgment** — the prohibition is real but cannot be reduced to a mechanical test (a
|
|
110
|
+
human/LLM judgment that the framing is not manipulative). It records intent and routes
|
|
111
|
+
to a judgment-based review rather than a green/red test.
|
|
112
|
+
|
|
113
|
+
Splitting these axes keeps the lifecycle enum free of a verification fact and lets the
|
|
114
|
+
prohibition adapter declare `test | judgment` without forking the shared lifecycle enum that
|
|
115
|
+
the edge-probe's `explicit | backstop` also uses.
|
|
116
|
+
|
|
117
|
+
## Output schema
|
|
118
|
+
|
|
119
|
+
The probe emits, per kept prohibition, an item of the form:
|
|
120
|
+
|
|
121
|
+
```
|
|
122
|
+
{ requirement_id, category, status, verification, resolution, reason, statement }
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
where `statement` is the must-NOT sentence and `category` is the values/safety/ethics class
|
|
126
|
+
(`values`, `fairness`, `privacy`, `transparency`, `safety`, …), plus a coverage summary:
|
|
127
|
+
|
|
128
|
+
```
|
|
129
|
+
coverage: { applicable, resolved, unresolved, byVerification: { test, judgment } }
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
`applicable` is the number of kept prohibitions, `resolved` = closed (`resolved` +
|
|
133
|
+
`dismissed`) status items, `unresolved` is the remainder, and `byVerification` breaks the
|
|
134
|
+
`resolved`-status items down by tier (`{ test, judgment }`). This JSON is the stable contract
|
|
135
|
+
both the reference implementation and any third-party port emit.
|
|
136
|
+
|
|
137
|
+
## Generic mapping (requirements → checks → verifier)
|
|
138
|
+
|
|
139
|
+
| Host structure | "requirement" | a `resolved`/`test` prohibition becomes | a `resolved`/`judgment` prohibition becomes |
|
|
140
|
+
|----------------|---------------|------------------------------------------|----------------------------------------------|
|
|
141
|
+
| GSD SPEC | a SPEC Requirement | a SPEC acceptance criterion (marked prohibition) that `plan-phase` lifts into `must_haves.prohibitions` | a `must_haves.prohibitions` item routed to judgment review |
|
|
142
|
+
| Gherkin feature | a Scenario | a negative `Then` assertion / tagged negative scenario | a tagged scenario routed to manual review |
|
|
143
|
+
| OpenAPI operation | an operation | a contract test asserting the forbidden behavior never occurs | a documented constraint flagged for review |
|
|
144
|
+
| Docstring contract | a documented behavior | a negative assertion in the contract test | a documented must-NOT for reviewers |
|
|
145
|
+
|
|
146
|
+
The portable invariant: a `resolved`/`test` prohibition produces **a checkable negative the
|
|
147
|
+
verifier iterates over** (GSD: a `must_haves.prohibitions` item with a test); a
|
|
148
|
+
`resolved`/`judgment` prohibition produces a recorded intent routed to judgment review. An
|
|
149
|
+
`unresolved` prohibition is an explicit assumption the downstream planner must surface, not
|
|
150
|
+
silently drop.
|
|
151
|
+
|
|
152
|
+
## Worked example (streak-reminder)
|
|
153
|
+
|
|
154
|
+
A single requirement to send a daily habit reminder. The edge-probe sees a `stateful`
|
|
155
|
+
requirement and asks about idempotency; the prohibition-probe asks the adversarial question
|
|
156
|
+
and surfaces what the reminder must never *become*. Stage 1 over-produces ("must not spam",
|
|
157
|
+
"must not throw on a deleted habit", "must not use shaming framing"); Stage 2 drops the
|
|
158
|
+
routine-engineering items and keeps the one genuine values prohibition:
|
|
159
|
+
|
|
160
|
+
```json prohibition-probe:01-streak-reminder/expected.json
|
|
161
|
+
{
|
|
162
|
+
"items": [
|
|
163
|
+
{
|
|
164
|
+
"requirement_id": "R1",
|
|
165
|
+
"category": "values",
|
|
166
|
+
"status": "resolved",
|
|
167
|
+
"verification": "judgment",
|
|
168
|
+
"resolution": null,
|
|
169
|
+
"reason": null,
|
|
170
|
+
"statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing"
|
|
171
|
+
}
|
|
172
|
+
],
|
|
173
|
+
"coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } }
|
|
174
|
+
}
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
The kept prohibition is `judgment`-tier: "manipulative framing" cannot be reduced to a
|
|
178
|
+
mechanical test, so it records intent and routes to judgment review — but it is now an
|
|
179
|
+
explicit acceptance criterion the spec must clear, not an unwritten assumption.
|
|
180
|
+
|
|
181
|
+
## Worked example (clean-utility)
|
|
182
|
+
|
|
183
|
+
A pure utility requirement — "deduplicate a list of integers" — has no user-facing surface,
|
|
184
|
+
no values/safety/ethics dimension. Stage 1 still over-produces ("must not mutate the input",
|
|
185
|
+
"must not change order"), but every candidate is routine engineering that Stage 2 drops (and
|
|
186
|
+
the edge-probe already owns). The correct precision outcome is an empty prohibition list — a
|
|
187
|
+
zero, not a false positive:
|
|
188
|
+
|
|
189
|
+
```json prohibition-probe:02-clean-utility/expected.json
|
|
190
|
+
{
|
|
191
|
+
"items": [],
|
|
192
|
+
"coverage": { "applicable": 0, "resolved": 0, "unresolved": 0, "byVerification": { "test": 0, "judgment": 0 } }
|
|
193
|
+
}
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
This is the precision discipline that keeps the probe from crying wolf: a utility with no
|
|
197
|
+
intent surface produces zero prohibitions, so a non-empty list always carries signal.
|
|
198
|
+
|
|
199
|
+
## Worked example (multi-prohibition)
|
|
200
|
+
|
|
201
|
+
A loan-decision requirement is the high-stakes case: it surfaces several distinct
|
|
202
|
+
prohibitions across categories. Stage 1 produces a long list including canon items
|
|
203
|
+
(prototype-pollution, generic GDPR retention) that canon-referral breadcrumbs out; Stage 2
|
|
204
|
+
keeps the three bespoke values/safety items — a `fairness` constraint, a `privacy` constraint
|
|
205
|
+
(`test`-tier, mechanically checkable against the audit log), and a `transparency` constraint:
|
|
206
|
+
|
|
207
|
+
```json prohibition-probe:03-multi-prohibition/expected.json
|
|
208
|
+
{
|
|
209
|
+
"items": [
|
|
210
|
+
{
|
|
211
|
+
"requirement_id": "R1",
|
|
212
|
+
"category": "fairness",
|
|
213
|
+
"status": "resolved",
|
|
214
|
+
"verification": "judgment",
|
|
215
|
+
"resolution": null,
|
|
216
|
+
"reason": null,
|
|
217
|
+
"statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate"
|
|
218
|
+
},
|
|
219
|
+
{
|
|
220
|
+
"requirement_id": "R1",
|
|
221
|
+
"category": "privacy",
|
|
222
|
+
"status": "resolved",
|
|
223
|
+
"verification": "test",
|
|
224
|
+
"resolution": null,
|
|
225
|
+
"reason": null,
|
|
226
|
+
"statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log"
|
|
227
|
+
},
|
|
228
|
+
{
|
|
229
|
+
"requirement_id": "R1",
|
|
230
|
+
"category": "transparency",
|
|
231
|
+
"status": "resolved",
|
|
232
|
+
"verification": "judgment",
|
|
233
|
+
"resolution": null,
|
|
234
|
+
"reason": null,
|
|
235
|
+
"statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)"
|
|
236
|
+
}
|
|
237
|
+
],
|
|
238
|
+
"coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } }
|
|
239
|
+
}
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
The `privacy` row is `test`-tier — "no raw SSN in the audit log" is a mechanical assertion —
|
|
243
|
+
while `fairness` and `transparency` are `judgment`-tier. The byVerification rollup
|
|
244
|
+
`{ test: 1, judgment: 2 }` is the count-preserved breakdown of the three `resolved`-status
|
|
245
|
+
items. Each worked-example block above is kept byte-for-byte (parsed-JSON) identical to its
|
|
246
|
+
fixture under `gsd-core/references/prohibition-probe-fixtures/` by
|
|
247
|
+
`tests/prohibition-probe.docs-fixtures.test.cjs`, so the doc and the reference data cannot
|
|
248
|
+
silently drift.
|
|
@@ -80,6 +80,20 @@ No "should feel good", "looks reasonable", or "generally works" — those are no
|
|
|
80
80
|
Acceptance Criteria above; `backstop` rows must be carried into plan-phase `must_haves`.
|
|
81
81
|
`⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.]
|
|
82
82
|
|
|
83
|
+
## Prohibitions (must-NOT)
|
|
84
|
+
|
|
85
|
+
**Coverage:** [resolved]/[applicable] applicable prohibitions resolved · [unresolved] unresolved
|
|
86
|
+
|
|
87
|
+
| Prohibition (must-NOT statement) | Requirement | Status | Verification / Reason |
|
|
88
|
+
|----------------------------------|-------------|--------|------------------------|
|
|
89
|
+
| [MUST NOT … must-NOT statement] | [Rn] | [resolved / dismissed / ⚠ UNRESOLVED] | [verification: test \| judgment, or dismissal reason] |
|
|
90
|
+
|
|
91
|
+
[Generated by the prohibition probe (Step 5.6). `resolved` prohibitions become NEGATIVE
|
|
92
|
+
acceptance criteria; a `resolved`/`test` row is a checkable negative the verifier iterates
|
|
93
|
+
over, a `resolved`/`judgment` row routes to judgment review. Resolved prohibitions are lifted
|
|
94
|
+
into `must_haves.prohibitions` by plan-phase. `dismissed` rows carry a required non-empty
|
|
95
|
+
reason. `⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.]
|
|
96
|
+
|
|
83
97
|
## Ambiguity Report
|
|
84
98
|
|
|
85
99
|
| Dimension | Score | Min | Status | Notes |
|
|
@@ -612,11 +612,7 @@ Extract `branching_strategy`, `phase_branch_template`, `milestone_branch_templat
|
|
|
612
612
|
|
|
613
613
|
Detect base branch:
|
|
614
614
|
```bash
|
|
615
|
-
BASE_BRANCH=$(gsd_run query
|
|
616
|
-
if [ -z "$BASE_BRANCH" ] || [ "$BASE_BRANCH" = "null" ]; then
|
|
617
|
-
BASE_BRANCH=$(git symbolic-ref refs/remotes/origin/HEAD 2>/dev/null | sed 's|^refs/remotes/origin/||')
|
|
618
|
-
BASE_BRANCH="${BASE_BRANCH:-main}"
|
|
619
|
-
fi
|
|
615
|
+
BASE_BRANCH=$(gsd_run query git.base-branch)
|
|
620
616
|
```
|
|
621
617
|
|
|
622
618
|
**If "none":** Skip to git_tag.
|
|
@@ -189,7 +189,7 @@ CURRENT_PLAN_ID="{phase_number}-{plan_padded}"
|
|
|
189
189
|
SUMMARY_PATH="{phase_dir}/{plan_padded}-SUMMARY.md"
|
|
190
190
|
PLAN_COMMITS=$(git log --oneline --grep="${CURRENT_PLAN_ID}" -30)
|
|
191
191
|
```
|
|
192
|
-
If production commits exist and `SUMMARY.md is missing`, stop before spawning a
|
|
192
|
+
If production commits exist and `SUMMARY.md is missing` (no `.planning/async-jobs/*.json` manifest matches it: a match is a legal `external_job_waiting` deferral - reconcile per `docs/reference/planning-artifacts.md`, never re-dispatch), stop before spawning a
|
|
193
193
|
new executor; continuing risks duplicate work and stale `STATE.md`/ROADMAP progress.
|
|
194
194
|
Offer these recovery options:
|
|
195
195
|
- `close out manually` — inspect commits, write SUMMARY.md, then update STATE/ROADMAP.
|
|
@@ -296,8 +296,9 @@ Check `branching_strategy` from init:
|
|
|
296
296
|
Fork the new phase branch off `origin/HEAD` (the project's default branch), not the current HEAD — otherwise consecutive phases compound and stay unpushed (#2916). If `$BRANCH_NAME` already exists locally, reuse it as-is.
|
|
297
297
|
|
|
298
298
|
```bash
|
|
299
|
-
DEFAULT_BRANCH=$(git
|
|
300
|
-
|
|
299
|
+
DEFAULT_BRANCH=$(gsd_run query git.base-branch 2>/dev/null \
|
|
300
|
+
|| git symbolic-ref --quiet --short refs/remotes/origin/HEAD 2>/dev/null | sed 's|^origin/||' \
|
|
301
|
+
|| echo main)
|
|
301
302
|
|
|
302
303
|
if git show-ref --verify --quiet "refs/heads/$BRANCH_NAME"; then
|
|
303
304
|
git switch "$BRANCH_NAME" || { echo "ERROR: Could not switch to existing branch '$BRANCH_NAME'." >&2; exit 1; }
|
|
@@ -13,10 +13,23 @@ Read config.json for planning behavior settings.
|
|
|
13
13
|
For each executed plan, the only complete close-out order is:
|
|
14
14
|
`production-code commit(s) -> SUMMARY commit -> STATE/ROADMAP update`.
|
|
15
15
|
|
|
16
|
-
|
|
17
|
-
actively working. Once production commits for a plan
|
|
18
|
-
committed SUMMARY.md is an illegal partial-plan state.
|
|
19
|
-
resume must detect that condition before dispatching
|
|
16
|
+
For a synchronous executor, the only legal half-state is mid-production-commits
|
|
17
|
+
while the executor is still actively working. Once production commits for a plan
|
|
18
|
+
exist, returning without a committed SUMMARY.md is an illegal partial-plan state.
|
|
19
|
+
The next execute-phase resume must detect that condition before dispatching
|
|
20
|
+
another executor.
|
|
21
|
+
|
|
22
|
+
**Async exception — `external_job_waiting`.** When an executor dispatches an
|
|
23
|
+
async external job (long-running compute) it commits an async-job manifest at
|
|
24
|
+
`.planning/async-jobs/<job>.json` and returns *without* SUMMARY.md. With a
|
|
25
|
+
manifest recording a non-terminal job for this plan, the SUMMARY-absent state is
|
|
26
|
+
a **legal deferred state** (`external_job_waiting`), not an illegal partial.
|
|
27
|
+
SUMMARY.md is deferred until the external job reaches a terminal state and its
|
|
28
|
+
output is verified. Resume reconciles against the manifest and must NOT
|
|
29
|
+
re-dispatch a fresh executor for a plan with a non-terminal manifest (that would
|
|
30
|
+
duplicate the external job). The manifest schema is the stability contract in
|
|
31
|
+
`docs/reference/planning-artifacts.md`; the scheduler adapter that *writes* it is
|
|
32
|
+
a capability (#1164), not core.
|
|
20
33
|
</atomic_close_out_invariant>
|
|
21
34
|
|
|
22
35
|
<available_agent_types>
|
|
@@ -47,7 +60,9 @@ If `.planning/` missing: error.
|
|
|
47
60
|
(ls .planning/phases/XX-name/*-SUMMARY.md 2>/dev/null || true) | sort
|
|
48
61
|
```
|
|
49
62
|
|
|
50
|
-
Find first PLAN without matching SUMMARY. Decimal phases supported (`01.1-hotfix/`)
|
|
63
|
+
Find first PLAN without matching SUMMARY. Decimal phases supported (`01.1-hotfix/`).
|
|
64
|
+
|
|
65
|
+
**Exclude `external_job_waiting` plans from selection.** When choosing the first PLAN that lacks a matching SUMMARY, skip any plan whose `plan_id` matches an async-job manifest in `.planning/async-jobs/` (any status) — that plan is `external_job_waiting` or awaiting reconciliation, never work to (re-)dispatch (re-dispatching would duplicate the external job). Reconcile via the manifest / safe_resume_gate instead.
|
|
51
66
|
|
|
52
67
|
```bash
|
|
53
68
|
PHASE=$(echo "$PLAN_PATH" | grep -oE '[0-9]+(\.[0-9]+)?-[0-9]+')
|
|
@@ -504,7 +519,7 @@ If `USER_SETUP_CREATED=true`: display `⚠️ USER SETUP REQUIRED` with path + e
|
|
|
504
519
|
|
|
505
520
|
| Condition | Route | Action |
|
|
506
521
|
|-----------|-------|--------|
|
|
507
|
-
| summaries < plans | **A: More plans** | Find next PLAN without SUMMARY. Yolo: auto-continue. Interactive: show next plan, suggest `/gsd:execute-phase {phase}` + `/gsd:verify-work`. STOP here. |
|
|
522
|
+
| summaries < plans | **A: More plans** | Find next PLAN without SUMMARY — skip any plan whose `plan_id` matches a non-terminal async-job manifest (`external_job_waiting`; see `identify_plan`). Yolo: auto-continue. Interactive: show next plan, suggest `/gsd:execute-phase {phase}` + `/gsd:verify-work`. STOP here. |
|
|
508
523
|
| summaries = plans, current < highest phase | **B: Phase done** | Show completion, suggest `/gsd:plan-phase {Z+1}` + `/gsd:verify-work {Z}` + `/gsd:discuss-phase {Z+1}` |
|
|
509
524
|
| summaries = plans, current = highest phase | **C: Milestone done** | Show banner, suggest `/gsd:complete-milestone` + `/gsd:verify-work` + `/gsd-add-phase` |
|
|
510
525
|
|
|
@@ -256,11 +256,15 @@ Check project status and intelligently route to next action.
|
|
|
256
256
|
Modes:
|
|
257
257
|
- **default** — progress report + intelligent routing
|
|
258
258
|
- **`--next`** — auto-advance to the next logical step (use `--next --force` to bypass safety gates)
|
|
259
|
+
- **`--next --auto`** — like `--next`, but chains steps automatically until milestone completion or a blocking decision
|
|
260
|
+
- **`--next --converge`** — when the next action is planning, route it through `/gsd:plan-review-convergence` instead of `/gsd:plan-phase`; requires `workflow.plan_review_convergence=true`. `--cross-ai` is an alias. Reviewer flags (`--codex`, `--gemini`, `--claude`, `--opencode`, `--ollama`, `--lm-studio`, `--llama-cpp`, `--all`) and `--max-cycles N` forward to the convergence loop.
|
|
259
261
|
- **`--forensic`** — append a 6-check integrity audit after the progress report
|
|
260
262
|
- **`--do "<text>"`** — smart router: dispatch freeform intent to the matching `/gsd-*` command (see *Smart Router* above)
|
|
261
263
|
|
|
262
264
|
Usage: `/gsd:progress`
|
|
263
265
|
Usage: `/gsd:progress --next`
|
|
266
|
+
Usage: `/gsd:progress --next --auto`
|
|
267
|
+
Usage: `/gsd:progress --next --auto --converge`
|
|
264
268
|
Usage: `/gsd:progress --forensic`
|
|
265
269
|
|
|
266
270
|
### Session Management
|
|
@@ -230,7 +230,7 @@ If the current phase directory exists but has neither CONTEXT.md nor RESEARCH.md
|
|
|
230
230
|
|
|
231
231
|
**Route 3: Phase has context but no plans → plan**
|
|
232
232
|
If the current phase has CONTEXT.md (or RESEARCH.md) but no PLAN.md files:
|
|
233
|
-
→ Next action: `/gsd:plan-phase <current-phase>`
|
|
233
|
+
→ Next action: `/gsd:plan-phase <current-phase>` (or `/gsd:plan-review-convergence <current-phase>` when `PLAN_STRATEGY=converge`)
|
|
234
234
|
|
|
235
235
|
**Route 4: Phase has plans but incomplete summaries → execute**
|
|
236
236
|
If plans exist but not all have matching summaries:
|
|
@@ -254,6 +254,47 @@ If STATE.md shows paused_at:
|
|
|
254
254
|
</step>
|
|
255
255
|
|
|
256
256
|
<step name="show_and_execute">
|
|
257
|
+
Parse the arguments passed to this workflow to detect the plan strategy and build convergence pass-through args:
|
|
258
|
+
|
|
259
|
+
```bash
|
|
260
|
+
PLAN_STRATEGY="local"
|
|
261
|
+
if echo "$ARGUMENTS" | grep -qE '(^|[[:space:]])\-\-(converge|cross-ai)([[:space:]]|$)'; then
|
|
262
|
+
PLAN_STRATEGY="converge"
|
|
263
|
+
fi
|
|
264
|
+
|
|
265
|
+
CONVERGENCE_ARGS=""
|
|
266
|
+
for REVIEW_FLAG in --codex --gemini --claude --opencode --ollama --lm-studio --llama-cpp --all --text; do
|
|
267
|
+
if echo "$ARGUMENTS" | grep -qE "(^|[[:space:]])${REVIEW_FLAG}([[:space:]]|$)"; then
|
|
268
|
+
CONVERGENCE_ARGS="${CONVERGENCE_ARGS} ${REVIEW_FLAG}"
|
|
269
|
+
fi
|
|
270
|
+
done
|
|
271
|
+
|
|
272
|
+
MAX_CYCLES_ARG=""
|
|
273
|
+
if echo "$ARGUMENTS" | grep -qE '\-\-max-cycles\s+[0-9]+'; then
|
|
274
|
+
MAX_CYCLES_ARG=$(echo "$ARGUMENTS" | grep -oE '\-\-max-cycles\s+[0-9]+' | awk '{print $2}')
|
|
275
|
+
CONVERGENCE_ARGS="${CONVERGENCE_ARGS} --max-cycles ${MAX_CYCLES_ARG}"
|
|
276
|
+
fi
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
If `PLAN_STRATEGY` is `converge`, fail fast unless the convergence feature gate is enabled:
|
|
280
|
+
|
|
281
|
+
```bash
|
|
282
|
+
if [ "$PLAN_STRATEGY" = "converge" ]; then
|
|
283
|
+
CONVERGENCE_ENABLED=$(gsd_run query config-get workflow.plan_review_convergence 2>/dev/null || echo "false")
|
|
284
|
+
if [ "$CONVERGENCE_ENABLED" != "true" ]; then
|
|
285
|
+
printf '%s\n' \
|
|
286
|
+
'/gsd:progress --next --converge is disabled (workflow.plan_review_convergence=false).' \
|
|
287
|
+
'' \
|
|
288
|
+
'Enable plan convergence with:' \
|
|
289
|
+
'' \
|
|
290
|
+
' gsd config-set workflow.plan_review_convergence true' \
|
|
291
|
+
'' \
|
|
292
|
+
'Then re-run with --converge.'
|
|
293
|
+
exit 1
|
|
294
|
+
fi
|
|
295
|
+
fi
|
|
296
|
+
```
|
|
297
|
+
|
|
257
298
|
Display the determination:
|
|
258
299
|
|
|
259
300
|
```
|
|
@@ -269,7 +310,9 @@ Display the determination:
|
|
|
269
310
|
Then immediately invoke the determined command via SlashCommand.
|
|
270
311
|
Do not ask for confirmation — the whole point of `/gsd:progress --next` is zero-friction advancement.
|
|
271
312
|
|
|
272
|
-
**
|
|
313
|
+
**Route 3 convergence override:** When the routing decision is Route 3 (plan) and `PLAN_STRATEGY=converge`, invoke `/gsd:plan-review-convergence <current-phase> ${CONVERGENCE_ARGS}` instead of `/gsd:plan-phase <current-phase>`.
|
|
314
|
+
|
|
315
|
+
**If `--auto` was passed:** after the determined command completes, automatically re-invoke `/gsd:progress --next --auto` (forwarding `--converge`/`--cross-ai` and any reviewer flags if they were originally passed) to continue chaining to the next step. Repeat until one of:
|
|
273
316
|
- A milestone completes (`/gsd:complete-milestone` is reached)
|
|
274
317
|
- A blocking decision is required (safety gate triggers, prior-phase completeness prompt, user input needed)
|
|
275
318
|
- An error or paused state is detected
|
|
@@ -296,4 +339,9 @@ Resume with: `/gsd:progress --next --auto` once resolved.
|
|
|
296
339
|
- [ ] Next action correctly determined from routing rules
|
|
297
340
|
- [ ] Command invoked immediately without user confirmation
|
|
298
341
|
- [ ] Clear status shown before invoking
|
|
342
|
+
- [ ] `--converge` routes Route 3 planning through `gsd-plan-review-convergence`
|
|
343
|
+
- [ ] `--cross-ai` is accepted as an alias for `--converge`
|
|
344
|
+
- [ ] `--converge` fails fast with enable instructions when `workflow.plan_review_convergence=false`
|
|
345
|
+
- [ ] `--converge` forwards reviewer selector flags and `--max-cycles N`
|
|
346
|
+
- [ ] Default planning remains `gsd-plan-phase` when convergence is not requested
|
|
299
347
|
</success_criteria>
|
|
@@ -48,7 +48,8 @@ If phase is detected, proceed with phase handoff path. Otherwise use the first m
|
|
|
48
48
|
6. **Human actions pending**: Things that need manual intervention (MCP setup, API keys, approvals, manual testing)
|
|
49
49
|
7. **Background processes**: Any running servers/watchers that were part of the workflow
|
|
50
50
|
8. **Files modified**: What's changed but not committed
|
|
51
|
-
9. **
|
|
51
|
+
9. **Outstanding async external jobs**: any `.planning/async-jobs/*.json` manifests for non-terminal jobs — record job id, backend, status, expected artifacts, verification + resume commands, and any watcher/daemon state. Do NOT cancel the external job; it keeps running across the pause.
|
|
52
|
+
10. **Blocking constraints**: Anti-patterns or methodological failures encountered during this session that a resuming agent MUST be aware of before proceeding. Only include items discovered through actual failure — not warnings or predictions. Assign each constraint a `severity`:
|
|
52
53
|
- `blocking` — The resuming agent MUST demonstrate understanding before proceeding. The discuss-phase and execute-phase workflows will enforce a mandatory understanding check.
|
|
53
54
|
- `advisory` — Important context but does not gate resumption.
|
|
54
55
|
|
|
@@ -93,6 +94,9 @@ timestamp=$(gsd_run query current-timestamp full --raw)
|
|
|
93
94
|
"blockers": [
|
|
94
95
|
{"description": "{blocker}", "type": "technical|human_action|external", "workaround": "{if any}"}
|
|
95
96
|
],
|
|
97
|
+
"async_jobs": [
|
|
98
|
+
{"manifest": ".planning/async-jobs/{job}.json", "job_id": "{id}", "backend": "{backend}", "status": "running", "submit_command": "{cmd}", "submitted_at": "{iso8601}", "expected_artifacts": ["..."], "verification_command": "{cmd}", "resume_command": "{cmd}"}
|
|
99
|
+
],
|
|
96
100
|
"human_actions_pending": [
|
|
97
101
|
{"action": "{what needs to be done}", "context": "{why}", "blocking": true}
|
|
98
102
|
],
|
|
@@ -104,6 +108,8 @@ timestamp=$(gsd_run query current-timestamp full --raw)
|
|
|
104
108
|
"context_notes": "{mental state, approach, what you were thinking}"
|
|
105
109
|
}
|
|
106
110
|
```
|
|
111
|
+
|
|
112
|
+
Any recorded `async_jobs` entries are the primary resume context on the next session — check them first before treating a PLAN-without-SUMMARY as incomplete work.
|
|
107
113
|
</step>
|
|
108
114
|
|
|
109
115
|
<step name="write">
|
|
@@ -892,6 +892,7 @@ Output consumed by /gsd:execute-phase. Plans need:
|
|
|
892
892
|
- Verification criteria
|
|
893
893
|
- must_haves for goal-backward verification
|
|
894
894
|
- If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths`, and every `backstop` edge into `must_haves.truths` as a non-inferable check (note it needs a held-out/property-based test). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them.
|
|
895
|
+
- If the SPEC has a `## Prohibitions` section, lift every resolved prohibition into the `must_haves.prohibitions:` sibling block (NOT `truths` — ADR-550 D3) carrying `statement` + `status` + `verification`; unresolved prohibitions are explicit assumptions — surface them in the plan, do not silently drop them. A prohibition is a must-NOT (negative) check that belongs in its own `must_haves.prohibitions` block. Never place a must-NOT under `must_haves.truths` — that block keeps positive-observable semantics only.
|
|
895
896
|
- **"Artifacts this phase produces" section (MANDATORY)** — list every symbol this phase creates: decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths. The plan-review-convergence source-grounding pass reads this section to exclude newly-created symbols from drift verification; omitting it causes new symbols to be flagged for acknowledgement.
|
|
896
897
|
</downstream_consumer>
|
|
897
898
|
|
|
@@ -938,6 +939,7 @@ Every task MUST include these fields — they are NOT optional:
|
|
|
938
939
|
- [ ] must_haves derived from phase goal
|
|
939
940
|
- [ ] Every PLAN.md includes an "Artifacts this phase produces" section listing symbols created by this phase (decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths)
|
|
940
941
|
- [ ] Every SPEC ## Edge Coverage covered/backstop edge is represented in a plan's must_haves (no silent drops)
|
|
942
|
+
- [ ] Every SPEC ## Prohibitions resolved item is represented in a plan's must_haves.prohibitions (no silent drops)
|
|
941
943
|
</quality_gate>
|
|
942
944
|
```
|
|
943
945
|
|
|
@@ -185,13 +185,23 @@ Run this pass unless `plan_review.source_grounding` is `false`. It verifies ever
|
|
|
185
185
|
|
|
186
186
|
1. **Enumerate cited symbols.** List every referenced symbol by kind, quoting the plan line for each (coverage must be auditable): decorators (`@name`), classes/methods (`Class.method`), functions (`module.function`), CLI flags (`--name`), file paths, dataclass/struct fields.
|
|
187
187
|
2. **Exclude new artifacts.** Do NOT verify symbols the plan declares under its "Artifacts this phase produces" section — those are created by this phase, not references to existing code.
|
|
188
|
-
3. **Resolve each remaining symbol** using the adapter
|
|
188
|
+
3. **Resolve each remaining symbol** using the effective authority adapter (resolved deterministically — see step 4a):
|
|
189
189
|
- `grep` — ripgrep / Read the source; confirm the name appears as a real declaration.
|
|
190
190
|
- `intel` — consult `.planning/intel/API-SURFACE.md` / `api-map.json` (only when `intel.enabled`).
|
|
191
191
|
Record one verdict per symbol: **VERIFIED** (quote `file:line`), **MISSING** (adapter can check this language/kind and the symbol is absent), **AMBIGUOUS** (multiple candidates), or **UNCHECKABLE** (adapter cannot analyze this language/kind — e.g. non-JS under `intel`, or any signature under `grep`). Never treat UNCHECKABLE as verified or missing.
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
-
|
|
192
|
+
4a. **Resolve effective authority** (deterministic — replaces manual `intel.enabled` reasoning):
|
|
193
|
+
```bash
|
|
194
|
+
EFFECTIVE_AUTHORITY=$(gsd_run drift-guard authority --raw)
|
|
195
|
+
```
|
|
196
|
+
4. **Severity & gating** — classify each symbol's verdict using the seam (do not apply the table manually):
|
|
197
|
+
```bash
|
|
198
|
+
# For each symbol, e.g.:
|
|
199
|
+
RESULT=$(gsd_run drift-guard severity --status <verdict> --authority "$EFFECTIVE_AUTHORITY")
|
|
200
|
+
# $RESULT is JSON: {"severity":"…","hardBlock":true|false}
|
|
201
|
+
```
|
|
202
|
+
- `hardBlock: true` (HIGH at authority `lsp`/`scip`) — stops the review cycle immediately; do not proceed until the plan author resolves the missing symbol.
|
|
203
|
+
- `hardBlock: false`, severity `needs-acknowledgement` — plan proceeds only if the author confirms the symbol is genuinely new or dynamically resolved, and that acknowledgement is recorded.
|
|
204
|
+
- `AMBIGUOUS` → MEDIUM. `UNCHECKABLE` → INFO.
|
|
195
205
|
- Signature mismatches cannot be asserted under `grep`/`intel`; report the signature as UNCHECKABLE.
|
|
196
206
|
5. **Coverage block.** Append a "Verification coverage" section to `REVIEWS.md` listing every UNCHECKABLE/skipped symbol and why — a clean review must never silently mean "nothing was checked."
|
|
197
207
|
|