@opengsd/gsd-core 1.6.0 → 1.7.0-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/agents/gsd-verifier.md +1 -0
- package/bin/gsd-mcp-server.js +31 -0
- package/bin/install.js +293 -1145
- package/commands/gsd/review.md +6 -0
- package/gemini-extension.json +1 -1
- package/gsd-core/bin/gsd-tools.cjs +116 -1
- package/gsd-core/bin/lib/adapter-declarative.cjs +35 -0
- package/gsd-core/bin/lib/adapter-imperative.cjs +52 -0
- package/gsd-core/bin/lib/assumption-delta.cjs +231 -0
- package/gsd-core/bin/lib/capability-lifecycle.cjs +7 -7
- package/gsd-core/bin/lib/capability-loader.cjs +18 -0
- package/gsd-core/bin/lib/capability-lock.cjs +2 -2
- package/gsd-core/bin/lib/capability-registry.cjs +889 -82
- package/gsd-core/bin/lib/capability-source.cjs +4 -4
- package/gsd-core/bin/lib/capability-validator.cjs +198 -0
- package/gsd-core/bin/lib/cli-skew-check.cjs +44 -0
- package/gsd-core/bin/lib/command-aliases.cjs +8 -0
- package/gsd-core/bin/lib/config.cjs +27 -0
- package/gsd-core/bin/lib/embedding-adapter.cjs +27 -0
- package/gsd-core/bin/lib/external-descriptor-trust.cjs +70 -0
- package/gsd-core/bin/lib/hook-bus.cjs +81 -0
- package/gsd-core/bin/lib/host-integration.cjs +408 -0
- package/gsd-core/bin/lib/init.cjs +1 -1
- package/gsd-core/bin/lib/install-engine.cjs +755 -0
- package/gsd-core/bin/lib/install-profiles.cjs +35 -4
- package/gsd-core/bin/lib/installer-migrations.cjs +1 -1
- package/gsd-core/bin/lib/mcp-server.cjs +194 -0
- package/gsd-core/bin/lib/milestone.cjs +27 -30
- package/gsd-core/bin/lib/model-adapter.cjs +50 -0
- package/gsd-core/bin/lib/phase.cjs +41 -72
- package/gsd-core/bin/lib/planning-workspace.cjs +1 -1
- package/gsd-core/bin/lib/probe-core.cjs +91 -1
- package/gsd-core/bin/lib/review-reviewer-selection.cjs +129 -13
- package/gsd-core/bin/lib/roadmap-upgrade.cjs +3 -2
- package/gsd-core/bin/lib/roadmap.cjs +17 -3
- package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +65 -9
- package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +54 -4
- package/gsd-core/bin/lib/runtime-artifact-layout.cjs +5 -2
- package/gsd-core/bin/lib/runtime-hooks-surface.cjs +1 -1
- package/gsd-core/bin/lib/runtime-name-policy.cjs +160 -30
- package/gsd-core/bin/lib/shell-command-projection.cjs +37 -1
- package/gsd-core/bin/lib/stale-bake-guard.cjs +254 -0
- package/gsd-core/bin/lib/state-command-router.cjs +4 -0
- package/gsd-core/bin/lib/state-io.cjs +55 -0
- package/gsd-core/bin/lib/state-transition.cjs +1588 -0
- package/gsd-core/bin/lib/state.cjs +306 -681
- package/gsd-core/bin/lib/surface.cjs +4 -1
- package/gsd-core/bin/lib/workstream.cjs +4 -4
- package/gsd-core/bin/shared/config-schema.manifest.json +9 -0
- package/gsd-core/references/honest-verifier.md +105 -0
- package/gsd-core/references/reviewer-instances.md +99 -0
- package/gsd-core/workflows/autonomous.md +9 -9
- package/gsd-core/workflows/manager.md +15 -15
- package/gsd-core/workflows/plan-phase.md +1 -1
- package/gsd-core/workflows/review.md +26 -0
- package/gsd-core/workflows/thread.md +4 -4
- package/gsd-core/workflows/verify-phase.md +11 -4
- package/hooks/dist/gsd-graphify-update.sh +7 -1
- package/hooks/gsd-graphify-update.sh +7 -1
- package/package.json +4 -4
- package/scripts/ci-test-scope.cjs +38 -9
- package/scripts/lint-allow-test-rule-refs.allowlist.json +0 -1
- package/scripts/lint-regression-test-names.allowlist.json +3 -0
- package/scripts/lint-test-file-count.allowlist.json +19 -5
- package/scripts/mutation-matrix.cjs +45 -3
- package/scripts/prompt-injection-scan.sh +8 -0
- package/scripts/lint-windows-test-portability.cjs +0 -178
|
@@ -43,6 +43,9 @@ const runtimeArtifactLayout = require("./runtime-artifact-layout.cjs");
|
|
|
43
43
|
const { findInstallSourceRoot } = runtimeArtifactLayout;
|
|
44
44
|
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
|
45
45
|
const runtimeArtifactConversion = require("./runtime-artifact-conversion.cjs");
|
|
46
|
+
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
|
47
|
+
const runtimeArtifactInstallPlan = require("./runtime-artifact-install-plan.cjs");
|
|
48
|
+
const { assertDestWithinConfigHome } = runtimeArtifactInstallPlan;
|
|
46
49
|
const SURFACE_FILE_NAME = '.gsd-surface.json';
|
|
47
50
|
/**
|
|
48
51
|
* Read the surface state from a runtime config directory.
|
|
@@ -302,7 +305,7 @@ function applySurface(runtimeConfigDir, layout, manifest, clusterMap, registry)
|
|
|
302
305
|
tempDirsToClean.push(rewritten);
|
|
303
306
|
}
|
|
304
307
|
}
|
|
305
|
-
const dest =
|
|
308
|
+
const dest = assertDestWithinConfigHome(layout.configDir, kind.destSubpath);
|
|
306
309
|
_syncGsdDir(staged, dest, kind, skillManifest);
|
|
307
310
|
}
|
|
308
311
|
}
|
|
@@ -67,7 +67,7 @@ function migrateToWorkstreams(cwd, workstreamName) {
|
|
|
67
67
|
const src = node_path_1.default.join(baseDir, item.name);
|
|
68
68
|
if (node_fs_1.default.existsSync(src)) {
|
|
69
69
|
const dest = node_path_1.default.join(wsDir, item.name);
|
|
70
|
-
|
|
70
|
+
(0, shell_command_projection_cjs_1.retryRenameSync)(src, dest);
|
|
71
71
|
filesMoved.push(item.name);
|
|
72
72
|
}
|
|
73
73
|
}
|
|
@@ -75,7 +75,7 @@ function migrateToWorkstreams(cwd, workstreamName) {
|
|
|
75
75
|
catch (err) {
|
|
76
76
|
for (const name of filesMoved) {
|
|
77
77
|
try {
|
|
78
|
-
|
|
78
|
+
(0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(wsDir, name), node_path_1.default.join(baseDir, name));
|
|
79
79
|
}
|
|
80
80
|
catch { /* ignore */ }
|
|
81
81
|
}
|
|
@@ -275,14 +275,14 @@ function cmdWorkstreamComplete(cwd, name, options, raw) {
|
|
|
275
275
|
try {
|
|
276
276
|
const entries = node_fs_1.default.readdirSync(wsDir, { withFileTypes: true });
|
|
277
277
|
for (const entry of entries) {
|
|
278
|
-
|
|
278
|
+
(0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(wsDir, entry.name), node_path_1.default.join(archivePath, entry.name));
|
|
279
279
|
filesMoved.push(entry.name);
|
|
280
280
|
}
|
|
281
281
|
}
|
|
282
282
|
catch (err) {
|
|
283
283
|
for (const fname of filesMoved) {
|
|
284
284
|
try {
|
|
285
|
-
|
|
285
|
+
(0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(archivePath, fname), node_path_1.default.join(wsDir, fname));
|
|
286
286
|
}
|
|
287
287
|
catch { /* ignore */ }
|
|
288
288
|
}
|
|
@@ -10,6 +10,10 @@
|
|
|
10
10
|
"brave_search",
|
|
11
11
|
"firecrawl",
|
|
12
12
|
"exa_search",
|
|
13
|
+
"tavily_search",
|
|
14
|
+
"ref_search",
|
|
15
|
+
"perplexity",
|
|
16
|
+
"jina",
|
|
13
17
|
"workflow.plan_check",
|
|
14
18
|
"workflow.verifier",
|
|
15
19
|
"workflow.auto_advance",
|
|
@@ -172,6 +176,11 @@
|
|
|
172
176
|
"source": "^review\\.max_prompt_tokens_per_reviewer\\.[a-zA-Z0-9_-]+$",
|
|
173
177
|
"description": "review.max_prompt_tokens_per_reviewer.<reviewer-slug>"
|
|
174
178
|
},
|
|
179
|
+
{
|
|
180
|
+
"topLevel": "review",
|
|
181
|
+
"source": "^review\\.reviewer_instances\\.[a-zA-Z0-9_-]+\\.(cli|model|agent)$",
|
|
182
|
+
"description": "review.reviewer_instances.<instance-name>.<cli|model|agent> (#1517)"
|
|
183
|
+
},
|
|
175
184
|
{
|
|
176
185
|
"topLevel": "model_policy",
|
|
177
186
|
"source": "^model_policy\\.runtime_tiers\\.[a-zA-Z0-9_-]+\\.(opus|sonnet|haiku)$",
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# Honest Verifier — Abstention on Non-Inferable Checks
|
|
2
|
+
|
|
3
|
+
Shared reference for the **verify** phase. The verify-time companion to the spec-time
|
|
4
|
+
`@~/.claude/gsd-core/references/edge-probe.md` (which *classifies* non-inferable checks) and
|
|
5
|
+
`@~/.claude/gsd-core/references/prohibition-probe.md` (whose judgment-tier disposition this mirrors).
|
|
6
|
+
This doc is written in generic `spec → predicate → verifier` terms with no tool-specific vocabulary,
|
|
7
|
+
so it is portable: copy it into any verification process.
|
|
8
|
+
|
|
9
|
+
## The problem it solves
|
|
10
|
+
|
|
11
|
+
A verifier is trustworthy on **inferable** checks — defects determined by the stated spec. On a
|
|
12
|
+
**non-inferable** check the correct answer is *not derivable from the spec alone* (e.g. "does `[1,2]`
|
|
13
|
+
touching `[2,3]` merge?", "is a 'character' a grapheme or a code unit?"). On these the verifier *does
|
|
14
|
+
not know that it does not know*: measured behavior is a **confident PASS on the blind-spot check ~100%
|
|
15
|
+
of the time** (mean confidence ~0.93), because a model cannot self-detect a gap it does not perceive.
|
|
16
|
+
|
|
17
|
+
The edge-probe already detects these at spec time and tags them `verification: backstop` (ADR-550
|
|
18
|
+
D7a). The honest verifier consumes that tag so the verifier **abstains** instead of confidently
|
|
19
|
+
false-passing — converting a silent false-pass (the worst failure: you don't know to look) into an
|
|
20
|
+
explicit, actionable "write a held-out test." Measured: the confident-false-pass rate on the blind
|
|
21
|
+
spot drops **100% → 17%** (N17).
|
|
22
|
+
|
|
23
|
+
## The two properties that define the design
|
|
24
|
+
|
|
25
|
+
1. **Exogenous, not endogenous.** The trigger is the *external tag* (`backstop`), never the verifier's
|
|
26
|
+
self-judgment. Asking the verifier to "abstain if unsure" barely moves the number (100% → 67%) and
|
|
27
|
+
only on ambiguity it already notices; on a true blind spot it stays confidently wrong. A confidence
|
|
28
|
+
gate cannot reach a blind spot the model does not feel — so there is **no "are you sure?" prompt**;
|
|
29
|
+
routing is on the pre-existing tag only.
|
|
30
|
+
2. **Routing, not diagnosis.** The verifier need not name the omitted rule (if it could, it wouldn't
|
|
31
|
+
be a blind spot). In testing, verifiers abstained correctly while citing the *wrong* edge. The
|
|
32
|
+
honest verdict requires only "I was told this is under-specified and I cannot rule it out." The
|
|
33
|
+
omitted rule is carried by a human-authored held-out test, not by the verifier.
|
|
34
|
+
|
|
35
|
+
## The disposition (the protocol)
|
|
36
|
+
|
|
37
|
+
For each `must_haves.truths` item:
|
|
38
|
+
|
|
39
|
+
| Item | Confirmable with explicit evidence? | Disposition |
|
|
40
|
+
|---|---|---|
|
|
41
|
+
| Inferable (plain string, or `verification: explicit`) | n/a — graded normally | ✓ VERIFIED / ✗ FAILED as usual; **never abstained** (over-abstention guard) |
|
|
42
|
+
| Non-inferable (`verification: backstop`) | **yes** (a wired held-out/property-based test that passes, or a directly-observed behavior) | ✓ VERIFIED |
|
|
43
|
+
| Non-inferable (`verification: backstop`) | **no** | **abstain** → ⚠️ `insufficient_spec`, flagged, → `human_needed` — **never `passed`** |
|
|
44
|
+
|
|
45
|
+
- **Explicit evidence** = a wired held-out/property-based test that passes, or a behavior the verifier
|
|
46
|
+
directly observed. Symbol presence + wiring is **not** explicit evidence for a non-inferable truth.
|
|
47
|
+
- **Never silent, never a hard halt.** *Interactive:* the abstained item routes to the end-of-phase
|
|
48
|
+
human checkpoint. *Autonomous (AFK):* it produces a prominent `unverified — held-out test
|
|
49
|
+
recommended` flag and the completion line reads "complete with N unverified non-inferable checks";
|
|
50
|
+
the run neither silently passes the blind spot nor hard-halts.
|
|
51
|
+
- **Distinguishable reason.** The abstain disposition carries `reason: insufficient_spec` so the
|
|
52
|
+
`human_needed` outcome is never conflated with an ordinary manual-UAT `human_needed`.
|
|
53
|
+
|
|
54
|
+
This is the verify-time half of ADR-550 Decision 4 (the never-silent-pass disposition), applied to the
|
|
55
|
+
edge `backstop` truth tier instead of the prohibition judgment tier — the same machinery, opposite
|
|
56
|
+
polarity (must-HAVE under-specified vs must-NOT irreducible).
|
|
57
|
+
|
|
58
|
+
## Deterministic engine surface
|
|
59
|
+
|
|
60
|
+
The CI-testable surface is the **deterministic disposition + projection**, never the LLM's judgment
|
|
61
|
+
(ADR-550 D5 — a test asserting the model's verdict is vacuous and rejected). In `probe-core`:
|
|
62
|
+
|
|
63
|
+
- `truthStatement(t)` / `truthVerification(t)` — normalizers; read a truth's statement and tier from
|
|
64
|
+
either the plain-string or object form (a truth-reader MUST normalize, never assume a string).
|
|
65
|
+
- `projectTruths(items)` — conservative serializer: a `backstop` truth → flat-scalar object
|
|
66
|
+
`{ statement, verification: backstop }`; every inferable truth → a bare string.
|
|
67
|
+
- `dispositionForUnverifiableTruth(truth, { evidence })` → `{ status, flagged, tier, reason }`:
|
|
68
|
+
`backstop` + no evidence → `unverified`/`flagged`/`insufficient_spec`; `backstop` + evidence →
|
|
69
|
+
`green`; non-`backstop` → `green` (over-abstention guard).
|
|
70
|
+
|
|
71
|
+
## Capable-tier requirement (a documented cost)
|
|
72
|
+
|
|
73
|
+
Abstention is **model-tier dependent** and this is a standing cost, not an assumption:
|
|
74
|
+
|
|
75
|
+
- The default `gsd-verifier` tier (`sonnet`, golden/balanced) heeds the exogenous tag reliably
|
|
76
|
+
(2/2 under testing).
|
|
77
|
+
- The **budget tier (`haiku`)** is the least flag-responsive (1/2, inconsistent) and **degrades toward
|
|
78
|
+
current behavior** (confident false-pass). Run honest-verifier on a capable tier; treat the budget
|
|
79
|
+
tier as best-effort. Re-validate when the `gsd-verifier` model tier changes or a new budget model is
|
|
80
|
+
adopted (captured as a test so a tier regression is caught, not discovered in production).
|
|
81
|
+
|
|
82
|
+
## Evidence and scope (stated honestly)
|
|
83
|
+
|
|
84
|
+
- **Evidence strength.** N17 is n=27 verdicts (3 models × 3 conditions × 3 tasks), 1 rep —
|
|
85
|
+
**direction-finding, not powered.** The blind-spot effect is large and monotone
|
|
86
|
+
(100% → 67% → 17%); the two costs are clean single events (a *false* tag made the strongest model
|
|
87
|
+
over-abstain on a real spec-determined bug; the weakest tier was flag-deaf) and they name exactly
|
|
88
|
+
the failure modes the over-abstention guard and the capable-tier requirement defend against.
|
|
89
|
+
- **Tag-precision coupling.** Quality is bounded by the edge-probe's `backstop` recall/precision — a
|
|
90
|
+
false non-inferable flag causes over-abstention. Positive coupling: improving the probe (#1110)
|
|
91
|
+
improves this for free. It adds no independent burden.
|
|
92
|
+
- **Explicit non-goals.** Does NOT identify the omitted rule; does NOT recalibrate decisive verdicts;
|
|
93
|
+
does NOT defend against *malicious compliance* (a self-graded review rationalizing away its own
|
|
94
|
+
findings). It raises the floor on *honest* uncertainty about non-inferable checks — that is the
|
|
95
|
+
whole claim.
|
|
96
|
+
|
|
97
|
+
## Distinct from neighbours
|
|
98
|
+
|
|
99
|
+
- **vs `PRESENT_BEHAVIOR_UNVERIFIED` (#966 axis):** that is the *inferable-but-unobserved* case — the
|
|
100
|
+
truth **can** be verified from the spec but was shortcut-passed on symbol presence; the fix is to
|
|
101
|
+
demand behavioral evidence. Honest-verifier is the *non-inferable* case — the truth **cannot** be
|
|
102
|
+
verified from the spec at all; the fix is to abstain and route to a held-out test. Orthogonal axes
|
|
103
|
+
(insufficient *evidence* vs insufficient *spec*); both feed the same `human_needed` sink.
|
|
104
|
+
- **vs prohibition judgment-tier (#644):** that disposes **must-NOT** constraints; honest-verifier
|
|
105
|
+
disposes **non-inferable positive truths**. Opposite polarity, same never-silent disposition.
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# Reviewer Instances (#1517)
|
|
2
|
+
|
|
3
|
+
Custom reviewer instances for `/gsd:review`: run one model-capable adapter (e.g. OpenCode)
|
|
4
|
+
as several independent reviewer identities in a single review pass. Loaded lazily by
|
|
5
|
+
`gsd-core/workflows/review.md` when `review.reviewer_instances` is configured. See
|
|
6
|
+
[ADR-1517](../docs/adr/1517-reviewer-instances-config-surface.md) for the contract.
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Config shape
|
|
11
|
+
|
|
12
|
+
A `review.reviewer_instances` object under the `review` namespace. Each entry maps an
|
|
13
|
+
instance name to `{ cli, model?, agent? }`:
|
|
14
|
+
|
|
15
|
+
```json
|
|
16
|
+
{
|
|
17
|
+
"review": {
|
|
18
|
+
"reviewer_instances": {
|
|
19
|
+
"opencode-deepseek": { "cli": "opencode", "model": "deepseek/deepseek-v4-pro", "agent": "review" },
|
|
20
|
+
"opencode-mimo": { "cli": "opencode", "model": "xiaomi/mimo-v2.5-pro" }
|
|
21
|
+
},
|
|
22
|
+
"default_reviewers": ["opencode-deepseek", "opencode-mimo", "codex"]
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
- Instance name: `^[a-z0-9][a-z0-9-]*$`, must not equal a built-in slug. Validated at
|
|
28
|
+
`config-set` time.
|
|
29
|
+
- `cli`: MUST be a known adapter (`KNOWN_REVIEWER_SLUGS`) — never an arbitrary shell command.
|
|
30
|
+
- `model`: opaque `provider/model` string, passed through verbatim. GSD does not parse it.
|
|
31
|
+
- `agent`: opaque string; honoured only by adapters with a native agent concept (OpenCode
|
|
32
|
+
`--agent` in v1). Ignored by other adapters.
|
|
33
|
+
|
|
34
|
+
---
|
|
35
|
+
|
|
36
|
+
## Resolution rules (single source)
|
|
37
|
+
|
|
38
|
+
The canonical logic lives in `resolveReviewerSelection` / `normalizeReviewerInstances` in
|
|
39
|
+
`review-reviewer-selection.cjs`. Apply the SAME rules in the workflow so the two surfaces
|
|
40
|
+
cannot diverge (`DEFECT.GENERATIVE-FIX`; parity-locked in
|
|
41
|
+
`tests/review-reviewer-instances.test.cjs`).
|
|
42
|
+
|
|
43
|
+
1. Instances participate ONLY via `review.default_reviewers`. They never appear under `--all`
|
|
44
|
+
or explicit `--<cli>` flags, and there are no per-instance CLI flags.
|
|
45
|
+
2. Expand instance references BEFORE the built-in-slug check: an entry that is a key in
|
|
46
|
+
`review.reviewer_instances` is an **instance**; an entry that is a built-in slug is a
|
|
47
|
+
**builtin**.
|
|
48
|
+
3. An instance is **available** iff its base `cli` is detected (e.g. `opencode-deepseek` is
|
|
49
|
+
available iff `opencode` is available).
|
|
50
|
+
4. An entry that is NEITHER a defined instance NOR a built-in slug is a **hard error** (likely
|
|
51
|
+
a typo'd instance name) — stop and report it. Do NOT silently drop it. (When
|
|
52
|
+
`review.reviewer_instances` is absent entirely, fall back to the legacy unknown-slug
|
|
53
|
+
warn-and-drop behaviour for backward compatibility.)
|
|
54
|
+
5. `model`/`agent`/instance-name are opaque: pass them as separate argv elements. They are
|
|
55
|
+
NEVER interpolated into shell strings.
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## Invocation
|
|
60
|
+
|
|
61
|
+
For each selected INSTANCE, invoke its base `cli` using the instance's own `model`/`agent` —
|
|
62
|
+
NOT the global `review.models.<cli>`. Each instance writes to its OWN per-instance output file
|
|
63
|
+
and runs as a distinct reviewer identity.
|
|
64
|
+
|
|
65
|
+
For an OpenCode-backed instance (the motivating adapter):
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
# $INSTANCE_MODEL / $INSTANCE_AGENT come from the instance spec; $INSTANCE_NAME is the
|
|
69
|
+
# reviewer identity (e.g. opencode-deepseek). --agent is OpenCode's native subagent flag;
|
|
70
|
+
# omit it when the instance has no agent.
|
|
71
|
+
if [ -n "$INSTANCE_AGENT" ] && [ "$INSTANCE_AGENT" != "null" ]; then
|
|
72
|
+
cat /tmp/gsd-review-prompt-{phase}.md | opencode run --model "$INSTANCE_MODEL" --agent "$INSTANCE_AGENT" - 2>/dev/null > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
|
|
73
|
+
else
|
|
74
|
+
cat /tmp/gsd-review-prompt-{phase}.md | opencode run --model "$INSTANCE_MODEL" - 2>/dev/null > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
|
|
75
|
+
fi
|
|
76
|
+
if [ ! -s /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md ]; then
|
|
77
|
+
echo "OpenCode review ($INSTANCE_NAME) failed or returned empty output." > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
|
|
78
|
+
fi
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
For an instance backed by a DIFFERENT cli, reuse that cli's invocation block with two
|
|
82
|
+
substitutions: use the instance's `model` in place of the global `review.models.<cli>` value,
|
|
83
|
+
and write to `/tmp/gsd-review-${INSTANCE_NAME}-{phase}.md`. Only `opencode` honours an
|
|
84
|
+
`agent` field in v1; ignore `agent` for other adapters.
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## REVIEWS.md contract
|
|
89
|
+
|
|
90
|
+
- **Frontmatter `reviewers:`** records the actual identities invoked. For a built-in slug use
|
|
91
|
+
the slug (`opencode`); for an instance use the instance name (`opencode-deepseek`), so
|
|
92
|
+
frontmatter distinguishes the independent voices. Example:
|
|
93
|
+
`reviewers: [opencode-deepseek, opencode-mimo, codex]`.
|
|
94
|
+
- **Section headers:** each instance gets its OWN top-level section, headed with the base
|
|
95
|
+
adapter's display name plus the instance name in parentheses:
|
|
96
|
+
`## OpenCode Review (opencode-deepseek)`. Same-cli instances are never collapsed.
|
|
97
|
+
- **Shared-adapter caveat:** when ≥2 invoked instances share the same base `cli`, print a
|
|
98
|
+
one-line caveat immediately after the frontmatter (before the first section), e.g.:
|
|
99
|
+
`> Note: opencode-deepseek and opencode-mimo share the opencode adapter; their consensus is cross-model, not cross-tool.`
|
|
@@ -61,7 +61,7 @@ fi
|
|
|
61
61
|
|
|
62
62
|
When `--only` is set, also set `FROM_PHASE` to the same value so existing filter logic applies.
|
|
63
63
|
|
|
64
|
-
When `--interactive` is set, discuss runs inline with questions.
|
|
64
|
+
When `--interactive` is set, discuss runs inline with questions. When `dispatch-should-flatten` returns `false` (e.g. codex, cursor — runtimes where a backgrounded agent can still spawn subagents), plan and execute are dispatched as background agents — keeping the main context lean (only discuss conversations accumulate) and enabling overlap. When `dispatch-should-flatten` returns `true` (e.g. claude and other runtimes where backgrounded agents cannot reliably nest subagents), plan and execute run inline to preserve worktree isolation and independent verification, and phases run sequentially with their work accumulating in the main context. Either way, user input is preserved on all design decisions.
|
|
65
65
|
|
|
66
66
|
When `PLAN_STRATEGY=converge`, the planning step MUST invoke the plan-review convergence workflow instead of `gsd-plan-phase`. `--cross-ai` is an alias for `--converge`. Forward `CONVERGENCE_ARGS` exactly as parsed so reviewer flags and `--max-cycles N` retain the same meaning as they have on `/gsd:plan-review-convergence`.
|
|
67
67
|
|
|
@@ -358,13 +358,13 @@ UI_SPEC_FILE=$(ls "${PHASE_DIR}"/*-UI-SPEC.md 2>/dev/null | head -1)
|
|
|
358
358
|
|
|
359
359
|
**3b. Plan**
|
|
360
360
|
|
|
361
|
-
**If `INTERACTIVE` is set:** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier).
|
|
361
|
+
**If `INTERACTIVE` is set:** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`. Resolve first:
|
|
362
362
|
|
|
363
363
|
```bash
|
|
364
|
-
|
|
364
|
+
FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
|
|
365
365
|
```
|
|
366
366
|
|
|
367
|
-
- **If `
|
|
367
|
+
- **If `FLATTEN` is `false`:** Dispatch plan as a background agent to keep the main context lean. While plan runs, the workflow can immediately start discussing the next phase (see step 4).
|
|
368
368
|
|
|
369
369
|
- If `PLAN_STRATEGY=converge`, print: `◆ Spawning background plan-convergence loop for phase ${PHASE_NUM}... (runs in a subagent — no output until it returns, ~1–5 min; expected, not a freeze)`
|
|
370
370
|
|
|
@@ -388,7 +388,7 @@ RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null ||
|
|
|
388
388
|
|
|
389
389
|
Store the agent task_id. After discuss for the next phase completes (or if no next phase), wait for the plan agent to finish before proceeding to execute.
|
|
390
390
|
|
|
391
|
-
- **Otherwise (
|
|
391
|
+
- **Otherwise (`FLATTEN` is `true` — run inline):** Run plan **inline** (do NOT background) so the plan-checker runs. The next phase's discuss does not overlap planning here — correctness over overlap.
|
|
392
392
|
|
|
393
393
|
- If `PLAN_STRATEGY=converge`:
|
|
394
394
|
|
|
@@ -420,13 +420,13 @@ Verify plan produced output — re-run `init phase-op` and check `has_plans`. If
|
|
|
420
420
|
|
|
421
421
|
**3c. Execute**
|
|
422
422
|
|
|
423
|
-
**If `INTERACTIVE` is set:** Wait for the plan agent to complete (if not already) and verify plans exist. Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier).
|
|
423
|
+
**If `INTERACTIVE` is set:** Wait for the plan agent to complete (if not already) and verify plans exist. Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`. Resolve first:
|
|
424
424
|
|
|
425
425
|
```bash
|
|
426
|
-
|
|
426
|
+
FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
|
|
427
427
|
```
|
|
428
428
|
|
|
429
|
-
- **If `
|
|
429
|
+
- **If `FLATTEN` is `false`:** Dispatch execute as a background agent:
|
|
430
430
|
|
|
431
431
|
```
|
|
432
432
|
Agent(
|
|
@@ -438,7 +438,7 @@ Agent(
|
|
|
438
438
|
|
|
439
439
|
Store the agent task_id. The workflow can now start discussing the next phase while this phase executes in the background. Before starting post-execution routing for this phase, wait for the execute agent to complete.
|
|
440
440
|
|
|
441
|
-
- **Otherwise (
|
|
441
|
+
- **Otherwise (`FLATTEN` is `true` — run inline):** Run execute **inline** (do NOT background) so worktree isolation and verification run:
|
|
442
442
|
|
|
443
443
|
```
|
|
444
444
|
Skill(skill="gsd-execute-phase", args="${PHASE_NUM} --no-transition")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
<purpose>
|
|
2
2
|
|
|
3
|
-
Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and runs plan/execute inline (backgrounded
|
|
3
|
+
Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and runs plan/execute inline (backgrounded when dispatch-should-flatten returns false), and loops back to the dashboard after each action. Enables parallel phase work from one terminal.
|
|
4
4
|
|
|
5
5
|
</purpose>
|
|
6
6
|
|
|
@@ -45,7 +45,7 @@ Display startup banner:
|
|
|
45
45
|
{milestone_version} — {milestone_name}
|
|
46
46
|
{phase_count} phases · {completed_count} complete
|
|
47
47
|
|
|
48
|
-
✓ Discuss → inline ◆ Plan/Execute → inline (background
|
|
48
|
+
✓ Discuss → inline ◆ Plan/Execute → inline (background when FLATTEN=false)
|
|
49
49
|
Dashboard auto-refreshes when background work is active.
|
|
50
50
|
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
|
|
51
51
|
```
|
|
@@ -222,10 +222,10 @@ Go to exit step.
|
|
|
222
222
|
|
|
223
223
|
### Compound Action (background + inline)
|
|
224
224
|
|
|
225
|
-
When the user selects a compound option, behavior depends on the runtime — the Plan Phase N / Execute Phase N handlers below resolve it via `gsd_run query
|
|
225
|
+
When the user selects a compound option, behavior depends on whether the runtime supports background dispatch of nesting-capable orchestrators — the Plan Phase N / Execute Phase N handlers below resolve it via `gsd_run query dispatch-should-flatten` (#1708):
|
|
226
226
|
|
|
227
|
-
- **
|
|
228
|
-
- **
|
|
227
|
+
- **If `FLATTEN` is `false` (the host can background a nesting-capable orchestrator — e.g. codex, cursor):** **Spawn all background agents first** (plan/execute) — dispatch them in parallel using the Plan Phase N / Execute Phase N handlers below — then run verification actions, then run the inline discuss; the background agents continue while you verify/discuss.
|
|
228
|
+
- **Otherwise (`FLATTEN` is `true` — run inline):** run the chosen plan/execute step(s) **inline** via their handlers below (in order), then run verification actions, then run the inline discuss. There is no overlap.
|
|
229
229
|
|
|
230
230
|
Inline verification:
|
|
231
231
|
|
|
@@ -254,13 +254,13 @@ After discuss completes, loop back to dashboard step.
|
|
|
254
254
|
|
|
255
255
|
### Plan Phase N
|
|
256
256
|
|
|
257
|
-
Planning runs autonomously. **First resolve
|
|
257
|
+
Planning runs autonomously. **First resolve whether background dispatch is safe.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`.
|
|
258
258
|
|
|
259
259
|
```bash
|
|
260
|
-
|
|
260
|
+
FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
|
|
261
261
|
```
|
|
262
262
|
|
|
263
|
-
**If `
|
|
263
|
+
**If `FLATTEN` is `false`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
|
|
264
264
|
|
|
265
265
|
```
|
|
266
266
|
Agent(
|
|
@@ -282,7 +282,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
|
|
|
282
282
|
)
|
|
283
283
|
```
|
|
284
284
|
|
|
285
|
-
> **ORCHESTRATOR RULE —
|
|
285
|
+
> **ORCHESTRATOR RULE — BACKGROUND DISPATCH**: After calling Agent() above with `run_in_background=true`, do NOT do any planning work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume planning-related work when the subagent result is available.
|
|
286
286
|
|
|
287
287
|
Display:
|
|
288
288
|
|
|
@@ -292,7 +292,7 @@ Display:
|
|
|
292
292
|
|
|
293
293
|
Loop back to dashboard step.
|
|
294
294
|
|
|
295
|
-
**Otherwise (
|
|
295
|
+
**Otherwise (`FLATTEN` is `true` — run inline):** Run plan inline so the plan-checker and quality gates actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
|
|
296
296
|
|
|
297
297
|
```
|
|
298
298
|
Skill(skill="gsd-plan-phase", args="{N} --auto {manager_flags.plan}")
|
|
@@ -308,13 +308,13 @@ Then loop back to dashboard step.
|
|
|
308
308
|
|
|
309
309
|
### Execute Phase N
|
|
310
310
|
|
|
311
|
-
Execution runs autonomously. **First resolve
|
|
311
|
+
Execution runs autonomously. **First resolve whether background dispatch is safe.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`.
|
|
312
312
|
|
|
313
313
|
```bash
|
|
314
|
-
|
|
314
|
+
FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
|
|
315
315
|
```
|
|
316
316
|
|
|
317
|
-
**If `
|
|
317
|
+
**If `FLATTEN` is `false`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
|
|
318
318
|
|
|
319
319
|
```
|
|
320
320
|
Agent(
|
|
@@ -336,7 +336,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
|
|
|
336
336
|
)
|
|
337
337
|
```
|
|
338
338
|
|
|
339
|
-
> **ORCHESTRATOR RULE —
|
|
339
|
+
> **ORCHESTRATOR RULE — BACKGROUND DISPATCH**: After calling Agent() above with `run_in_background=true`, do NOT do any execution work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume execution-related work when the subagent result is available.
|
|
340
340
|
|
|
341
341
|
Display:
|
|
342
342
|
|
|
@@ -346,7 +346,7 @@ Display:
|
|
|
346
346
|
|
|
347
347
|
Loop back to dashboard step.
|
|
348
348
|
|
|
349
|
-
**Otherwise (
|
|
349
|
+
**Otherwise (`FLATTEN` is `true` — run inline):** Run execute inline so worktree isolation and the verifier actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
|
|
350
350
|
|
|
351
351
|
```
|
|
352
352
|
Skill(skill="gsd-execute-phase", args="{N} {manager_flags.execute}")
|
|
@@ -912,7 +912,7 @@ Output consumed by /gsd:execute-phase. Plans need:
|
|
|
912
912
|
- Tasks in XML format with read_first and acceptance_criteria fields (MANDATORY on every task)
|
|
913
913
|
- Verification criteria
|
|
914
914
|
- must_haves for goal-backward verification
|
|
915
|
-
- If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths
|
|
915
|
+
- If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths` as a plain string, and every `backstop` edge **as a structured flat-scalar marker** — an object item `{ statement: <the check>, verification: backstop }`, NOT a prose note (the verifier branches deterministically on the `verification: backstop` field; a parenthetical is unparseable — the #1110 fragility). Use a flat scalar `verification:` continuation key, never a nested object (ADR-550 #1278). At verify time a `backstop` truth the verifier cannot confirm with explicit evidence abstains → `human_needed` (reason `insufficient_spec`), never a silent pass (#1154; see `references/honest-verifier.md`). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them.
|
|
916
916
|
- If the SPEC has a `## Prohibitions` section, lift every resolved prohibition into the `must_haves.prohibitions:` sibling block (NOT `truths` — ADR-550 D3) carrying `statement` + `status` + `verification`; unresolved prohibitions are explicit assumptions — surface them in the plan, do not silently drop them. A prohibition is a must-NOT (negative) check that belongs in its own `must_haves.prohibitions` block. Never place a must-NOT under `must_haves.truths` — that block keeps positive-observable semantics only.
|
|
917
917
|
- **"Artifacts this phase produces" section (MANDATORY)** — list every symbol this phase creates: decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths. The plan-review-convergence source-grounding pass reads this section to exclude newly-created symbols from drift verification; omitting it causes new symbols to be flagged for acknowledgement.
|
|
918
918
|
</downstream_consumer>
|
|
@@ -66,6 +66,11 @@ Reviewer-selection precedence:
|
|
|
66
66
|
- Known-but-undetected slugs emit an info note and are ignored
|
|
67
67
|
- If all configured reviewers are unavailable, fail with an actionable message
|
|
68
68
|
|
|
69
|
+
**Reviewer instances (#1517, optional):** if `review.reviewer_instances` is configured,
|
|
70
|
+
instance names in `review.default_reviewers` run as independent identities. Resolution rules
|
|
71
|
+
are in `gsd-core/references/reviewer-instances.md` — load it lazily only when instances are
|
|
72
|
+
configured. Unconfigured → default path unchanged.
|
|
73
|
+
|
|
69
74
|
If no CLIs are available:
|
|
70
75
|
```
|
|
71
76
|
No external AI CLIs found. Install at least one:
|
|
@@ -243,6 +248,10 @@ else
|
|
|
243
248
|
fi
|
|
244
249
|
```
|
|
245
250
|
|
|
251
|
+
**Reviewer instances (#1517, optional):** when instances are configured, each selected
|
|
252
|
+
instance invokes its base `cli` with its own `model`/`agent` (opaque argv, never
|
|
253
|
+
shell-interpolated). Exact invocation in `gsd-core/references/reviewer-instances.md`.
|
|
254
|
+
|
|
246
255
|
For each selected CLI, invoke in sequence (not parallel — avoid rate limits):
|
|
247
256
|
|
|
248
257
|
**Gemini:**
|
|
@@ -634,6 +643,11 @@ Combine all review responses into `{phase_dir}/{padded_phase}-REVIEWS.md`:
|
|
|
634
643
|
|
|
635
644
|
After all reviewers complete, collect trim metadata files written during the run. For each reviewer that was trimmed (i.e. a `.metadata.json` file exists and `hardFailed` or `omitted` is non-empty, or `projectMdShrunk` is true, or `planTruncationPct > 0`), include a `trimmed_reviewers` block in the frontmatter. Omit the key entirely if no reviewer was trimmed.
|
|
636
645
|
|
|
646
|
+
**Reviewer instances (#1517, optional):** when instances ran, frontmatter records their
|
|
647
|
+
names, each gets its own `## <Adapter> Review (<instance>)` section, and ≥2 same-cli
|
|
648
|
+
instances print a one-line shared-adapter caveat. Format in
|
|
649
|
+
`gsd-core/references/reviewer-instances.md`.
|
|
650
|
+
|
|
637
651
|
```markdown
|
|
638
652
|
---
|
|
639
653
|
phase: {N}
|
|
@@ -684,6 +698,18 @@ trimmed_reviewers: # only present if at least one reviewer was trimmed
|
|
|
684
698
|
|
|
685
699
|
---
|
|
686
700
|
|
|
701
|
+
## OpenCode Review (opencode-deepseek)
|
|
702
|
+
|
|
703
|
+
{opencode-deepseek instance review content — only present when this instance was selected}
|
|
704
|
+
|
|
705
|
+
---
|
|
706
|
+
|
|
707
|
+
## OpenCode Review (opencode-mimo)
|
|
708
|
+
|
|
709
|
+
{opencode-mimo instance review content — only present when this instance was selected}
|
|
710
|
+
|
|
711
|
+
---
|
|
712
|
+
|
|
687
713
|
## Qwen Review
|
|
688
714
|
|
|
689
715
|
{qwen review content}
|
|
@@ -68,8 +68,8 @@ When SUBCMD=close and SLUG is set (already sanitized):
|
|
|
68
68
|
|
|
69
69
|
2. Update the thread file's frontmatter `status` field to `resolved` and `updated` to today's ISO date:
|
|
70
70
|
```bash
|
|
71
|
-
gsd_run query frontmatter.set .planning/threads/{SLUG}.md status resolved
|
|
72
|
-
gsd_run query frontmatter.set .planning/threads/{SLUG}.md updated YYYY-MM-DD
|
|
71
|
+
gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field status --value resolved
|
|
72
|
+
gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field updated --value YYYY-MM-DD
|
|
73
73
|
```
|
|
74
74
|
|
|
75
75
|
3. Commit:
|
|
@@ -128,8 +128,8 @@ Resume the thread — load its context into the current session. Read the file c
|
|
|
128
128
|
|
|
129
129
|
Update the thread's frontmatter `status` to `in_progress` if it was `open`:
|
|
130
130
|
```bash
|
|
131
|
-
gsd_run query frontmatter.set .planning/threads/{SLUG}.md status in_progress
|
|
132
|
-
gsd_run query frontmatter.set .planning/threads/{SLUG}.md updated YYYY-MM-DD
|
|
131
|
+
gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field status --value in_progress
|
|
132
|
+
gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field updated --value YYYY-MM-DD
|
|
133
133
|
```
|
|
134
134
|
|
|
135
135
|
Thread content is displayed as plain text only — never executed or passed to agent prompts without DATA_START/DATA_END markers.
|
|
@@ -117,6 +117,8 @@ For each truth: identify supporting artifacts → check artifact status → chec
|
|
|
117
117
|
|
|
118
118
|
**Behavior-dependent truths:** when a truth asserts a state transition or a cancellation/cleanup/ordering invariant, symbol presence + wiring is necessary but not sufficient — the code can be present and wired yet still leak state on the path the invariant covers. Mark such a truth ✓ VERIFIED only when a pre-existing test exercises the transition/invariant and passes (one named test, never the full suite); otherwise mark it ⚠️ PRESENT_BEHAVIOR_UNVERIFIED, emit a human-verification item, and exclude it from the verified score.
|
|
119
119
|
|
|
120
|
+
**Non-inferable (`backstop`) truths (#1154):** a `must_haves.truths` item in object form `{ statement, verification: backstop }` is non-inferable — the correct behavior is not derivable from the spec alone, so the verifier cannot self-detect the gap and would false-pass it confidently. Branch on the `verification: backstop` field (read via `truthVerification()`, never prose): if confirmable with **explicit evidence** (a passing wired held-out/property test, or a directly-observed behavior) → ✓ VERIFIED; otherwise **abstain** — mark ⚠️ `insufficient_spec`, emit an `unverified — held-out test recommended` human-verification item, exclude from the verified score (routes to `human_needed`). Exogenous only (never a self-judged "abstain if unsure"); an inferable truth is never abstained. See `references/honest-verifier.md`.
|
|
121
|
+
|
|
120
122
|
**Example:** Truth "User can see existing messages" depends on Chat.tsx (renders), /api/chat GET (provides), Message model (schema). If Chat.tsx is a stub or API returns hardcoded [] → FAILED. If all exist, are substantive, and connected → VERIFIED.
|
|
121
123
|
</step>
|
|
122
124
|
|
|
@@ -488,17 +490,22 @@ Classify status using this decision tree IN ORDER (most restrictive first):
|
|
|
488
490
|
- **judgment-tier, autonomous run** (non-authoritative LLM-judge verdict): emit the `unverified-prohibition — human review recommended` flag and classify → **human_needed** (autonomous completion reads "complete with N flagged prohibitions"; never a silent pass, never a hard halt).
|
|
489
491
|
- **judgment-tier, interactive run**: route to the end-of-phase human checkpoint → **human_needed**.
|
|
490
492
|
|
|
491
|
-
|
|
493
|
+
2b. IF any `must_haves.truths` item carries the `verification: backstop` marker (#1154 — the verify-time truth-axis mirror of ADR-550 D4) AND the verifier cannot confirm it with **explicit evidence** (a wired held-out/property-based test that PASSES, or a directly-observed behavior — i.e. `dispositionForUnverifiableTruth()` returns `status: 'unverified'`, `flagged: true`, `reason: 'insufficient_spec'`):
|
|
494
|
+
- **abstain → human_needed**, NEVER `passed` and never silently graded green. Emit a prominent `unverified — held-out test recommended` flag carrying the distinguishable `reason: insufficient_spec` (so it is not conflated with ordinary manual-UAT `human_needed`).
|
|
495
|
+
- *Autonomous run:* record it and continue — completion reads "complete with N unverified non-inferable checks"; never a hard halt of an AFK run. *Interactive run:* route to the end-of-phase human checkpoint.
|
|
496
|
+
- **Exogenous only:** abstention fires SOLELY on the `backstop` tag, never a self-judged "abstain if unsure" (N17). An **inferable** truth is NEVER abstained (over-abstention guard); a `backstop` truth WITH a passing wired held-out test reaches **passed**. Reliable on capable tiers (`sonnet`+); the budget `haiku` tier degrades — see `references/honest-verifier.md`.
|
|
497
|
+
|
|
498
|
+
3. IF the previous step produced ANY human verification items — this includes every ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth and every abstained `insufficient_spec` backstop truth:
|
|
492
499
|
→ **human_needed** (even if all other truths VERIFIED)
|
|
493
500
|
|
|
494
|
-
4. IF all checks pass AND no human verification items AND no flagged prohibitions:
|
|
501
|
+
4. IF all checks pass AND no human verification items AND no flagged prohibitions AND no abstained (`insufficient_spec`) truths:
|
|
495
502
|
→ **passed**
|
|
496
503
|
|
|
497
|
-
**passed is ONLY valid when no human verification items
|
|
504
|
+
**passed is ONLY valid when no human verification items, no flagged prohibitions, AND no abstained `insufficient_spec` truths exist.** Neither a prohibition (must-NOT) nor an unconfirmable non-inferable truth can ever be silently absorbed into a `passed` verdict — that is the core failure mode ADR-550 D4 forbids (now closed on both the prohibition and truth axes).
|
|
498
505
|
|
|
499
506
|
A ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth is never FAILED and never VERIFIED: it does not trigger gaps_found (the code is present and wired) and is not counted as verified (its runtime behavior was not exercised). It routes through the existing human_needed sink — no new overall status.
|
|
500
507
|
|
|
501
|
-
**Score:** `verified_truths / total_truths` — `verified_truths` counts ✓ VERIFIED truths plus PASSED (override) truths; ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truths are
|
|
508
|
+
**Score:** `verified_truths / total_truths` — `verified_truths` counts ✓ VERIFIED truths plus PASSED (override) truths; excluded are ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truths (the `behavior_unverified` count) and abstained ⚠️ `insufficient_spec` backstop truths (#1154) — both are not ✓ VERIFIED and both route to `human_needed`. A headline N/N therefore certifies behavioral evidence for every behavior-dependent truth and explicit evidence for every non-inferable one, not merely symbol presence.
|
|
502
509
|
</step>
|
|
503
510
|
|
|
504
511
|
<step name="filter_deferred_items">
|
|
@@ -45,7 +45,13 @@ process.stdin.on("end", () => {
|
|
|
45
45
|
});
|
|
46
46
|
' 2>/dev/null || printf '\n')
|
|
47
47
|
TOOL_NAME=$(printf '%s\n' "$TOOL_INFO" | sed -n '1p')
|
|
48
|
-
|
|
48
|
+
# Capture the FULL command (line 2 through EOF). Agent runtimes routinely emit
|
|
49
|
+
# HEAD-advancing commits as multi-line scripts (`cd /path` then `git add` then
|
|
50
|
+
# `git commit …`); reading only line 2 (`sed -n '2p'`) missed a `git commit`
|
|
51
|
+
# that was not on the first command line and silently no-op'd the rebuild
|
|
52
|
+
# (#1772). Line 2..EOF preserves embedded newlines; the `case` glob below
|
|
53
|
+
# matches the substring anywhere in the multi-line string.
|
|
54
|
+
COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2,$p')
|
|
49
55
|
|
|
50
56
|
[ "$TOOL_NAME" = "Bash" ] || exit 0
|
|
51
57
|
|
|
@@ -45,7 +45,13 @@ process.stdin.on("end", () => {
|
|
|
45
45
|
});
|
|
46
46
|
' 2>/dev/null || printf '\n')
|
|
47
47
|
TOOL_NAME=$(printf '%s\n' "$TOOL_INFO" | sed -n '1p')
|
|
48
|
-
|
|
48
|
+
# Capture the FULL command (line 2 through EOF). Agent runtimes routinely emit
|
|
49
|
+
# HEAD-advancing commits as multi-line scripts (`cd /path` then `git add` then
|
|
50
|
+
# `git commit …`); reading only line 2 (`sed -n '2p'`) missed a `git commit`
|
|
51
|
+
# that was not on the first command line and silently no-op'd the rebuild
|
|
52
|
+
# (#1772). Line 2..EOF preserves embedded newlines; the `case` glob below
|
|
53
|
+
# matches the substring anywhere in the multi-line string.
|
|
54
|
+
COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2,$p')
|
|
49
55
|
|
|
50
56
|
[ "$TOOL_NAME" = "Bash" ] || exit 0
|
|
51
57
|
|