@opengsd/gsd-core 1.6.0 → 1.7.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/agents/gsd-verifier.md +1 -0
  3. package/bin/gsd-mcp-server.js +31 -0
  4. package/bin/install.js +293 -1145
  5. package/commands/gsd/review.md +6 -0
  6. package/gemini-extension.json +1 -1
  7. package/gsd-core/bin/gsd-tools.cjs +116 -1
  8. package/gsd-core/bin/lib/adapter-declarative.cjs +35 -0
  9. package/gsd-core/bin/lib/adapter-imperative.cjs +52 -0
  10. package/gsd-core/bin/lib/assumption-delta.cjs +231 -0
  11. package/gsd-core/bin/lib/capability-lifecycle.cjs +7 -7
  12. package/gsd-core/bin/lib/capability-loader.cjs +18 -0
  13. package/gsd-core/bin/lib/capability-lock.cjs +2 -2
  14. package/gsd-core/bin/lib/capability-registry.cjs +889 -82
  15. package/gsd-core/bin/lib/capability-source.cjs +4 -4
  16. package/gsd-core/bin/lib/capability-validator.cjs +198 -0
  17. package/gsd-core/bin/lib/cli-skew-check.cjs +44 -0
  18. package/gsd-core/bin/lib/command-aliases.cjs +8 -0
  19. package/gsd-core/bin/lib/config.cjs +27 -0
  20. package/gsd-core/bin/lib/embedding-adapter.cjs +27 -0
  21. package/gsd-core/bin/lib/external-descriptor-trust.cjs +70 -0
  22. package/gsd-core/bin/lib/hook-bus.cjs +81 -0
  23. package/gsd-core/bin/lib/host-integration.cjs +408 -0
  24. package/gsd-core/bin/lib/init.cjs +1 -1
  25. package/gsd-core/bin/lib/install-engine.cjs +755 -0
  26. package/gsd-core/bin/lib/install-profiles.cjs +35 -4
  27. package/gsd-core/bin/lib/installer-migrations.cjs +1 -1
  28. package/gsd-core/bin/lib/mcp-server.cjs +194 -0
  29. package/gsd-core/bin/lib/milestone.cjs +27 -30
  30. package/gsd-core/bin/lib/model-adapter.cjs +50 -0
  31. package/gsd-core/bin/lib/phase.cjs +41 -72
  32. package/gsd-core/bin/lib/planning-workspace.cjs +1 -1
  33. package/gsd-core/bin/lib/probe-core.cjs +91 -1
  34. package/gsd-core/bin/lib/review-reviewer-selection.cjs +129 -13
  35. package/gsd-core/bin/lib/roadmap-upgrade.cjs +3 -2
  36. package/gsd-core/bin/lib/roadmap.cjs +17 -3
  37. package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +65 -9
  38. package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +54 -4
  39. package/gsd-core/bin/lib/runtime-artifact-layout.cjs +5 -2
  40. package/gsd-core/bin/lib/runtime-hooks-surface.cjs +1 -1
  41. package/gsd-core/bin/lib/runtime-name-policy.cjs +160 -30
  42. package/gsd-core/bin/lib/shell-command-projection.cjs +37 -1
  43. package/gsd-core/bin/lib/stale-bake-guard.cjs +254 -0
  44. package/gsd-core/bin/lib/state-command-router.cjs +4 -0
  45. package/gsd-core/bin/lib/state-io.cjs +55 -0
  46. package/gsd-core/bin/lib/state-transition.cjs +1588 -0
  47. package/gsd-core/bin/lib/state.cjs +306 -681
  48. package/gsd-core/bin/lib/surface.cjs +4 -1
  49. package/gsd-core/bin/lib/workstream.cjs +4 -4
  50. package/gsd-core/bin/shared/config-schema.manifest.json +9 -0
  51. package/gsd-core/references/honest-verifier.md +105 -0
  52. package/gsd-core/references/reviewer-instances.md +99 -0
  53. package/gsd-core/workflows/autonomous.md +9 -9
  54. package/gsd-core/workflows/manager.md +15 -15
  55. package/gsd-core/workflows/plan-phase.md +1 -1
  56. package/gsd-core/workflows/review.md +26 -0
  57. package/gsd-core/workflows/thread.md +4 -4
  58. package/gsd-core/workflows/verify-phase.md +11 -4
  59. package/hooks/dist/gsd-graphify-update.sh +7 -1
  60. package/hooks/gsd-graphify-update.sh +7 -1
  61. package/package.json +4 -4
  62. package/scripts/ci-test-scope.cjs +38 -9
  63. package/scripts/lint-allow-test-rule-refs.allowlist.json +0 -1
  64. package/scripts/lint-regression-test-names.allowlist.json +3 -0
  65. package/scripts/lint-test-file-count.allowlist.json +19 -5
  66. package/scripts/mutation-matrix.cjs +45 -3
  67. package/scripts/prompt-injection-scan.sh +8 -0
  68. package/scripts/lint-windows-test-portability.cjs +0 -178
@@ -43,6 +43,9 @@ const runtimeArtifactLayout = require("./runtime-artifact-layout.cjs");
43
43
  const { findInstallSourceRoot } = runtimeArtifactLayout;
44
44
  // eslint-disable-next-line @typescript-eslint/no-require-imports
45
45
  const runtimeArtifactConversion = require("./runtime-artifact-conversion.cjs");
46
+ // eslint-disable-next-line @typescript-eslint/no-require-imports
47
+ const runtimeArtifactInstallPlan = require("./runtime-artifact-install-plan.cjs");
48
+ const { assertDestWithinConfigHome } = runtimeArtifactInstallPlan;
46
49
  const SURFACE_FILE_NAME = '.gsd-surface.json';
47
50
  /**
48
51
  * Read the surface state from a runtime config directory.
@@ -302,7 +305,7 @@ function applySurface(runtimeConfigDir, layout, manifest, clusterMap, registry)
302
305
  tempDirsToClean.push(rewritten);
303
306
  }
304
307
  }
305
- const dest = node_path_1.default.join(layout.configDir, kind.destSubpath);
308
+ const dest = assertDestWithinConfigHome(layout.configDir, kind.destSubpath);
306
309
  _syncGsdDir(staged, dest, kind, skillManifest);
307
310
  }
308
311
  }
@@ -67,7 +67,7 @@ function migrateToWorkstreams(cwd, workstreamName) {
67
67
  const src = node_path_1.default.join(baseDir, item.name);
68
68
  if (node_fs_1.default.existsSync(src)) {
69
69
  const dest = node_path_1.default.join(wsDir, item.name);
70
- node_fs_1.default.renameSync(src, dest);
70
+ (0, shell_command_projection_cjs_1.retryRenameSync)(src, dest);
71
71
  filesMoved.push(item.name);
72
72
  }
73
73
  }
@@ -75,7 +75,7 @@ function migrateToWorkstreams(cwd, workstreamName) {
75
75
  catch (err) {
76
76
  for (const name of filesMoved) {
77
77
  try {
78
- node_fs_1.default.renameSync(node_path_1.default.join(wsDir, name), node_path_1.default.join(baseDir, name));
78
+ (0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(wsDir, name), node_path_1.default.join(baseDir, name));
79
79
  }
80
80
  catch { /* ignore */ }
81
81
  }
@@ -275,14 +275,14 @@ function cmdWorkstreamComplete(cwd, name, options, raw) {
275
275
  try {
276
276
  const entries = node_fs_1.default.readdirSync(wsDir, { withFileTypes: true });
277
277
  for (const entry of entries) {
278
- node_fs_1.default.renameSync(node_path_1.default.join(wsDir, entry.name), node_path_1.default.join(archivePath, entry.name));
278
+ (0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(wsDir, entry.name), node_path_1.default.join(archivePath, entry.name));
279
279
  filesMoved.push(entry.name);
280
280
  }
281
281
  }
282
282
  catch (err) {
283
283
  for (const fname of filesMoved) {
284
284
  try {
285
- node_fs_1.default.renameSync(node_path_1.default.join(archivePath, fname), node_path_1.default.join(wsDir, fname));
285
+ (0, shell_command_projection_cjs_1.retryRenameSync)(node_path_1.default.join(archivePath, fname), node_path_1.default.join(wsDir, fname));
286
286
  }
287
287
  catch { /* ignore */ }
288
288
  }
@@ -10,6 +10,10 @@
10
10
  "brave_search",
11
11
  "firecrawl",
12
12
  "exa_search",
13
+ "tavily_search",
14
+ "ref_search",
15
+ "perplexity",
16
+ "jina",
13
17
  "workflow.plan_check",
14
18
  "workflow.verifier",
15
19
  "workflow.auto_advance",
@@ -172,6 +176,11 @@
172
176
  "source": "^review\\.max_prompt_tokens_per_reviewer\\.[a-zA-Z0-9_-]+$",
173
177
  "description": "review.max_prompt_tokens_per_reviewer.<reviewer-slug>"
174
178
  },
179
+ {
180
+ "topLevel": "review",
181
+ "source": "^review\\.reviewer_instances\\.[a-zA-Z0-9_-]+\\.(cli|model|agent)$",
182
+ "description": "review.reviewer_instances.<instance-name>.<cli|model|agent> (#1517)"
183
+ },
175
184
  {
176
185
  "topLevel": "model_policy",
177
186
  "source": "^model_policy\\.runtime_tiers\\.[a-zA-Z0-9_-]+\\.(opus|sonnet|haiku)$",
@@ -0,0 +1,105 @@
1
+ # Honest Verifier — Abstention on Non-Inferable Checks
2
+
3
+ Shared reference for the **verify** phase. The verify-time companion to the spec-time
4
+ `@~/.claude/gsd-core/references/edge-probe.md` (which *classifies* non-inferable checks) and
5
+ `@~/.claude/gsd-core/references/prohibition-probe.md` (whose judgment-tier disposition this mirrors).
6
+ This doc is written in generic `spec → predicate → verifier` terms with no tool-specific vocabulary,
7
+ so it is portable: copy it into any verification process.
8
+
9
+ ## The problem it solves
10
+
11
+ A verifier is trustworthy on **inferable** checks — defects determined by the stated spec. On a
12
+ **non-inferable** check the correct answer is *not derivable from the spec alone* (e.g. "does `[1,2]`
13
+ touching `[2,3]` merge?", "is a 'character' a grapheme or a code unit?"). On these the verifier *does
14
+ not know that it does not know*: measured behavior is a **confident PASS on the blind-spot check ~100%
15
+ of the time** (mean confidence ~0.93), because a model cannot self-detect a gap it does not perceive.
16
+
17
+ The edge-probe already detects these at spec time and tags them `verification: backstop` (ADR-550
18
+ D7a). The honest verifier consumes that tag so the verifier **abstains** instead of confidently
19
+ false-passing — converting a silent false-pass (the worst failure: you don't know to look) into an
20
+ explicit, actionable "write a held-out test." Measured: the confident-false-pass rate on the blind
21
+ spot drops **100% → 17%** (N17).
22
+
23
+ ## The two properties that define the design
24
+
25
+ 1. **Exogenous, not endogenous.** The trigger is the *external tag* (`backstop`), never the verifier's
26
+ self-judgment. Asking the verifier to "abstain if unsure" barely moves the number (100% → 67%) and
27
+ only on ambiguity it already notices; on a true blind spot it stays confidently wrong. A confidence
28
+ gate cannot reach a blind spot the model does not feel — so there is **no "are you sure?" prompt**;
29
+ routing is on the pre-existing tag only.
30
+ 2. **Routing, not diagnosis.** The verifier need not name the omitted rule (if it could, it wouldn't
31
+ be a blind spot). In testing, verifiers abstained correctly while citing the *wrong* edge. The
32
+ honest verdict requires only "I was told this is under-specified and I cannot rule it out." The
33
+ omitted rule is carried by a human-authored held-out test, not by the verifier.
34
+
35
+ ## The disposition (the protocol)
36
+
37
+ For each `must_haves.truths` item:
38
+
39
+ | Item | Confirmable with explicit evidence? | Disposition |
40
+ |---|---|---|
41
+ | Inferable (plain string, or `verification: explicit`) | n/a — graded normally | ✓ VERIFIED / ✗ FAILED as usual; **never abstained** (over-abstention guard) |
42
+ | Non-inferable (`verification: backstop`) | **yes** (a wired held-out/property-based test that passes, or a directly-observed behavior) | ✓ VERIFIED |
43
+ | Non-inferable (`verification: backstop`) | **no** | **abstain** → ⚠️ `insufficient_spec`, flagged, → `human_needed` — **never `passed`** |
44
+
45
+ - **Explicit evidence** = a wired held-out/property-based test that passes, or a behavior the verifier
46
+ directly observed. Symbol presence + wiring is **not** explicit evidence for a non-inferable truth.
47
+ - **Never silent, never a hard halt.** *Interactive:* the abstained item routes to the end-of-phase
48
+ human checkpoint. *Autonomous (AFK):* it produces a prominent `unverified — held-out test
49
+ recommended` flag and the completion line reads "complete with N unverified non-inferable checks";
50
+ the run neither silently passes the blind spot nor hard-halts.
51
+ - **Distinguishable reason.** The abstain disposition carries `reason: insufficient_spec` so the
52
+ `human_needed` outcome is never conflated with an ordinary manual-UAT `human_needed`.
53
+
54
+ This is the verify-time half of ADR-550 Decision 4 (the never-silent-pass disposition), applied to the
55
+ edge `backstop` truth tier instead of the prohibition judgment tier — the same machinery, opposite
56
+ polarity (must-HAVE under-specified vs must-NOT irreducible).
57
+
58
+ ## Deterministic engine surface
59
+
60
+ The CI-testable surface is the **deterministic disposition + projection**, never the LLM's judgment
61
+ (ADR-550 D5 — a test asserting the model's verdict is vacuous and rejected). In `probe-core`:
62
+
63
+ - `truthStatement(t)` / `truthVerification(t)` — normalizers; read a truth's statement and tier from
64
+ either the plain-string or object form (a truth-reader MUST normalize, never assume a string).
65
+ - `projectTruths(items)` — conservative serializer: a `backstop` truth → flat-scalar object
66
+ `{ statement, verification: backstop }`; every inferable truth → a bare string.
67
+ - `dispositionForUnverifiableTruth(truth, { evidence })` → `{ status, flagged, tier, reason }`:
68
+ `backstop` + no evidence → `unverified`/`flagged`/`insufficient_spec`; `backstop` + evidence →
69
+ `green`; non-`backstop` → `green` (over-abstention guard).
70
+
71
+ ## Capable-tier requirement (a documented cost)
72
+
73
+ Abstention is **model-tier dependent** and this is a standing cost, not an assumption:
74
+
75
+ - The default `gsd-verifier` tier (`sonnet`, golden/balanced) heeds the exogenous tag reliably
76
+ (2/2 under testing).
77
+ - The **budget tier (`haiku`)** is the least flag-responsive (1/2, inconsistent) and **degrades toward
78
+ current behavior** (confident false-pass). Run honest-verifier on a capable tier; treat the budget
79
+ tier as best-effort. Re-validate when the `gsd-verifier` model tier changes or a new budget model is
80
+ adopted (captured as a test so a tier regression is caught, not discovered in production).
81
+
82
+ ## Evidence and scope (stated honestly)
83
+
84
+ - **Evidence strength.** N17 is n=27 verdicts (3 models × 3 conditions × 3 tasks), 1 rep —
85
+ **direction-finding, not powered.** The blind-spot effect is large and monotone
86
+ (100% → 67% → 17%); the two costs are clean single events (a *false* tag made the strongest model
87
+ over-abstain on a real spec-determined bug; the weakest tier was flag-deaf) and they name exactly
88
+ the failure modes the over-abstention guard and the capable-tier requirement defend against.
89
+ - **Tag-precision coupling.** Quality is bounded by the edge-probe's `backstop` recall/precision — a
90
+ false non-inferable flag causes over-abstention. Positive coupling: improving the probe (#1110)
91
+ improves this for free. It adds no independent burden.
92
+ - **Explicit non-goals.** Does NOT identify the omitted rule; does NOT recalibrate decisive verdicts;
93
+ does NOT defend against *malicious compliance* (a self-graded review rationalizing away its own
94
+ findings). It raises the floor on *honest* uncertainty about non-inferable checks — that is the
95
+ whole claim.
96
+
97
+ ## Distinct from neighbours
98
+
99
+ - **vs `PRESENT_BEHAVIOR_UNVERIFIED` (#966 axis):** that is the *inferable-but-unobserved* case — the
100
+ truth **can** be verified from the spec but was shortcut-passed on symbol presence; the fix is to
101
+ demand behavioral evidence. Honest-verifier is the *non-inferable* case — the truth **cannot** be
102
+ verified from the spec at all; the fix is to abstain and route to a held-out test. Orthogonal axes
103
+ (insufficient *evidence* vs insufficient *spec*); both feed the same `human_needed` sink.
104
+ - **vs prohibition judgment-tier (#644):** that disposes **must-NOT** constraints; honest-verifier
105
+ disposes **non-inferable positive truths**. Opposite polarity, same never-silent disposition.
@@ -0,0 +1,99 @@
1
+ # Reviewer Instances (#1517)
2
+
3
+ Custom reviewer instances for `/gsd:review`: run one model-capable adapter (e.g. OpenCode)
4
+ as several independent reviewer identities in a single review pass. Loaded lazily by
5
+ `gsd-core/workflows/review.md` when `review.reviewer_instances` is configured. See
6
+ [ADR-1517](../docs/adr/1517-reviewer-instances-config-surface.md) for the contract.
7
+
8
+ ---
9
+
10
+ ## Config shape
11
+
12
+ A `review.reviewer_instances` object under the `review` namespace. Each entry maps an
13
+ instance name to `{ cli, model?, agent? }`:
14
+
15
+ ```json
16
+ {
17
+ "review": {
18
+ "reviewer_instances": {
19
+ "opencode-deepseek": { "cli": "opencode", "model": "deepseek/deepseek-v4-pro", "agent": "review" },
20
+ "opencode-mimo": { "cli": "opencode", "model": "xiaomi/mimo-v2.5-pro" }
21
+ },
22
+ "default_reviewers": ["opencode-deepseek", "opencode-mimo", "codex"]
23
+ }
24
+ }
25
+ ```
26
+
27
+ - Instance name: `^[a-z0-9][a-z0-9-]*$`, must not equal a built-in slug. Validated at
28
+ `config-set` time.
29
+ - `cli`: MUST be a known adapter (`KNOWN_REVIEWER_SLUGS`) — never an arbitrary shell command.
30
+ - `model`: opaque `provider/model` string, passed through verbatim. GSD does not parse it.
31
+ - `agent`: opaque string; honoured only by adapters with a native agent concept (OpenCode
32
+ `--agent` in v1). Ignored by other adapters.
33
+
34
+ ---
35
+
36
+ ## Resolution rules (single source)
37
+
38
+ The canonical logic lives in `resolveReviewerSelection` / `normalizeReviewerInstances` in
39
+ `review-reviewer-selection.cjs`. Apply the SAME rules in the workflow so the two surfaces
40
+ cannot diverge (`DEFECT.GENERATIVE-FIX`; parity-locked in
41
+ `tests/review-reviewer-instances.test.cjs`).
42
+
43
+ 1. Instances participate ONLY via `review.default_reviewers`. They never appear under `--all`
44
+ or explicit `--<cli>` flags, and there are no per-instance CLI flags.
45
+ 2. Expand instance references BEFORE the built-in-slug check: an entry that is a key in
46
+ `review.reviewer_instances` is an **instance**; an entry that is a built-in slug is a
47
+ **builtin**.
48
+ 3. An instance is **available** iff its base `cli` is detected (e.g. `opencode-deepseek` is
49
+ available iff `opencode` is available).
50
+ 4. An entry that is NEITHER a defined instance NOR a built-in slug is a **hard error** (likely
51
+ a typo'd instance name) — stop and report it. Do NOT silently drop it. (When
52
+ `review.reviewer_instances` is absent entirely, fall back to the legacy unknown-slug
53
+ warn-and-drop behaviour for backward compatibility.)
54
+ 5. `model`/`agent`/instance-name are opaque: pass them as separate argv elements. They are
55
+ NEVER interpolated into shell strings.
56
+
57
+ ---
58
+
59
+ ## Invocation
60
+
61
+ For each selected INSTANCE, invoke its base `cli` using the instance's own `model`/`agent` —
62
+ NOT the global `review.models.<cli>`. Each instance writes to its OWN per-instance output file
63
+ and runs as a distinct reviewer identity.
64
+
65
+ For an OpenCode-backed instance (the motivating adapter):
66
+
67
+ ```bash
68
+ # $INSTANCE_MODEL / $INSTANCE_AGENT come from the instance spec; $INSTANCE_NAME is the
69
+ # reviewer identity (e.g. opencode-deepseek). --agent is OpenCode's native subagent flag;
70
+ # omit it when the instance has no agent.
71
+ if [ -n "$INSTANCE_AGENT" ] && [ "$INSTANCE_AGENT" != "null" ]; then
72
+ cat /tmp/gsd-review-prompt-{phase}.md | opencode run --model "$INSTANCE_MODEL" --agent "$INSTANCE_AGENT" - 2>/dev/null > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
73
+ else
74
+ cat /tmp/gsd-review-prompt-{phase}.md | opencode run --model "$INSTANCE_MODEL" - 2>/dev/null > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
75
+ fi
76
+ if [ ! -s /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md ]; then
77
+ echo "OpenCode review ($INSTANCE_NAME) failed or returned empty output." > /tmp/gsd-review-${INSTANCE_NAME}-{phase}.md
78
+ fi
79
+ ```
80
+
81
+ For an instance backed by a DIFFERENT cli, reuse that cli's invocation block with two
82
+ substitutions: use the instance's `model` in place of the global `review.models.<cli>` value,
83
+ and write to `/tmp/gsd-review-${INSTANCE_NAME}-{phase}.md`. Only `opencode` honours an
84
+ `agent` field in v1; ignore `agent` for other adapters.
85
+
86
+ ---
87
+
88
+ ## REVIEWS.md contract
89
+
90
+ - **Frontmatter `reviewers:`** records the actual identities invoked. For a built-in slug use
91
+ the slug (`opencode`); for an instance use the instance name (`opencode-deepseek`), so
92
+ frontmatter distinguishes the independent voices. Example:
93
+ `reviewers: [opencode-deepseek, opencode-mimo, codex]`.
94
+ - **Section headers:** each instance gets its OWN top-level section, headed with the base
95
+ adapter's display name plus the instance name in parentheses:
96
+ `## OpenCode Review (opencode-deepseek)`. Same-cli instances are never collapsed.
97
+ - **Shared-adapter caveat:** when ≥2 invoked instances share the same base `cli`, print a
98
+ one-line caveat immediately after the frontmatter (before the first section), e.g.:
99
+ `> Note: opencode-deepseek and opencode-mimo share the opencode adapter; their consensus is cross-model, not cross-tool.`
@@ -61,7 +61,7 @@ fi
61
61
 
62
62
  When `--only` is set, also set `FROM_PHASE` to the same value so existing filter logic applies.
63
63
 
64
- When `--interactive` is set, discuss runs inline with questions. On Codex, where a backgrounded agent can still spawn subagents, plan and execute are dispatched as background agents — keeping the main context lean (only discuss conversations accumulate) and enabling overlap. On every other runtime (Claude Code and all other non-Codex runtimes), backgrounded agents cannot reliably nest subagents, so plan and execute run inline to preserve worktree isolation and independent verification, and phases run sequentially with their work accumulating in the main context. Either way, user input is preserved on all design decisions.
64
+ When `--interactive` is set, discuss runs inline with questions. When `dispatch-should-flatten` returns `false` (e.g. codex, cursor — runtimes where a backgrounded agent can still spawn subagents), plan and execute are dispatched as background agents — keeping the main context lean (only discuss conversations accumulate) and enabling overlap. When `dispatch-should-flatten` returns `true` (e.g. claude and other runtimes where backgrounded agents cannot reliably nest subagents), plan and execute run inline to preserve worktree isolation and independent verification, and phases run sequentially with their work accumulating in the main context. Either way, user input is preserved on all design decisions.
65
65
 
66
66
  When `PLAN_STRATEGY=converge`, the planning step MUST invoke the plan-review convergence workflow instead of `gsd-plan-phase`. `--cross-ai` is an alias for `--converge`. Forward `CONVERGENCE_ARGS` exactly as parsed so reviewer flags and `--max-cycles N` retain the same meaning as they have on `/gsd:plan-review-convergence`.
67
67
 
@@ -358,13 +358,13 @@ UI_SPEC_FILE=$(ls "${PHASE_DIR}"/*-UI-SPEC.md 2>/dev/null | head -1)
358
358
 
359
359
  **3b. Plan**
360
360
 
361
- **If `INTERACTIVE` is set:** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background. Resolve the runtime first:
361
+ **If `INTERACTIVE` is set:** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`. Resolve first:
362
362
 
363
363
  ```bash
364
- RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
364
+ FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
365
365
  ```
366
366
 
367
- - **If `RUNTIME` is `codex`:** Dispatch plan as a background agent to keep the main context lean. While plan runs, the workflow can immediately start discussing the next phase (see step 4).
367
+ - **If `FLATTEN` is `false`:** Dispatch plan as a background agent to keep the main context lean. While plan runs, the workflow can immediately start discussing the next phase (see step 4).
368
368
 
369
369
  - If `PLAN_STRATEGY=converge`, print: `◆ Spawning background plan-convergence loop for phase ${PHASE_NUM}... (runs in a subagent — no output until it returns, ~1–5 min; expected, not a freeze)`
370
370
 
@@ -388,7 +388,7 @@ RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null ||
388
388
 
389
389
  Store the agent task_id. After discuss for the next phase completes (or if no next phase), wait for the plan agent to finish before proceeding to execute.
390
390
 
391
- - **Otherwise (Claude Code or any other non-Codex runtime):** Run plan **inline** (do NOT background) so the plan-checker runs. The next phase's discuss does not overlap planning here — correctness over overlap.
391
+ - **Otherwise (`FLATTEN` is `true` — run inline):** Run plan **inline** (do NOT background) so the plan-checker runs. The next phase's discuss does not overlap planning here — correctness over overlap.
392
392
 
393
393
  - If `PLAN_STRATEGY=converge`:
394
394
 
@@ -420,13 +420,13 @@ Verify plan produced output — re-run `init phase-op` and check `has_plans`. If
420
420
 
421
421
  **3c. Execute**
422
422
 
423
- **If `INTERACTIVE` is set:** Wait for the plan agent to complete (if not already) and verify plans exist. Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background. Resolve the runtime first:
423
+ **If `INTERACTIVE` is set:** Wait for the plan agent to complete (if not already) and verify plans exist. Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`. Resolve first:
424
424
 
425
425
  ```bash
426
- RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
426
+ FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
427
427
  ```
428
428
 
429
- - **If `RUNTIME` is `codex`:** Dispatch execute as a background agent:
429
+ - **If `FLATTEN` is `false`:** Dispatch execute as a background agent:
430
430
 
431
431
  ```
432
432
  Agent(
@@ -438,7 +438,7 @@ Agent(
438
438
 
439
439
  Store the agent task_id. The workflow can now start discussing the next phase while this phase executes in the background. Before starting post-execution routing for this phase, wait for the execute agent to complete.
440
440
 
441
- - **Otherwise (Claude Code or any other non-Codex runtime):** Run execute **inline** (do NOT background) so worktree isolation and verification run:
441
+ - **Otherwise (`FLATTEN` is `true` — run inline):** Run execute **inline** (do NOT background) so worktree isolation and verification run:
442
442
 
443
443
  ```
444
444
  Skill(skill="gsd-execute-phase", args="${PHASE_NUM} --no-transition")
@@ -1,6 +1,6 @@
1
1
  <purpose>
2
2
 
3
- Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and runs plan/execute inline (backgrounded only on Codex), and loops back to the dashboard after each action. Enables parallel phase work from one terminal.
3
+ Interactive command center for managing a milestone from a single terminal. Shows a dashboard of all phases with visual status, dispatches discuss inline and runs plan/execute inline (backgrounded when dispatch-should-flatten returns false), and loops back to the dashboard after each action. Enables parallel phase work from one terminal.
4
4
 
5
5
  </purpose>
6
6
 
@@ -45,7 +45,7 @@ Display startup banner:
45
45
  {milestone_version} — {milestone_name}
46
46
  {phase_count} phases · {completed_count} complete
47
47
 
48
- ✓ Discuss → inline ◆ Plan/Execute → inline (background on Codex)
48
+ ✓ Discuss → inline ◆ Plan/Execute → inline (background when FLATTEN=false)
49
49
  Dashboard auto-refreshes when background work is active.
50
50
  ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
51
51
  ```
@@ -222,10 +222,10 @@ Go to exit step.
222
222
 
223
223
  ### Compound Action (background + inline)
224
224
 
225
- When the user selects a compound option, behavior depends on the runtime — the Plan Phase N / Execute Phase N handlers below resolve it via `gsd_run query config-get runtime`:
225
+ When the user selects a compound option, behavior depends on whether the runtime supports background dispatch of nesting-capable orchestrators — the Plan Phase N / Execute Phase N handlers below resolve it via `gsd_run query dispatch-should-flatten` (#1708):
226
226
 
227
- - **On Codex:** **Spawn all background agents first** (plan/execute) — dispatch them in parallel using the Plan Phase N / Execute Phase N handlers below — then run verification actions, then run the inline discuss; the background agents continue while you verify/discuss.
228
- - **On Claude Code or any other non-Codex runtime:** run the chosen plan/execute step(s) **inline** via their handlers below (in order), then run verification actions, then run the inline discuss. There is no overlap.
227
+ - **If `FLATTEN` is `false` (the host can background a nesting-capable orchestrator — e.g. codex, cursor):** **Spawn all background agents first** (plan/execute) — dispatch them in parallel using the Plan Phase N / Execute Phase N handlers below — then run verification actions, then run the inline discuss; the background agents continue while you verify/discuss.
228
+ - **Otherwise (`FLATTEN` is `true` — run inline):** run the chosen plan/execute step(s) **inline** via their handlers below (in order), then run verification actions, then run the inline discuss. There is no overlap.
229
229
 
230
230
  Inline verification:
231
231
 
@@ -254,13 +254,13 @@ After discuss completes, loop back to dashboard step.
254
254
 
255
255
  ### Plan Phase N
256
256
 
257
- Planning runs autonomously. **First resolve the runtime.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background.
257
+ Planning runs autonomously. **First resolve whether background dispatch is safe.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`.
258
258
 
259
259
  ```bash
260
- RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
260
+ FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
261
261
  ```
262
262
 
263
- **If `RUNTIME` is `codex`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
263
+ **If `FLATTEN` is `false`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
264
264
 
265
265
  ```
266
266
  Agent(
@@ -282,7 +282,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
282
282
  )
283
283
  ```
284
284
 
285
- > **ORCHESTRATOR RULE — CODEX RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any planning work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume planning-related work when the subagent result is available.
285
+ > **ORCHESTRATOR RULE — BACKGROUND DISPATCH**: After calling Agent() above with `run_in_background=true`, do NOT do any planning work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume planning-related work when the subagent result is available.
286
286
 
287
287
  Display:
288
288
 
@@ -292,7 +292,7 @@ Display:
292
292
 
293
293
  Loop back to dashboard step.
294
294
 
295
- **Otherwise (Claude Code or any other non-Codex runtime):** Run plan inline so the plan-checker and quality gates actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
295
+ **Otherwise (`FLATTEN` is `true` — run inline):** Run plan inline so the plan-checker and quality gates actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
296
296
 
297
297
  ```
298
298
  Skill(skill="gsd-plan-phase", args="{N} --auto {manager_flags.plan}")
@@ -308,13 +308,13 @@ Then loop back to dashboard step.
308
308
 
309
309
  ### Execute Phase N
310
310
 
311
- Execution runs autonomously. **First resolve the runtime.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). Among supported runtimes only **Codex** (`spawn_agent`) can do this; Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except Codex, which is dispatched in the background.
311
+ Execution runs autonomously. **First resolve whether background dispatch is safe.** Background dispatch is only safe on a runtime where a backgrounded agent can still nest the pipeline's subagents (plan-checker / worktree executors / verifier). This is determined from the documentation-sourced dispatch capability in the registry (#1708); Claude Code's backgrounded agents have no `Agent`/`Task` tool, and every other runtime either prohibits nested subagents or disables them by default. So run **inline** everywhere except where `dispatch-should-flatten` returns `false`.
312
312
 
313
313
  ```bash
314
- RUNTIME=$(gsd_run query config-get runtime --default claude --raw 2>/dev/null || echo "claude")
314
+ FLATTEN=$(gsd_run query dispatch-should-flatten --raw 2>/dev/null || echo "true")
315
315
  ```
316
316
 
317
- **If `RUNTIME` is `codex`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
317
+ **If `FLATTEN` is `false`:** Spawn a background agent that delegates to the Skill pipeline with any configured flags:
318
318
 
319
319
  ```
320
320
  Agent(
@@ -336,7 +336,7 @@ Important: You are running in the background. Do NOT use AskUserQuestion — mak
336
336
  )
337
337
  ```
338
338
 
339
- > **ORCHESTRATOR RULE — CODEX RUNTIME**: After calling Agent() above with `run_in_background=true`, do NOT do any execution work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume execution-related work when the subagent result is available.
339
+ > **ORCHESTRATOR RULE — BACKGROUND DISPATCH**: After calling Agent() above with `run_in_background=true`, do NOT do any execution work for this phase independently. Return to the dashboard immediately and wait for the background agent to report back. Only resume execution-related work when the subagent result is available.
340
340
 
341
341
  Display:
342
342
 
@@ -346,7 +346,7 @@ Display:
346
346
 
347
347
  Loop back to dashboard step.
348
348
 
349
- **Otherwise (Claude Code or any other non-Codex runtime):** Run execute inline so worktree isolation and the verifier actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
349
+ **Otherwise (`FLATTEN` is `true` — run inline):** Run execute inline so worktree isolation and the verifier actually run — do NOT wrap it in `Agent(run_in_background=true, …)`:
350
350
 
351
351
  ```
352
352
  Skill(skill="gsd-execute-phase", args="{N} {manager_flags.execute}")
@@ -912,7 +912,7 @@ Output consumed by /gsd:execute-phase. Plans need:
912
912
  - Tasks in XML format with read_first and acceptance_criteria fields (MANDATORY on every task)
913
913
  - Verification criteria
914
914
  - must_haves for goal-backward verification
915
- - If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths`, and every `backstop` edge into `must_haves.truths` as a non-inferable check (note it needs a held-out/property-based test). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them.
915
+ - If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths` as a plain string, and every `backstop` edge **as a structured flat-scalar marker** — an object item `{ statement: <the check>, verification: backstop }`, NOT a prose note (the verifier branches deterministically on the `verification: backstop` field; a parenthetical is unparseable — the #1110 fragility). Use a flat scalar `verification:` continuation key, never a nested object (ADR-550 #1278). At verify time a `backstop` truth the verifier cannot confirm with explicit evidence abstains → `human_needed` (reason `insufficient_spec`), never a silent pass (#1154; see `references/honest-verifier.md`). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them.
916
916
  - If the SPEC has a `## Prohibitions` section, lift every resolved prohibition into the `must_haves.prohibitions:` sibling block (NOT `truths` — ADR-550 D3) carrying `statement` + `status` + `verification`; unresolved prohibitions are explicit assumptions — surface them in the plan, do not silently drop them. A prohibition is a must-NOT (negative) check that belongs in its own `must_haves.prohibitions` block. Never place a must-NOT under `must_haves.truths` — that block keeps positive-observable semantics only.
917
917
  - **"Artifacts this phase produces" section (MANDATORY)** — list every symbol this phase creates: decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths. The plan-review-convergence source-grounding pass reads this section to exclude newly-created symbols from drift verification; omitting it causes new symbols to be flagged for acknowledgement.
918
918
  </downstream_consumer>
@@ -66,6 +66,11 @@ Reviewer-selection precedence:
66
66
  - Known-but-undetected slugs emit an info note and are ignored
67
67
  - If all configured reviewers are unavailable, fail with an actionable message
68
68
 
69
+ **Reviewer instances (#1517, optional):** if `review.reviewer_instances` is configured,
70
+ instance names in `review.default_reviewers` run as independent identities. Resolution rules
71
+ are in `gsd-core/references/reviewer-instances.md` — load it lazily only when instances are
72
+ configured. Unconfigured → default path unchanged.
73
+
69
74
  If no CLIs are available:
70
75
  ```
71
76
  No external AI CLIs found. Install at least one:
@@ -243,6 +248,10 @@ else
243
248
  fi
244
249
  ```
245
250
 
251
+ **Reviewer instances (#1517, optional):** when instances are configured, each selected
252
+ instance invokes its base `cli` with its own `model`/`agent` (opaque argv, never
253
+ shell-interpolated). Exact invocation in `gsd-core/references/reviewer-instances.md`.
254
+
246
255
  For each selected CLI, invoke in sequence (not parallel — avoid rate limits):
247
256
 
248
257
  **Gemini:**
@@ -634,6 +643,11 @@ Combine all review responses into `{phase_dir}/{padded_phase}-REVIEWS.md`:
634
643
 
635
644
  After all reviewers complete, collect trim metadata files written during the run. For each reviewer that was trimmed (i.e. a `.metadata.json` file exists and `hardFailed` or `omitted` is non-empty, or `projectMdShrunk` is true, or `planTruncationPct > 0`), include a `trimmed_reviewers` block in the frontmatter. Omit the key entirely if no reviewer was trimmed.
636
645
 
646
+ **Reviewer instances (#1517, optional):** when instances ran, frontmatter records their
647
+ names, each gets its own `## <Adapter> Review (<instance>)` section, and ≥2 same-cli
648
+ instances print a one-line shared-adapter caveat. Format in
649
+ `gsd-core/references/reviewer-instances.md`.
650
+
637
651
  ```markdown
638
652
  ---
639
653
  phase: {N}
@@ -684,6 +698,18 @@ trimmed_reviewers: # only present if at least one reviewer was trimmed
684
698
 
685
699
  ---
686
700
 
701
+ ## OpenCode Review (opencode-deepseek)
702
+
703
+ {opencode-deepseek instance review content — only present when this instance was selected}
704
+
705
+ ---
706
+
707
+ ## OpenCode Review (opencode-mimo)
708
+
709
+ {opencode-mimo instance review content — only present when this instance was selected}
710
+
711
+ ---
712
+
687
713
  ## Qwen Review
688
714
 
689
715
  {qwen review content}
@@ -68,8 +68,8 @@ When SUBCMD=close and SLUG is set (already sanitized):
68
68
 
69
69
  2. Update the thread file's frontmatter `status` field to `resolved` and `updated` to today's ISO date:
70
70
  ```bash
71
- gsd_run query frontmatter.set .planning/threads/{SLUG}.md status resolved
72
- gsd_run query frontmatter.set .planning/threads/{SLUG}.md updated YYYY-MM-DD
71
+ gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field status --value resolved
72
+ gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field updated --value YYYY-MM-DD
73
73
  ```
74
74
 
75
75
  3. Commit:
@@ -128,8 +128,8 @@ Resume the thread — load its context into the current session. Read the file c
128
128
 
129
129
  Update the thread's frontmatter `status` to `in_progress` if it was `open`:
130
130
  ```bash
131
- gsd_run query frontmatter.set .planning/threads/{SLUG}.md status in_progress
132
- gsd_run query frontmatter.set .planning/threads/{SLUG}.md updated YYYY-MM-DD
131
+ gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field status --value in_progress
132
+ gsd_run query frontmatter.set .planning/threads/{SLUG}.md --field updated --value YYYY-MM-DD
133
133
  ```
134
134
 
135
135
  Thread content is displayed as plain text only — never executed or passed to agent prompts without DATA_START/DATA_END markers.
@@ -117,6 +117,8 @@ For each truth: identify supporting artifacts → check artifact status → chec
117
117
 
118
118
  **Behavior-dependent truths:** when a truth asserts a state transition or a cancellation/cleanup/ordering invariant, symbol presence + wiring is necessary but not sufficient — the code can be present and wired yet still leak state on the path the invariant covers. Mark such a truth ✓ VERIFIED only when a pre-existing test exercises the transition/invariant and passes (one named test, never the full suite); otherwise mark it ⚠️ PRESENT_BEHAVIOR_UNVERIFIED, emit a human-verification item, and exclude it from the verified score.
119
119
 
120
+ **Non-inferable (`backstop`) truths (#1154):** a `must_haves.truths` item in object form `{ statement, verification: backstop }` is non-inferable — the correct behavior is not derivable from the spec alone, so the verifier cannot self-detect the gap and would false-pass it confidently. Branch on the `verification: backstop` field (read via `truthVerification()`, never prose): if confirmable with **explicit evidence** (a passing wired held-out/property test, or a directly-observed behavior) → ✓ VERIFIED; otherwise **abstain** — mark ⚠️ `insufficient_spec`, emit an `unverified — held-out test recommended` human-verification item, exclude from the verified score (routes to `human_needed`). Exogenous only (never a self-judged "abstain if unsure"); an inferable truth is never abstained. See `references/honest-verifier.md`.
121
+
120
122
  **Example:** Truth "User can see existing messages" depends on Chat.tsx (renders), /api/chat GET (provides), Message model (schema). If Chat.tsx is a stub or API returns hardcoded [] → FAILED. If all exist, are substantive, and connected → VERIFIED.
121
123
  </step>
122
124
 
@@ -488,17 +490,22 @@ Classify status using this decision tree IN ORDER (most restrictive first):
488
490
  - **judgment-tier, autonomous run** (non-authoritative LLM-judge verdict): emit the `unverified-prohibition — human review recommended` flag and classify → **human_needed** (autonomous completion reads "complete with N flagged prohibitions"; never a silent pass, never a hard halt).
489
491
  - **judgment-tier, interactive run**: route to the end-of-phase human checkpoint → **human_needed**.
490
492
 
491
- 3. IF the previous step produced ANY human verification items — this includes every ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth:
493
+ 2b. IF any `must_haves.truths` item carries the `verification: backstop` marker (#1154 — the verify-time truth-axis mirror of ADR-550 D4) AND the verifier cannot confirm it with **explicit evidence** (a wired held-out/property-based test that PASSES, or a directly-observed behavior — i.e. `dispositionForUnverifiableTruth()` returns `status: 'unverified'`, `flagged: true`, `reason: 'insufficient_spec'`):
494
+ - **abstain → human_needed**, NEVER `passed` and never silently graded green. Emit a prominent `unverified — held-out test recommended` flag carrying the distinguishable `reason: insufficient_spec` (so it is not conflated with ordinary manual-UAT `human_needed`).
495
+ - *Autonomous run:* record it and continue — completion reads "complete with N unverified non-inferable checks"; never a hard halt of an AFK run. *Interactive run:* route to the end-of-phase human checkpoint.
496
+ - **Exogenous only:** abstention fires SOLELY on the `backstop` tag, never a self-judged "abstain if unsure" (N17). An **inferable** truth is NEVER abstained (over-abstention guard); a `backstop` truth WITH a passing wired held-out test reaches **passed**. Reliable on capable tiers (`sonnet`+); the budget `haiku` tier degrades — see `references/honest-verifier.md`.
497
+
498
+ 3. IF the previous step produced ANY human verification items — this includes every ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth and every abstained `insufficient_spec` backstop truth:
492
499
  → **human_needed** (even if all other truths VERIFIED)
493
500
 
494
- 4. IF all checks pass AND no human verification items AND no flagged prohibitions:
501
+ 4. IF all checks pass AND no human verification items AND no flagged prohibitions AND no abstained (`insufficient_spec`) truths:
495
502
  → **passed**
496
503
 
497
- **passed is ONLY valid when no human verification items AND no flagged prohibitions exist.** A prohibition (must-NOT) can never be silently absorbed into a `passed` verdict — that is the core failure mode ADR-550 D4 forbids.
504
+ **passed is ONLY valid when no human verification items, no flagged prohibitions, AND no abstained `insufficient_spec` truths exist.** Neither a prohibition (must-NOT) nor an unconfirmable non-inferable truth can ever be silently absorbed into a `passed` verdict — that is the core failure mode ADR-550 D4 forbids (now closed on both the prohibition and truth axes).
498
505
 
499
506
  A ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth is never FAILED and never VERIFIED: it does not trigger gaps_found (the code is present and wired) and is not counted as verified (its runtime behavior was not exercised). It routes through the existing human_needed sink — no new overall status.
500
507
 
501
- **Score:** `verified_truths / total_truths` — `verified_truths` counts ✓ VERIFIED truths plus PASSED (override) truths; ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truths are the only ones excluded, reported separately as the `behavior_unverified` count. A headline N/N therefore certifies behavioral evidence for every behavior-dependent truth, not merely symbol presence.
508
+ **Score:** `verified_truths / total_truths` — `verified_truths` counts ✓ VERIFIED truths plus PASSED (override) truths; excluded are ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truths (the `behavior_unverified` count) and abstained ⚠️ `insufficient_spec` backstop truths (#1154) — both are not ✓ VERIFIED and both route to `human_needed`. A headline N/N therefore certifies behavioral evidence for every behavior-dependent truth and explicit evidence for every non-inferable one, not merely symbol presence.
502
509
  </step>
503
510
 
504
511
  <step name="filter_deferred_items">
@@ -45,7 +45,13 @@ process.stdin.on("end", () => {
45
45
  });
46
46
  ' 2>/dev/null || printf '\n')
47
47
  TOOL_NAME=$(printf '%s\n' "$TOOL_INFO" | sed -n '1p')
48
- COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2p')
48
+ # Capture the FULL command (line 2 through EOF). Agent runtimes routinely emit
49
+ # HEAD-advancing commits as multi-line scripts (`cd /path` then `git add` then
50
+ # `git commit …`); reading only line 2 (`sed -n '2p'`) missed a `git commit`
51
+ # that was not on the first command line and silently no-op'd the rebuild
52
+ # (#1772). Line 2..EOF preserves embedded newlines; the `case` glob below
53
+ # matches the substring anywhere in the multi-line string.
54
+ COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2,$p')
49
55
 
50
56
  [ "$TOOL_NAME" = "Bash" ] || exit 0
51
57
 
@@ -45,7 +45,13 @@ process.stdin.on("end", () => {
45
45
  });
46
46
  ' 2>/dev/null || printf '\n')
47
47
  TOOL_NAME=$(printf '%s\n' "$TOOL_INFO" | sed -n '1p')
48
- COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2p')
48
+ # Capture the FULL command (line 2 through EOF). Agent runtimes routinely emit
49
+ # HEAD-advancing commits as multi-line scripts (`cd /path` then `git add` then
50
+ # `git commit …`); reading only line 2 (`sed -n '2p'`) missed a `git commit`
51
+ # that was not on the first command line and silently no-op'd the rebuild
52
+ # (#1772). Line 2..EOF preserves embedded newlines; the `case` glob below
53
+ # matches the substring anywhere in the multi-line string.
54
+ COMMAND=$(printf '%s\n' "$TOOL_INFO" | sed -n '2,$p')
49
55
 
50
56
  [ "$TOOL_NAME" = "Bash" ] || exit 0
51
57