@opengsd/gsd-core 1.10.0 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (328) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.claude-plugin/plugin.json +1 -1
  3. package/agents/gsd-debug-session-manager.md +11 -0
  4. package/agents/gsd-doc-synthesizer.md +2 -4
  5. package/agents/gsd-executor.md +5 -5
  6. package/agents/gsd-mempalace-curator.md +5 -2
  7. package/agents/gsd-phase-researcher.md +20 -1
  8. package/agents/gsd-plan-checker.md +37 -0
  9. package/agents/gsd-planner.md +44 -46
  10. package/agents/gsd-user-profiler.md +3 -0
  11. package/agents/gsd-verifier.md +12 -3
  12. package/bin/install.js +841 -971
  13. package/bin/lib/ui-safety-gate.cjs +2 -0
  14. package/commands/gsd/code-review.md +1 -1
  15. package/commands/gsd/execute-phase.md +1 -1
  16. package/commands/gsd/map-codebase.md +1 -1
  17. package/commands/gsd/mempalace-capture.md +1 -1
  18. package/commands/gsd/mempalace-recall.md +1 -1
  19. package/commands/gsd/new-milestone.md +1 -1
  20. package/commands/gsd/quick.md +1 -1
  21. package/commands/gsd/review-backlog.md +2 -1
  22. package/commands/gsd/verify-work.md +1 -1
  23. package/gsd-core/bin/gsd-tools.cjs +469 -88
  24. package/gsd-core/bin/lib/active-workstream-store.cjs +138 -22
  25. package/gsd-core/bin/lib/agent-install-check.cjs +230 -32
  26. package/gsd-core/bin/lib/api-coverage.cjs +3 -5
  27. package/gsd-core/bin/lib/artifacts.cjs +3 -0
  28. package/gsd-core/bin/lib/assumption-delta.cjs +2 -4
  29. package/gsd-core/bin/lib/audit-command-router.cjs +9 -2
  30. package/gsd-core/bin/lib/audit.cjs +876 -240
  31. package/gsd-core/bin/lib/broken-windows.cjs +1 -1
  32. package/gsd-core/bin/lib/capability-consent.cjs +149 -15
  33. package/gsd-core/bin/lib/capability-lifecycle.cjs +45 -0
  34. package/gsd-core/bin/lib/capability-registry.cjs +575 -101
  35. package/gsd-core/bin/lib/capability-source.cjs +92 -0
  36. package/gsd-core/bin/lib/capability-trust.cjs +444 -25
  37. package/gsd-core/bin/lib/capability-validator.cjs +495 -22
  38. package/gsd-core/bin/lib/capability-writer.cjs +3 -2
  39. package/gsd-core/bin/lib/check-command-router.cjs +71 -37
  40. package/gsd-core/bin/lib/claude-orchestration.cjs +56 -3
  41. package/gsd-core/bin/lib/codex-agent-toml.cjs +329 -0
  42. package/gsd-core/bin/lib/command-aliases.cjs +22 -0
  43. package/gsd-core/bin/lib/command-roster.cjs +44 -1
  44. package/gsd-core/bin/lib/commands.cjs +651 -86
  45. package/gsd-core/bin/lib/commonjs-marker.cjs +12 -6
  46. package/gsd-core/bin/lib/complexity-trigger.cjs +1172 -0
  47. package/gsd-core/bin/lib/config-loader.cjs +75 -0
  48. package/gsd-core/bin/lib/config.cjs +10 -1
  49. package/gsd-core/bin/lib/core-utils.cjs +127 -29
  50. package/gsd-core/bin/lib/decisions.cjs +23 -0
  51. package/gsd-core/bin/lib/fallow-runner.cjs +20 -44
  52. package/gsd-core/bin/lib/frontmatter.cjs +155 -20
  53. package/gsd-core/bin/lib/gap-checker.cjs +68 -7
  54. package/gsd-core/bin/lib/git-base-branch.cjs +102 -0
  55. package/gsd-core/bin/lib/gsd2-import.cjs +10 -1
  56. package/gsd-core/bin/lib/health-diagnostic-rules/agent-install.cjs +101 -0
  57. package/gsd-core/bin/lib/health-diagnostic-rules/config-validation.cjs +348 -0
  58. package/gsd-core/bin/lib/health-diagnostic-rules/consistency.cjs +145 -0
  59. package/gsd-core/bin/lib/health-diagnostic-rules/install-surface-shadowing.cjs +98 -0
  60. package/gsd-core/bin/lib/health-diagnostic-rules/milestone-archive-hygiene.cjs +100 -0
  61. package/gsd-core/bin/lib/health-diagnostic-rules/phase-structure.cjs +222 -0
  62. package/gsd-core/bin/lib/health-diagnostic-rules/roadmap-disk-consistency.cjs +265 -0
  63. package/gsd-core/bin/lib/health-diagnostic-rules/root-existence.cjs +161 -0
  64. package/gsd-core/bin/lib/health-diagnostic-rules/state-consistency.cjs +303 -0
  65. package/gsd-core/bin/lib/health-diagnostic-rules/worktree-health.cjs +173 -0
  66. package/gsd-core/bin/lib/health-diagnostic-types.cjs +68 -0
  67. package/gsd-core/bin/lib/health-diagnostic.cjs +431 -0
  68. package/gsd-core/bin/lib/host-runtime-detection.cjs +134 -0
  69. package/gsd-core/bin/lib/init.cjs +321 -129
  70. package/gsd-core/bin/lib/install-effort-resolver.cjs +73 -30
  71. package/gsd-core/bin/lib/install-engine.cjs +745 -258
  72. package/gsd-core/bin/lib/install-fs-adapter.cjs +262 -0
  73. package/gsd-core/bin/lib/install-model-override-resolver.cjs +203 -0
  74. package/gsd-core/bin/lib/install-profiles.cjs +134 -57
  75. package/gsd-core/bin/lib/install-scope.cjs +270 -0
  76. package/gsd-core/bin/lib/install-shadow-report.cjs +385 -0
  77. package/gsd-core/bin/lib/installed-surface-resolver.cjs +381 -0
  78. package/gsd-core/bin/lib/installer-migrations.cjs +138 -31
  79. package/gsd-core/bin/lib/io.cjs +10 -0
  80. package/gsd-core/bin/lib/markdown-sectionizer.cjs +2 -1
  81. package/gsd-core/bin/lib/markdown-table.cjs +133 -20
  82. package/gsd-core/bin/lib/milestone-lock.cjs +248 -0
  83. package/gsd-core/bin/lib/milestone.cjs +754 -70
  84. package/gsd-core/bin/lib/model-catalog.cjs +59 -1
  85. package/gsd-core/bin/lib/model-resolver.cjs +183 -40
  86. package/gsd-core/bin/lib/normalize-test-command.cjs +1 -1
  87. package/gsd-core/bin/lib/pattern.cjs +122 -0
  88. package/gsd-core/bin/lib/phase-estimation.cjs +1 -1
  89. package/gsd-core/bin/lib/phase-id.cjs +444 -36
  90. package/gsd-core/bin/lib/phase-lifecycle.cjs +28 -3
  91. package/gsd-core/bin/lib/phase-locator.cjs +125 -18
  92. package/gsd-core/bin/lib/phase.cjs +646 -143
  93. package/gsd-core/bin/lib/plan-dependency-graph.cjs +72 -1
  94. package/gsd-core/bin/lib/plan-drift-guard.cjs +120 -0
  95. package/gsd-core/bin/lib/plan-scan.cjs +86 -2
  96. package/gsd-core/bin/lib/planning-scope.cjs +31 -0
  97. package/gsd-core/bin/lib/planning-snapshot.cjs +890 -0
  98. package/gsd-core/bin/lib/planning-workspace.cjs +56 -6
  99. package/gsd-core/bin/lib/probe-core.cjs +1 -1
  100. package/gsd-core/bin/lib/profile-output.cjs +1 -1
  101. package/gsd-core/bin/lib/refactor-trigger-command-router.cjs +740 -0
  102. package/gsd-core/bin/lib/retired-artifact-cleanup.cjs +11 -6
  103. package/gsd-core/bin/lib/review-lane-descriptor.cjs +13 -4
  104. package/gsd-core/bin/lib/review-lane-invocation.cjs +30 -0
  105. package/gsd-core/bin/lib/review-lane-runner.cjs +421 -66
  106. package/gsd-core/bin/lib/review-reviewer-selection.cjs +13 -18
  107. package/gsd-core/bin/lib/roadmap-command-router.cjs +34 -0
  108. package/gsd-core/bin/lib/roadmap-parser.cjs +943 -184
  109. package/gsd-core/bin/lib/roadmap-upgrade.cjs +37 -10
  110. package/gsd-core/bin/lib/roadmap.cjs +385 -94
  111. package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +608 -46
  112. package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +14 -2
  113. package/gsd-core/bin/lib/runtime-artifact-layout.cjs +426 -55
  114. package/gsd-core/bin/lib/runtime-config-adapter-registry.cjs +3 -2
  115. package/gsd-core/bin/lib/runtime-homes.cjs +69 -3
  116. package/gsd-core/bin/lib/runtime-hooks-surface.cjs +115 -3
  117. package/gsd-core/bin/lib/runtime-name-policy.cjs +3 -1
  118. package/gsd-core/bin/lib/runtime-slash.cjs +27 -9
  119. package/gsd-core/bin/lib/security.cjs +104 -5
  120. package/gsd-core/bin/lib/shell-command-projection.cjs +275 -3
  121. package/gsd-core/bin/lib/smart-entry.cjs +142 -22
  122. package/gsd-core/bin/lib/state-command-router.cjs +5 -1
  123. package/gsd-core/bin/lib/state-document.cjs +152 -8
  124. package/gsd-core/bin/lib/state-transition.cjs +371 -117
  125. package/gsd-core/bin/lib/state.cjs +1794 -357
  126. package/gsd-core/bin/lib/surface.cjs +23 -9
  127. package/gsd-core/bin/lib/text-lines.cjs +80 -0
  128. package/gsd-core/bin/lib/token-scanner.cjs +76 -0
  129. package/gsd-core/bin/lib/uat-predicate.cjs +9 -3
  130. package/gsd-core/bin/lib/uat.cjs +399 -56
  131. package/gsd-core/bin/lib/ui-frontend-evidence.cjs +157 -0
  132. package/gsd-core/bin/lib/ui-safety-gate.cjs +14 -5
  133. package/gsd-core/bin/lib/unusable-input.cjs +24 -0
  134. package/gsd-core/bin/lib/update-context.cjs +8 -2
  135. package/gsd-core/bin/lib/user-artifact-staging.cjs +705 -0
  136. package/gsd-core/bin/lib/validate.cjs +20 -6
  137. package/gsd-core/bin/lib/vendor/README.md +37 -0
  138. package/gsd-core/bin/lib/vendor/re2js.cjs +6480 -0
  139. package/gsd-core/bin/lib/vendor/re2js.d.cts +938 -0
  140. package/gsd-core/bin/lib/verification-command-router.cjs +2 -1
  141. package/gsd-core/bin/lib/verification.cjs +258 -8
  142. package/gsd-core/bin/lib/verify.cjs +368 -888
  143. package/gsd-core/bin/lib/workstream-inventory-builder.cjs +53 -32
  144. package/gsd-core/bin/lib/workstream-inventory.cjs +63 -10
  145. package/gsd-core/bin/lib/workstream.cjs +2 -2
  146. package/gsd-core/bin/lib/worktree-safety.cjs +176 -9
  147. package/gsd-core/bin/shared/config-defaults.manifest.json +1 -0
  148. package/gsd-core/bin/shared/config-schema.manifest.json +7 -1
  149. package/gsd-core/references/agent-contracts.md +43 -26
  150. package/gsd-core/references/checkpoints.md +2 -2
  151. package/gsd-core/references/context-budget.md +1 -1
  152. package/gsd-core/references/dispatch-isolation-gate.md +138 -0
  153. package/gsd-core/references/doc-conflict-engine.md +1 -1
  154. package/gsd-core/references/execute-mvp-tdd.md +3 -3
  155. package/gsd-core/references/execute-phase-between-wave-reset.md +6 -2
  156. package/gsd-core/references/execute-phase-context-guard.md +1 -1
  157. package/gsd-core/references/execute-phase-response-language.md +1 -1
  158. package/gsd-core/references/execute-phase-wave-guard.md +6 -2
  159. package/gsd-core/references/gate-prompts.md +1 -1
  160. package/gsd-core/references/git-planning-commit.md +2 -1
  161. package/gsd-core/references/loop-hook-dispatch.md +39 -2
  162. package/gsd-core/references/model-profiles.md +12 -4
  163. package/gsd-core/references/mvp-concepts.md +9 -9
  164. package/gsd-core/references/planner-guidance.md +3 -9
  165. package/gsd-core/references/planner-preconditions.md +1 -1
  166. package/gsd-core/references/planner-reviews.md +1 -1
  167. package/gsd-core/references/planning-config.md +8 -6
  168. package/gsd-core/references/revision-loop.md +1 -1
  169. package/gsd-core/references/specless-probe-fallback.md +1 -1
  170. package/gsd-core/references/universal-anti-patterns.md +3 -3
  171. package/gsd-core/references/verifier-phase-gates.md +192 -0
  172. package/gsd-core/references/verify-mvp-mode.md +1 -1
  173. package/gsd-core/references/workstream-flag.md +22 -6
  174. package/gsd-core/templates/discussion-log.md +1 -1
  175. package/gsd-core/templates/phase-prompt.md +2 -4
  176. package/gsd-core/templates/state.md +4 -4
  177. package/gsd-core/templates/verification-report.md +9 -1
  178. package/gsd-core/workflows/ai-integration-phase.md +9 -11
  179. package/gsd-core/workflows/autonomous.md +1 -1
  180. package/gsd-core/workflows/cleanup.md +62 -3
  181. package/gsd-core/workflows/code-review/steps/structural-pre-pass.md +13 -3
  182. package/gsd-core/workflows/code-review-fix.md +37 -10
  183. package/gsd-core/workflows/code-review.md +38 -12
  184. package/gsd-core/workflows/complete-milestone.md +141 -18
  185. package/gsd-core/workflows/debug.md +7 -5
  186. package/gsd-core/workflows/diagnose-issues.md +35 -9
  187. package/gsd-core/workflows/discuss-phase/modes/chain.md +2 -1
  188. package/gsd-core/workflows/discuss-phase/modes/default.md +1 -1
  189. package/gsd-core/workflows/discuss-phase-assumptions.md +2 -1
  190. package/gsd-core/workflows/edit-phase.md +26 -1
  191. package/gsd-core/workflows/eval-review.md +3 -5
  192. package/gsd-core/workflows/execute-phase/steps/executor-isolation-dispatch.md +31 -6
  193. package/gsd-core/workflows/execute-phase/steps/per-plan-executor-routing.md +77 -0
  194. package/gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md +2 -0
  195. package/gsd-core/workflows/execute-phase.md +38 -50
  196. package/gsd-core/workflows/execute-plan.md +36 -4
  197. package/gsd-core/workflows/explore.md +131 -4
  198. package/gsd-core/workflows/fast.md +10 -2
  199. package/gsd-core/workflows/health.md +73 -4
  200. package/gsd-core/workflows/import.md +4 -4
  201. package/gsd-core/workflows/ingest-docs.md +5 -5
  202. package/gsd-core/workflows/mvp-phase.md +6 -3
  203. package/gsd-core/workflows/new-milestone.md +14 -9
  204. package/gsd-core/workflows/new-project.md +14 -14
  205. package/gsd-core/workflows/next.md +12 -0
  206. package/gsd-core/workflows/plan-phase.md +41 -17
  207. package/gsd-core/workflows/plan-review-convergence.md +50 -2
  208. package/gsd-core/workflows/progress.md +34 -6
  209. package/gsd-core/workflows/quick/steps/plan-checker-loop.md +4 -4
  210. package/gsd-core/workflows/quick/steps/quick-verification.md +27 -6
  211. package/gsd-core/workflows/quick/steps/research-phase.md +2 -2
  212. package/gsd-core/workflows/quick.md +35 -15
  213. package/gsd-core/workflows/review.md +26 -5
  214. package/gsd-core/workflows/secure-phase.md +1 -1
  215. package/gsd-core/workflows/session-report.md +2 -1
  216. package/gsd-core/workflows/settings.md +66 -2
  217. package/gsd-core/workflows/ship.md +104 -44
  218. package/gsd-core/workflows/spec-phase.md +30 -12
  219. package/gsd-core/workflows/sync-skills.md +63 -8
  220. package/gsd-core/workflows/transition.md +46 -11
  221. package/gsd-core/workflows/ui-phase.md +5 -5
  222. package/gsd-core/workflows/ui-review.md +2 -2
  223. package/gsd-core/workflows/update.md +1 -1
  224. package/gsd-core/workflows/validate-phase.md +1 -1
  225. package/gsd-core/workflows/verify-work.md +9 -7
  226. package/hooks/dist/gsd-agent-isolation-guard.js +103 -14
  227. package/hooks/dist/gsd-check-update-worker.js +56 -13
  228. package/hooks/dist/gsd-check-update.js +19 -1
  229. package/hooks/dist/gsd-cursor-pre-tool.js +0 -3
  230. package/hooks/dist/gsd-cursor-subagent-start.js +77 -2
  231. package/hooks/dist/gsd-cursor-subagent-stop.js +3 -2
  232. package/hooks/dist/gsd-prompt-guard.js +21 -20
  233. package/hooks/dist/gsd-read-injection-scanner.js +38 -24
  234. package/hooks/dist/gsd-statusline.js +18 -0
  235. package/hooks/dist/gsd-update-banner.js +22 -1
  236. package/hooks/dist/gsd-workflow-guard.js +134 -36
  237. package/hooks/dist/lib/git-cmd.js +92 -59
  238. package/hooks/dist/lib/injection-patterns.js +45 -0
  239. package/hooks/dist/lib/isolation-deny-reason.js +39 -0
  240. package/hooks/dist/lib/isolation-sentinel.js +9 -0
  241. package/hooks/gsd-agent-isolation-guard.js +103 -14
  242. package/hooks/gsd-check-update-worker.js +56 -13
  243. package/hooks/gsd-check-update.js +19 -1
  244. package/hooks/gsd-cursor-pre-tool.js +0 -3
  245. package/hooks/gsd-cursor-subagent-start.js +77 -2
  246. package/hooks/gsd-cursor-subagent-stop.js +3 -2
  247. package/hooks/gsd-prompt-guard.js +21 -20
  248. package/hooks/gsd-read-injection-scanner.js +38 -24
  249. package/hooks/gsd-statusline.js +18 -0
  250. package/hooks/gsd-update-banner.js +22 -1
  251. package/hooks/gsd-workflow-guard.js +134 -36
  252. package/hooks/lib/git-cmd.js +92 -59
  253. package/hooks/lib/injection-patterns.js +45 -0
  254. package/hooks/lib/isolation-deny-reason.js +39 -0
  255. package/hooks/lib/isolation-sentinel.js +9 -0
  256. package/package.json +21 -9
  257. package/pi/gsd.cjs +19 -5
  258. package/scripts/baselines/planning-prompt-drift-baseline.json +4 -0
  259. package/scripts/baselines/planning-snapshot-bypass-baseline.json +12 -0
  260. package/scripts/baselines/unreachable-guard-drift-baseline.json +4 -0
  261. package/scripts/changeset/lint.cjs +60 -5
  262. package/scripts/check-alias-drift.cjs +7 -43
  263. package/scripts/check-contract-drift.cjs +297 -0
  264. package/scripts/ci-test-scope.cjs +19 -2
  265. package/scripts/command-contract-helpers.cjs +903 -1
  266. package/scripts/gen-adr-index.cjs +728 -38
  267. package/scripts/gen-capability-registry.cjs +3 -15
  268. package/scripts/gen-context-index.cjs +2 -11
  269. package/scripts/gen-health-docs.cjs +390 -0
  270. package/scripts/gen-inventory-manifest.cjs +50 -4
  271. package/scripts/gen-loop-host-contract.cjs +4 -24
  272. package/scripts/gen-registry.cjs +3 -14
  273. package/scripts/lib/alias-drift-families.cjs +46 -0
  274. package/scripts/lib/drift-scan.cjs +278 -0
  275. package/scripts/lint-allow-test-rule-refs.allowlist.json +1 -26
  276. package/scripts/lint-allow-test-rule-refs.effective-ceiling.json +4 -0
  277. package/scripts/lint-allow-test-rule-refs.unverified-ceiling.json +3 -0
  278. package/scripts/lint-canary-version-leak.cjs +73 -0
  279. package/scripts/lint-command-contract.cjs +96 -13
  280. package/scripts/lint-completion-predicate-drift.cjs +933 -0
  281. package/scripts/lint-completion-ratio-drift.cjs +214 -0
  282. package/scripts/lint-default-flip-documentation.cjs +193 -0
  283. package/scripts/lint-eslint-glob-coverage.allowlist.json +34 -0
  284. package/scripts/lint-eslint-glob-coverage.cjs +340 -0
  285. package/scripts/lint-frontmatter-scalar-broad-grep.cjs +237 -0
  286. package/scripts/lint-health-diagnostic-rule-table.cjs +404 -0
  287. package/scripts/lint-hooks-runtime-build-seam.cjs +262 -0
  288. package/scripts/lint-milestone-window-drift.cjs +468 -0
  289. package/scripts/lint-phase-enumeration-drift.cjs +479 -0
  290. package/scripts/lint-plan-count-drift.cjs +318 -0
  291. package/scripts/lint-planning-artifact-writer-drift.cjs +398 -0
  292. package/scripts/lint-planning-prompt-drift.cjs +434 -0
  293. package/scripts/lint-planning-snapshot-bypass-drift.cjs +544 -0
  294. package/scripts/lint-regression-test-names.cjs +15 -13
  295. package/scripts/lint-removed-but-needed.cjs +320 -0
  296. package/scripts/lint-state-field-drift.cjs +805 -0
  297. package/scripts/lint-state-write-path-drift.cjs +1045 -0
  298. package/scripts/lint-test-file-count.allowlist.json +21 -10
  299. package/scripts/lint-unreachable-guard-drift.cjs +843 -0
  300. package/scripts/lint-vendored-deps.cjs +124 -0
  301. package/scripts/pr-changed-files.cjs +63 -0
  302. package/scripts/pr-template-policy.cjs +14 -4
  303. package/scripts/prompt-injection-scan.sh +25 -0
  304. package/scripts/require-issue-link-policy.cjs +192 -0
  305. package/scripts/state-write-path-drift-baseline.json +19 -0
  306. package/scripts/sync-runtime-launcher.cjs +2 -4
  307. package/skills/gsd-autonomous/SKILL.md +0 -1
  308. package/skills/gsd-code-review/SKILL.md +1 -1
  309. package/skills/gsd-execute-phase/SKILL.md +1 -2
  310. package/skills/gsd-map-codebase/SKILL.md +1 -1
  311. package/skills/gsd-mempalace-capture/SKILL.md +1 -1
  312. package/skills/gsd-mempalace-recall/SKILL.md +1 -1
  313. package/skills/gsd-new-milestone/SKILL.md +1 -1
  314. package/skills/gsd-next/SKILL.md +0 -1
  315. package/skills/gsd-plan-phase/SKILL.md +0 -1
  316. package/skills/gsd-progress/SKILL.md +0 -1
  317. package/skills/gsd-quick/SKILL.md +1 -1
  318. package/skills/gsd-review-backlog/SKILL.md +2 -1
  319. package/skills/gsd-stats/SKILL.md +0 -1
  320. package/skills/gsd-verify-work/SKILL.md +1 -1
  321. package/vscode/package.json +1 -1
  322. package/gsd-core/workflows/discovery-phase.md +0 -298
  323. package/gsd-core/workflows/plan-milestone-gaps.md +0 -281
  324. package/gsd-core/workflows/verify-phase.md +0 -574
  325. package/scripts/affected-tests-lib.cjs +0 -554
  326. package/scripts/lint-allow-test-rule-refs.cjs +0 -162
  327. package/scripts/run-affected-tests.cjs +0 -7
  328. package/scripts/run-tests.cjs +0 -1051
@@ -27,19 +27,205 @@
27
27
  * "failed" and "ran cleanly with nothing to report" IS the defect this epic closes (#2494/#2605).
28
28
  */
29
29
  Object.defineProperty(exports, "__esModule", { value: true });
30
+ exports.MODEL_VALUE_MAX = exports.BANNER_SCAN_LINES = exports.UNRESOLVED_MODEL = exports.MODEL_SOURCE = void 0;
31
+ exports.parseModelBanner = parseModelBanner;
32
+ exports.parseTranscriptModel = parseTranscriptModel;
30
33
  exports.checkEgressHost = checkEgressHost;
31
34
  exports.probeLane = probeLane;
32
35
  exports.writeReviewOrStub = writeReviewOrStub;
33
36
  exports.handleOpencodeOutput = handleOpencodeOutput;
34
37
  exports.antigravityWatermark = antigravityWatermark;
35
38
  exports.antigravityTranscriptFallback = antigravityTranscriptFallback;
39
+ exports.antigravityModel = antigravityModel;
40
+ exports.resolveSpawnModel = resolveSpawnModel;
36
41
  exports.antigravityPrompt = antigravityPrompt;
37
42
  exports.antigravityArgv = antigravityArgv;
38
43
  exports.antigravityDiagnostic = antigravityDiagnostic;
39
44
  exports.stampBlindReview = stampBlindReview;
45
+ exports.stampUngroundedReview = stampUngroundedReview;
40
46
  exports.runOpenAiCompatible = runOpenAiCompatible;
41
47
  exports.runLane = runLane;
42
48
  const review_lane_invocation_cjs_1 = require("./review-lane-invocation.cjs");
49
+ /* ------------------------------------------------------------------ *
50
+ * #2295 — the resolved model
51
+ * ------------------------------------------------------------------ */
52
+ /**
53
+ * How a lane's resolved model was recovered. FROZEN — adding a member is three coordinated
54
+ * changes (enum + emitting site + the test locking `Object.keys(MODEL_SOURCE).sort()`), the same
55
+ * discipline `PARITY_VIOLATION` and `LANE_UNAVAILABLE` already carry.
56
+ *
57
+ * The source travels with the value on purpose. Postel's robustness principle is usually quoted
58
+ * as "be liberal in what you accept", but its modern caveat is the load-bearing half here:
59
+ * liberal must not mean GUESS SILENTLY. Two of these arms parse third-party text this project
60
+ * does not own — a CLI's startup banner and an undocumented on-disk session log — so a bare
61
+ * model string would be an unattributable claim. Recording HOW it was recovered lets a reader
62
+ * weigh `pinned` (certain) against `banner` (heuristic) without leaving the file.
63
+ */
64
+ exports.MODEL_SOURCE = Object.freeze({
65
+ /** `review.models.<slug>`, or an ADR-1517 instance `--model`, that really reached the invocation. */
66
+ PINNED: 'pinned',
67
+ /** An OpenAI-compatible server echoed the model it actually ran. The most authoritative arm. */
68
+ SERVED: 'served',
69
+ /** openai-http: discovered from `/v1/models`, or the declared `fallbackModel`; the server did not echo one. */
70
+ REQUESTED: 'requested',
71
+ /** The CLI's own startup banner named it. File-output lanes only — see `resolveSpawnModel`. */
72
+ BANNER: 'banner',
73
+ /** The lane handler's own on-disk session log named it (`agy`'s `transcript_full.jsonl`). */
74
+ TRANSCRIPT: 'transcript',
75
+ /** Nothing recoverable. An explicit non-answer, never an omitted field. */
76
+ UNKNOWN: 'unknown',
77
+ });
78
+ /** The one shape every unresolvable case returns, so callers never hand-build it inconsistently. */
79
+ exports.UNRESOLVED_MODEL = Object.freeze({ value: null, source: exports.MODEL_SOURCE.UNKNOWN });
80
+ /**
81
+ * How far into captured output a startup banner may appear, in lines. A banner is by definition
82
+ * the FIRST thing a CLI prints; scanning further only raises the odds of matching something that
83
+ * is not one.
84
+ */
85
+ exports.BANNER_SCAN_LINES = 40;
86
+ /** Longest plausible model identifier. Anything past this is not a model name, it is a payload. */
87
+ exports.MODEL_VALUE_MAX = 200;
88
+ /**
89
+ * C0 controls (0x00-0x1F), DEL (0x7F) and C1 controls (0x80-0x9F). A model identifier never
90
+ * legitimately contains one, and a newline in particular is the frontmatter-injection vector
91
+ * this guards against — a recorded model value is written verbatim into REVIEWS.md YAML
92
+ * frontmatter, so a value carrying `\n` could forge arbitrary sibling keys. Deliberately does
93
+ * NOT include `:` — `llama3:70b` and `qwen2.5:7b` are legitimate model ids.
94
+ */
95
+ const CONTROL_CHAR_RE = /[\u0000-\u001F\u007F-\u009F]/;
96
+ /**
97
+ * A recovered model value, or `null`. Shares `configString`'s unset-shape rule, length-caps it,
98
+ * then REJECTS (never strips or escapes) a value carrying a control character — see
99
+ * `CONTROL_CHAR_RE`. Rejecting rather than sanitizing means an anomalous value is recorded as
100
+ * `unknown` rather than silently rewritten into something that merely looks safe.
101
+ */
102
+ function normalizeModelValue(raw) {
103
+ const value = (0, review_lane_invocation_cjs_1.configString)(raw);
104
+ if (value === null)
105
+ return null;
106
+ if (value.length > exports.MODEL_VALUE_MAX)
107
+ return null;
108
+ return CONTROL_CHAR_RE.test(value) ? null : value;
109
+ }
110
+ /**
111
+ * The one place a `ResolvedModel` is built. Normalizing here rather than per-arm is what makes
112
+ * the `value !== null` ⟺ `source !== 'unknown'` invariant structural instead of a convention
113
+ * five call sites have to remember — and it is the single choke point where a hostile value is
114
+ * refused before it can reach the REVIEWS.md frontmatter a lane's result is rendered into.
115
+ */
116
+ function recordedModel(raw, source) {
117
+ const value = normalizeModelValue(raw);
118
+ return value === null ? exports.UNRESOLVED_MODEL : { value, source };
119
+ }
120
+ /**
121
+ * The reasoning effort GSD applied to this invocation, folded into the recorded value (#2295).
122
+ *
123
+ * The issue asks for `gpt-5.6-sol (reasoning=high)`, and the Antigravity lane already reports its
124
+ * own tier the same way (`Gemini 3.5 Flash (Medium)`) — so effort belongs in the model designation
125
+ * a human compares, not in a separate field they would have to join by hand.
126
+ *
127
+ * The source is GSD's OWN resolved execution policy, not the CLI's config or banner, so this arm
128
+ * is certain in a way the banner and transcript arms are not. `MODEL_VALUE_MAX` bounds the model
129
+ * id the suffix is appended to; the suffix itself is GSD-owned and bounded, so it is deliberately
130
+ * outside that cap rather than able to push a legitimate id over it.
131
+ */
132
+ function withEffort(resolved, effort) {
133
+ if (resolved.value === null)
134
+ return resolved;
135
+ const normalized = normalizeModelValue(effort);
136
+ if (normalized === null)
137
+ return resolved;
138
+ return { value: `${resolved.value} (reasoning=${normalized})`, source: resolved.source };
139
+ }
140
+ /** A line that IS a `model:` declaration — leading banner chrome allowed, trailing prose not. */
141
+ const BANNER_LINE_RE = /^[\s>*|-]*model\s*:\s*(.+)$/i;
142
+ /**
143
+ * The model a CLI named in its own startup banner, or `null` (#2295).
144
+ *
145
+ * TOTAL: never throws, for any string. Deliberately dull — a bounded line window and one anchored
146
+ * regex — because this is the cleverest code in the change and Kernighan's Law says debugging is
147
+ * twice as hard as writing.
148
+ *
149
+ * AMBIGUITY IS NOT RESOLVED, IT IS REFUSED. Two DIFFERENT candidate values in the window means we
150
+ * cannot tell which one ran, and picking the first would attribute a review to a model on a coin
151
+ * flip. Repetition of one identical value is not ambiguity and is accepted.
152
+ */
153
+ function parseModelBanner(text) {
154
+ const lines = String(text ?? '').split(/\r?\n/).slice(0, exports.BANNER_SCAN_LINES);
155
+ const found = new Set();
156
+ for (const line of lines) {
157
+ const m = BANNER_LINE_RE.exec(line);
158
+ if (!m)
159
+ continue;
160
+ const value = normalizeModelValue(m[1]);
161
+ if (value !== null)
162
+ found.add(value);
163
+ }
164
+ return found.size === 1 ? [...found][0] : null;
165
+ }
166
+ /** An own, string-valued `model` key on a plain object — never a prototype member, never coerced. */
167
+ function ownModel(node) {
168
+ if (node === null || typeof node !== 'object' || Array.isArray(node))
169
+ return null;
170
+ const record = node;
171
+ if (!Object.prototype.hasOwnProperty.call(record, 'model'))
172
+ return null;
173
+ return normalizeModelValue(record.model);
174
+ }
175
+ /**
176
+ * Key names the depth-2 wrapper scan below refuses to descend through, even though `JSON.parse`
177
+ * gives each an ordinary OWN data property here (never the real `Object.prototype` accessor — see
178
+ * `resolveConvId`'s `#3118` note on the same class of trap). The transcript is third-party JSON on
179
+ * a trust boundary; an entry SHAPED like `{"constructor":{"model":"x"}}` must never resolve as
180
+ * though "constructor" were a legitimate settings-wrapper key.
181
+ */
182
+ const UNSAFE_WRAPPER_KEYS = new Set(['__proto__', 'constructor', 'prototype']);
183
+ /**
184
+ * The session model named by the LAST settings-shaped entry of a `transcript_full.jsonl`, or `null`.
185
+ *
186
+ * TOTAL: never throws, for any string. Every line is independently parsed, so one truncated or
187
+ * garbage line cannot poison the file.
188
+ *
189
+ * SCOPE IS BOUNDED AT DEPTH TWO, AND THAT BOUND IS THE DESIGN. The transcript is an undocumented
190
+ * third-party format; its settings entry may carry `model` at the top level or one level down
191
+ * under a wrapper whose key name we cannot know without guessing. A depth is knowable; a key name
192
+ * is not. A recursive search over attacker-adjacent JSON would be both unbounded and a licence to
193
+ * match any `model`-ish key anywhere, so anything deeper degrades to `null` — which the maintainer
194
+ * ruled an acceptable recorded value, unlike a wrong one.
195
+ *
196
+ * `typeof null === 'object'`, so the null guard in `ownModel` is explicit rather than implied —
197
+ * the same trap `resolveConvId` documents at #3118.
198
+ */
199
+ function parseTranscriptModel(text) {
200
+ let latest = null;
201
+ for (const line of String(text ?? '').split(/\r?\n/)) {
202
+ if (!line.trim())
203
+ continue;
204
+ let entry;
205
+ try {
206
+ entry = JSON.parse(line);
207
+ }
208
+ catch {
209
+ continue; // one bad line is not a bad file
210
+ }
211
+ const direct = ownModel(entry);
212
+ if (direct !== null) {
213
+ latest = direct;
214
+ continue;
215
+ }
216
+ if (entry === null || typeof entry !== 'object' || Array.isArray(entry))
217
+ continue;
218
+ const record = entry;
219
+ for (const key of Object.keys(record)) {
220
+ if (UNSAFE_WRAPPER_KEYS.has(key))
221
+ continue;
222
+ const nested = ownModel(record[key]);
223
+ if (nested !== null)
224
+ latest = nested;
225
+ }
226
+ }
227
+ return latest;
228
+ }
43
229
  /**
44
230
  * Compare a lane's re-resolved egress destination against the one the user consented to.
45
231
  *
@@ -236,58 +422,37 @@ function handleOpencodeOutput(rawStdout) {
236
422
  diagnostic: `stop reason=${stopReason}, output tokens=${outputTokens}`,
237
423
  };
238
424
  }
239
- /**
240
- * `antigravity` — three layers, a two-level timeout, and a stale-response watermark.
241
- *
242
- * Layer 1 is stdout, which works on macOS/Linux/WSL. On native Windows `agy -p` silently produces
243
- * no stdout despite the API call succeeding (an upstream `text_drip.go` non-TTY flush bug), so
244
- * layer 2 reads the transcript `agy` always persists to disk. Layer 3 is a diagnostic stub.
245
- *
246
- * THE WATERMARK IS THE SUBTLE PART. Without it, layer 2 reads the last `PLANNER_RESPONSE` in the
247
- * transcript regardless of when it was written — including one from a PREVIOUS invocation in the
248
- * same workspace, silently presenting a stale review as this run's. So the transcript line count is
249
- * snapshotted BEFORE the spawn and only lines appended after it are considered. If the conversation
250
- * id changed, `agy` started a fresh session and every line is new (skip 0).
251
- *
252
- * `takeWatermark` must therefore be called before `runAntigravity`; the plan's outer timeout is the
253
- * external wall-clock cap that `--print-timeout` cannot provide (it cannot fire before a session
254
- * exists — a process can stall pre-session and outlive its own native timeout, #2073 mode 3).
255
- *
256
- * KNOWN LIMIT — the watermark is sequential, not concurrent. `last_conversations.json` is keyed by
257
- * WORKSPACE, so two `/gsd:review` runs against the same repo at the same time resolve the same
258
- * conversation id and share one transcript. The fallback then takes "the latest DONE
259
- * PLANNER_RESPONSE after the watermark" with nothing to tell the two runs apart, so one run could
260
- * read the other's response. This cannot be closed here: `agy` exposes no per-invocation id to
261
- * filter on, and the transcript carries none. It is stated rather than silently tolerated because
262
- * the guarantee this function advertises ("never stale") holds only for sequential use, and a
263
- * future reader deserves to know which half is actually guaranteed.
264
- */
265
425
  function antigravityWatermark(workspace, deps) {
266
- const cachePath = `${deps.homeDir}/.gemini/antigravity-cli/cache/last_conversations.json`;
267
- if (!deps.exists(cachePath))
268
- return { convId: '', lines: 0 };
269
- let cache;
270
- try {
271
- cache = JSON.parse(deps.readFile(cachePath));
272
- }
273
- catch {
274
- return { convId: '', lines: 0 };
275
- }
276
- const convId = resolveConvId(cache, workspace);
426
+ const convId = resolveWorkspaceConvId(workspace, deps);
277
427
  if (!convId)
278
- return { convId: '', lines: 0 };
428
+ return { convId: '', lines: 0, fullLines: 0 };
279
429
  const tx = transcriptPath(deps.homeDir, convId);
280
- if (!deps.exists(tx))
281
- return { convId, lines: 0 };
282
- try {
283
- return { convId, lines: deps.readFile(tx).split(/\r?\n/).filter((l) => l.trim()).length };
430
+ let lines = 0;
431
+ let unreadable;
432
+ if (deps.exists(tx)) {
433
+ try {
434
+ lines = deps.readFile(tx).split(/\r?\n/).filter((l) => l.trim()).length;
435
+ }
436
+ catch {
437
+ // #3118: this conv-id pre-dates this run, so its transcript exists but this run cannot
438
+ // verify its line count. Reporting `lines: 0` would assert a fact we could not check — flag
439
+ // it instead so the fallback can decline rather than silently skip zero and replay a stale
440
+ // response.
441
+ unreadable = true;
442
+ }
284
443
  }
285
- catch {
286
- // #3118: this conv-id pre-dates this run, so its transcript exists but this run cannot verify
287
- // its line count. Reporting `lines: 0` would assert a fact we could not check — flag it instead
288
- // so the fallback can decline rather than silently skip zero and replay a stale response.
289
- return { convId, lines: 0, unreadable: true };
444
+ const fullTx = fullTranscriptPath(deps.homeDir, convId);
445
+ let fullLines = 0;
446
+ let fullUnreadable;
447
+ if (deps.exists(fullTx)) {
448
+ try {
449
+ fullLines = deps.readFile(fullTx).split(/\r?\n/).filter((l) => l.trim()).length;
450
+ }
451
+ catch {
452
+ fullUnreadable = true;
453
+ }
290
454
  }
455
+ return { convId, lines, ...(unreadable ? { unreadable } : {}), fullLines, ...(fullUnreadable ? { fullUnreadable } : {}) };
291
456
  }
292
457
  /**
293
458
  * Workspace lookup is case-insensitive — the leg's jq did `ascii_downcase` on both sides.
@@ -314,17 +479,21 @@ function resolveConvId(cache, workspace) {
314
479
  }
315
480
  return '';
316
481
  }
317
- function transcriptPath(homeDir, convId) {
318
- return `${homeDir}/.gemini/antigravity-cli/brain/${convId}/.system_generated/logs/transcript.jsonl`;
319
- }
320
482
  /**
321
- * Layer 2: the newest `PLANNER_RESPONSE` written AFTER the watermark, or `''`.
483
+ * The `agy` conversation id for this workspace, or `''` when the cache is absent, unreadable or
484
+ * names none.
322
485
  *
323
- * Returning `''` rather than the newest entry overall is the whole point — an empty result lets
324
- * layer 3 fire with an honest diagnostic, where a stale one would be indistinguishable from a
325
- * successful review.
486
+ * Reads and parses `last_conversations.json` once, so `antigravityWatermark`,
487
+ * `antigravityTranscriptFallback` and `antigravityModel` — three callers as of #2295 — share one
488
+ * lookup instead of each hand-rolling the same exists/readFile/JSON.parse/resolveConvId sequence.
489
+ *
490
+ * #3118: a successful `JSON.parse` does not by itself make the payload a usable object —
491
+ * `JSON.parse('null')` succeeds and returns `null`, so a truncated/zeroed cache file slips past a
492
+ * parse-only try/catch; `resolveConvId`'s own guard handles the object-shape half of that trap.
493
+ * Workspace lookup is case-insensitive — the leg's jq did `ascii_downcase` on both sides — which
494
+ * `resolveConvId` implements.
326
495
  */
327
- function antigravityTranscriptFallback(workspace, mark, deps) {
496
+ function resolveWorkspaceConvId(workspace, deps) {
328
497
  const cachePath = `${deps.homeDir}/.gemini/antigravity-cli/cache/last_conversations.json`;
329
498
  if (!deps.exists(cachePath))
330
499
  return '';
@@ -335,7 +504,24 @@ function antigravityTranscriptFallback(workspace, mark, deps) {
335
504
  catch {
336
505
  return '';
337
506
  }
338
- const convId = resolveConvId(cache, workspace);
507
+ return resolveConvId(cache, workspace);
508
+ }
509
+ function transcriptPath(homeDir, convId) {
510
+ return `${homeDir}/.gemini/antigravity-cli/brain/${convId}/.system_generated/logs/transcript.jsonl`;
511
+ }
512
+ /** The sibling log that carries `agy`'s SETTINGS entries (and so the session's model), not the review body. */
513
+ function fullTranscriptPath(homeDir, convId) {
514
+ return `${homeDir}/.gemini/antigravity-cli/brain/${convId}/.system_generated/logs/transcript_full.jsonl`;
515
+ }
516
+ /**
517
+ * Layer 2: the newest `PLANNER_RESPONSE` written AFTER the watermark, or `''`.
518
+ *
519
+ * Returning `''` rather than the newest entry overall is the whole point — an empty result lets
520
+ * layer 3 fire with an honest diagnostic, where a stale one would be indistinguishable from a
521
+ * successful review.
522
+ */
523
+ function antigravityTranscriptFallback(workspace, mark, deps) {
524
+ const convId = resolveWorkspaceConvId(workspace, deps);
339
525
  if (!convId)
340
526
  return '';
341
527
  // #3118: the watermark could not read this conversation's transcript, so there is no trustworthy
@@ -373,6 +559,97 @@ function antigravityTranscriptFallback(workspace, mark, deps) {
373
559
  }
374
560
  return latest;
375
561
  }
562
+ /**
563
+ * The model `agy` ran under, recovered from its own `transcript_full.jsonl`, or `null` (#2295).
564
+ *
565
+ * THE STALENESS RULE HERE IS DELIBERATELY LOOSER THAN `antigravityTranscriptFallback`'s, and the
566
+ * difference is the whole reason this is a separate function rather than a flag on that one.
567
+ *
568
+ * For a REVIEW BODY, a pre-watermark entry is fatal: it would present a previous run's review as
569
+ * this one's. For the MODEL it is not. `last_conversations.json` is keyed by WORKSPACE, so a
570
+ * matching conv-id means `agy` reused the SAME SESSION — and that session's model IS the model
571
+ * this run ran under. `agy` reuses sessions per workspace, so a strict post-watermark-only scan
572
+ * would report `unknown` for most real runs while being no more correct.
573
+ *
574
+ * A DIFFERENT conv-id means a fresh session, where every line is already this run's.
575
+ * `fullUnreadable` still declines outright (#3118's fail-closed shape): a file that indisputably
576
+ * exists but could not be read is not the same fact as an absent one.
577
+ */
578
+ function antigravityModel(workspace, mark, deps) {
579
+ const convId = resolveWorkspaceConvId(workspace, deps);
580
+ if (!convId)
581
+ return null;
582
+ // #3118, applied to the model arm: a file that indisputably exists but could not be read is not
583
+ // the same fact as an absent one — decline rather than guess.
584
+ if (mark.fullUnreadable === true && convId === mark.convId)
585
+ return null;
586
+ const fullTx = fullTranscriptPath(deps.homeDir, convId);
587
+ if (!deps.exists(fullTx))
588
+ return null;
589
+ let fullText;
590
+ try {
591
+ fullText = deps.readFile(fullTx);
592
+ }
593
+ catch {
594
+ return null;
595
+ }
596
+ // Same conv-id ⇒ agy reused this session; only lines beyond the watermark are guaranteed new,
597
+ // but see the doc-comment above for why a pre-watermark fallback still applies for the MODEL.
598
+ // Different id ⇒ fresh session, so every line is already this run's.
599
+ const sameSession = convId === mark.convId;
600
+ const lines = fullText.split(/\r?\n/).filter((l) => l.trim());
601
+ const skip = sameSession ? mark.fullLines : 0;
602
+ const afterWatermark = parseTranscriptModel(lines.slice(skip).join('\n'));
603
+ if (afterWatermark !== null)
604
+ return afterWatermark;
605
+ // Nothing new since the watermark. For a same-session reuse, the session's own (pre-watermark)
606
+ // model is still this run's model — the whole point of the looser rule above.
607
+ return sameSession ? parseTranscriptModel(fullText) : null;
608
+ }
609
+ /**
610
+ * The model a spawned lane ran under. Precedence is TOTAL and ORDERED (#2295).
611
+ *
612
+ * `pinned` first because it is the only arm that is certain. Then the handler's own transcript,
613
+ * then the startup banner.
614
+ *
615
+ * THE BANNER ARM IS GATED ON `outputTarget.kind === 'file'`, and that gate is the single most
616
+ * important line in this function. A lane whose review comes back on STDOUT has its review text
617
+ * in exactly the buffer the banner scan would read — so a review that merely DISCUSSES a model
618
+ * ("model: gpt-5 is the wrong choice here") would be recorded as that lane's resolved model. Only
619
+ * a lane that writes its review to a FILE has a stdout stream that is banner and nothing else.
620
+ * The condition is derived from DECLARED DATA rather than from a slug check, so it covers today's
621
+ * one file-output lane and any future one without naming either. (`stampBlindReview` anchors its
622
+ * own tells to the first five lines for the same class of reason.)
623
+ *
624
+ * `repoRoot` is the workspace `agy` keys its conversation cache by, so it is required rather than
625
+ * defaulted — an empty workspace would silently resolve no conversation and look identical to "no
626
+ * model recorded".
627
+ */
628
+ function resolveSpawnModel(plan, out, mark, deps, repoRoot) {
629
+ try {
630
+ if (plan.model)
631
+ return withEffort(recordedModel(plan.model, exports.MODEL_SOURCE.PINNED), plan.effort);
632
+ if (plan.handler === 'antigravity') {
633
+ let transcript;
634
+ try {
635
+ transcript = antigravityModel(repoRoot, mark, deps);
636
+ }
637
+ catch {
638
+ return exports.UNRESOLVED_MODEL;
639
+ }
640
+ return withEffort(recordedModel(transcript, exports.MODEL_SOURCE.TRANSCRIPT), plan.effort);
641
+ }
642
+ if (plan.outputTarget.kind === 'file') {
643
+ const banner = parseModelBanner(out.stdout ?? '') ?? parseModelBanner(out.stderr ?? '');
644
+ return withEffort(recordedModel(banner, exports.MODEL_SOURCE.BANNER), plan.effort);
645
+ }
646
+ return exports.UNRESOLVED_MODEL;
647
+ }
648
+ catch {
649
+ // A model arm must NEVER fail the lane — see the module banner's TOTAL contract.
650
+ return exports.UNRESOLVED_MODEL;
651
+ }
652
+ }
376
653
  /**
377
654
  * Antigravity's prompt variant (#2176).
378
655
  *
@@ -478,6 +755,50 @@ function stampBlindReview(review) {
478
755
  'review — down-weight its verdict in the Consensus Summary.\n\n' +
479
756
  review);
480
757
  }
758
+ /**
759
+ * A `path/to/file:line`-shaped source citation (#3194).
760
+ *
761
+ * The Review Instructions (review.md) require every reviewer to "cite concrete
762
+ * `path/to/file:line` evidence"; this recognizes that shape in review output. Two anchors
763
+ * make the match a source citation rather than any `colon-digits`:
764
+ * - the token before the colon must contain a path separator (`/` or `\`) or end in a
765
+ * `.extension`, so a bare PLAN-line reference — "see line 42", "L12-L18", the invented
766
+ * references measured in #3194 — does not match;
767
+ * - the token may not itself contain `:` and may not start immediately after `/` or `:`,
768
+ * so a URL (`http://localhost:8080`) and its host:port do not match either.
769
+ *
770
+ * KNOWN LIMIT (deliberate, #3194 scope): presence is checked, not resolution. A citation to
771
+ * a line that does not exist still counts — catching invented references that look impeccable
772
+ * requires repo access at stamp time and is follow-up material, not part of this fix.
773
+ */
774
+ const SOURCE_CITATION_RE = /(?<![/:])(?:[^\s:]*[/\\][^\s:]*|[^\s:]*\.[A-Za-z0-9]{1,16}):[0-9]+/;
775
+ /** The marker the Consensus Summary step recognizes and down-weights (#3194). */
776
+ const UNGROUNDED_MARKER = '[reviewed-without-source-citations]';
777
+ /**
778
+ * Stamp a machine-readable marker when a source-grounded lane's review cites no `file:line`
779
+ * evidence (#3194).
780
+ *
781
+ * The sibling of `stampBlindReview`, for a different failure mode: that one fires when a
782
+ * reviewer REPORTS it had no repo access; this one fires when a lane that DECLARES
783
+ * `source-grounded` evidence delivered none — the review restates the plan's own claims with
784
+ * at most invented plan-line references. Either way the Consensus Summary must not count the
785
+ * verdict at full weight, which is why both prepend a marker the consensus step recognizes.
786
+ *
787
+ * Idempotent and empty-safe: an already-stamped review passes through unchanged, and an empty
788
+ * review is left to the empty-output policy's diagnostic stub.
789
+ */
790
+ function stampUngroundedReview(review) {
791
+ if ((0, review_lane_invocation_cjs_1.isEmptyReview)(review))
792
+ return review;
793
+ if (review.startsWith(`> ${UNGROUNDED_MARKER}`))
794
+ return review;
795
+ if (SOURCE_CITATION_RE.test(review))
796
+ return review;
797
+ return (`> ${UNGROUNDED_MARKER} This reviewer declared source-grounded evidence but cited no ` +
798
+ 'file:line source evidence, so it reviewed the pasted plan text only — down-weight its ' +
799
+ 'verdict in the Consensus Summary.\n\n' +
800
+ review);
801
+ }
481
802
  /**
482
803
  * `openai-compatible` — model discovery, the chat-completions round trip, and the served-model
483
804
  * mismatch warning.
@@ -485,6 +806,12 @@ function stampBlindReview(review) {
485
806
  * The raw body is returned alongside the content because an OpenAI-compatible server reports errors
486
807
  * with an HTTP 4xx/5xx and the JSON in the BODY. The bash piped the response straight into `jq`,
487
808
  * which discarded exactly that evidence; the stub appends it now.
809
+ *
810
+ * MODEL (#2295): what actually ran beats what was asked for. If the response echoes a `model`
811
+ * field, that is `SERVED` — the most authoritative arm, since it is the server's own report of
812
+ * what it ran. Otherwise the request falls back to `REQUESTED`: the discovered or `fallbackModel`
813
+ * value that was actually sent, recorded even when the request itself failed — the ADR-2782
814
+ * served-model mismatch warning above is preserved unchanged.
488
815
  */
489
816
  async function runOpenAiCompatible(plan, promptText, deps) {
490
817
  let model = plan.model;
@@ -504,12 +831,14 @@ async function runOpenAiCompatible(plan, promptText, deps) {
504
831
  }
505
832
  if (!model)
506
833
  model = plan.fallbackModel;
834
+ const requested = recordedModel(model, exports.MODEL_SOURCE.REQUESTED);
507
835
  const body = JSON.stringify({ model, messages: [{ role: 'user', content: promptText }] });
508
836
  const res = await deps.httpJson(plan.url, { method: 'POST', body, timeoutMs: plan.timeoutMs });
509
837
  if (!res.ok && !res.body) {
510
- return { review: '', rawBody: res.error ?? `HTTP ${res.status}` };
838
+ return { review: '', rawBody: res.error ?? `HTTP ${res.status}`, model: requested };
511
839
  }
512
840
  let review = '';
841
+ let served = null;
513
842
  try {
514
843
  const parsed = JSON.parse(res.body);
515
844
  // A server that quietly serves a different model than requested produces a review the user
@@ -518,6 +847,9 @@ async function runOpenAiCompatible(plan, promptText, deps) {
518
847
  deps.warn(`${plan.slug} served model '${parsed.model}' but '${model}' was requested. ` +
519
848
  `Review may be from a different model.`);
520
849
  }
850
+ const servedModel = recordedModel(parsed.model, exports.MODEL_SOURCE.SERVED);
851
+ if (servedModel.source !== exports.MODEL_SOURCE.UNKNOWN)
852
+ served = servedModel;
521
853
  const content = parsed.choices?.[0]?.message?.content;
522
854
  if (typeof content === 'string')
523
855
  review = content;
@@ -525,7 +857,7 @@ async function runOpenAiCompatible(plan, promptText, deps) {
525
857
  catch {
526
858
  /* leave review empty — the raw body carries the diagnosis */
527
859
  }
528
- return { review, rawBody: res.body };
860
+ return { review, rawBody: res.body, model: served ?? requested };
529
861
  }
530
862
  /* ------------------------------------------------------------------ *
531
863
  * Orchestration
@@ -537,7 +869,8 @@ async function runOpenAiCompatible(plan, promptText, deps) {
537
869
  * lane whose destination changed never receives the plan text even once.
538
870
  */
539
871
  async function runLane(plan, deps, opts) {
540
- const base = { slug: plan.slug, stubbed: false };
872
+ // Nothing ran for either early exit below, so there is nothing to attribute a model to (#2295).
873
+ const base = { slug: plan.slug, stubbed: false, model: exports.UNRESOLVED_MODEL };
541
874
  if (plan.transport === 'openai-http') {
542
875
  const egress = checkEgressHost(opts.consentedHost, plan.host);
543
876
  if (!egress.allowed) {
@@ -562,11 +895,20 @@ async function runLane(plan, deps, opts) {
562
895
  }
563
896
  function runSpawnLane(plan, deps, repoRoot) {
564
897
  const input = plan.stdin && deps.exists(plan.stdin) ? deps.readFile(plan.stdin) : undefined;
565
- const mark = plan.handler === 'antigravity' ? antigravityWatermark(repoRoot, deps) : { convId: '', lines: 0 };
898
+ const mark = plan.handler === 'antigravity'
899
+ ? antigravityWatermark(repoRoot, deps)
900
+ : { convId: '', lines: 0, fullLines: 0 };
566
901
  const argv = plan.handler === 'antigravity'
567
902
  ? antigravityArgv(plan.argv, plan.promptPath, repoRoot, deps)
568
903
  : plan.argv;
569
- const out = deps.spawn(plan.binary, argv, { input, timeoutMs: plan.timeoutMs });
904
+ const out = deps.spawn(plan.binary, argv, {
905
+ input,
906
+ timeoutMs: plan.timeoutMs,
907
+ ...(plan.env ? { env: plan.env } : {}),
908
+ });
909
+ // The model arm reads the RAW spawn outcome here, deliberately, so it sees `out.stdout`/`out.stderr`
910
+ // exactly as the process emitted them — before the handlers below reassign `review` (#2295).
911
+ const model = resolveSpawnModel(plan, out, mark, deps, repoRoot);
570
912
  // #3086: surface spawn errors (ENOENT, ETIMEDOUT, etc.) that would otherwise
571
913
  // be silently dropped — the review path read only stdout/stderr and treated
572
914
  // an empty-stderr spawn failure as "the model had nothing to say".
@@ -602,17 +944,30 @@ function runSpawnLane(plan, deps, repoRoot) {
602
944
  // Layer 3. `emptyOutput: 'handler-owned'` means the generic stub does not fire for this lane,
603
945
  // so if nothing is written here the lane goes out empty — the #2073 failure itself.
604
946
  if ((0, review_lane_invocation_cjs_1.isEmptyReview)(review)) {
947
+ // The model is kept on this stub path deliberately (#2295): #2073 mode 2 is exactly a
948
+ // pinned model that 404s server-side and exits 0 with empty output, so the model IS the
949
+ // diagnosis — dropping it here would throw away the one piece of evidence the stub exists
950
+ // to preserve.
605
951
  deps.writeFile(plan.reviewPath, `${antigravityDiagnostic(deps)}\n`);
606
- return { slug: plan.slug, ok: true, stubbed: true };
952
+ return { slug: plan.slug, ok: true, stubbed: true, model };
607
953
  }
608
954
  }
955
+ // #3194: verify the declared evidence class against the review's actual output. A
956
+ // source-grounded lane whose review cites no file:line evidence reviewed the plan text
957
+ // only; stamp it so the Consensus Summary down-weights the verdict instead of silently
958
+ // trusting the lane's declaration. diff-only lanes are exempt — their verdict is already
959
+ // folded in as a diff observation, and the citation check must not change that surface.
960
+ if (plan.evidenceClass !== 'diff-only')
961
+ review = stampUngroundedReview(review);
609
962
  const { stubbed } = writeReviewOrStub(plan, review, deps, extra);
610
- return { slug: plan.slug, ok: true, stubbed };
963
+ return { slug: plan.slug, ok: true, stubbed, model };
611
964
  }
612
965
  async function runHttpLane(plan, deps) {
613
966
  const promptText = deps.exists(plan.promptPath) ? deps.readFile(plan.promptPath) : '';
614
- const { review, rawBody } = await runOpenAiCompatible(plan, promptText, deps);
967
+ const { review, rawBody, model } = await runOpenAiCompatible(plan, promptText, deps);
968
+ // #3194: same verification on the http path — see runSpawnLane.
969
+ const stamped = plan.evidenceClass !== 'diff-only' ? stampUngroundedReview(review) : review;
615
970
  deps.writeFile(plan.errPath, '');
616
- const { stubbed } = writeReviewOrStub(plan, review, deps, rawBody);
617
- return { slug: plan.slug, ok: true, stubbed };
971
+ const { stubbed } = writeReviewOrStub(plan, stamped, deps, rawBody);
972
+ return { slug: plan.slug, ok: true, stubbed, model };
618
973
  }