opencode-codeops 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/CHANGELOG.md +179 -0
  2. package/LICENSE +21 -0
  3. package/README.md +171 -0
  4. package/_shared/auto-design.md +129 -0
  5. package/_shared/layout-convention.md +198 -0
  6. package/_shared/quality-profile.md +134 -0
  7. package/_shared/recommendation-hardening.md +166 -0
  8. package/_shared/scope-expansion-control.md +176 -0
  9. package/_shared/spec-first-ordering.md +79 -0
  10. package/_shared/zero-ambiguity-gate.md +311 -0
  11. package/agent-templates/codebase-scout.md +17 -0
  12. package/agent-templates/concurrency-auditor.md +5 -0
  13. package/agent-templates/design-challenger.md +26 -0
  14. package/agent-templates/financial-integrity-auditor.md +5 -0
  15. package/agent-templates/perf-auditor.md +23 -0
  16. package/agent-templates/phase-reviewer.md +54 -0
  17. package/agent-templates/plan-task-executor-opus.md +46 -0
  18. package/agent-templates/plan-task-executor.md +43 -0
  19. package/agent-templates/preflight-auditor.md +45 -0
  20. package/agent-templates/security-auditor.md +42 -0
  21. package/agent-templates/semantics-reviewer.md +5 -0
  22. package/agent-templates/spec-test-author.md +29 -0
  23. package/agents/concurrency-auditor.md +15 -0
  24. package/agents/correctness-reviewer.md +66 -0
  25. package/agents/demanding-executor.md +58 -0
  26. package/agents/design-challenger.md +38 -0
  27. package/agents/executor.md +55 -0
  28. package/agents/explorer.md +29 -0
  29. package/agents/financial-integrity-auditor.md +15 -0
  30. package/agents/performance-auditor.md +35 -0
  31. package/agents/preflight-auditor.md +57 -0
  32. package/agents/security-auditor.md +54 -0
  33. package/agents/semantics-reviewer.md +15 -0
  34. package/agents/spec-test-author.md +41 -0
  35. package/bin/codeops-worktree +244 -0
  36. package/bin/index.mjs +106 -0
  37. package/bin/install-agents.mjs +453 -0
  38. package/bin/install-skills.mjs +466 -0
  39. package/bin/lib/opencode-install.mjs +185 -0
  40. package/install.sh +55 -0
  41. package/package.json +73 -0
  42. package/plugin/index.ts +181 -0
  43. package/references/domains/compiler-and-language.md +28 -0
  44. package/references/domains/data-and-migration.md +22 -0
  45. package/references/domains/distributed-and-concurrent.md +26 -0
  46. package/references/domains/financial-system.md +28 -0
  47. package/references/domains/selection.md +19 -0
  48. package/references/domains/web-application.md +23 -0
  49. package/schemas/codeops-config.schema.json +56 -0
  50. package/scripts/check-version.mjs +163 -0
  51. package/scripts/codeops-migrate.sh +355 -0
  52. package/scripts/codeops-roadmap-compact.sh +232 -0
  53. package/scripts/codeops-roadmap-sync.sh +275 -0
  54. package/scripts/codeops_outcomes.py +155 -0
  55. package/scripts/codeops_plan.py +239 -0
  56. package/scripts/codeops_plan_migrate.py +318 -0
  57. package/scripts/codeops_worktree_snapshot.py +99 -0
  58. package/scripts/install_agents.py +288 -0
  59. package/scripts/release.mjs +533 -0
  60. package/skills/analyze-project/SKILL.md +28 -0
  61. package/skills/clean-comments/SKILL.md +22 -0
  62. package/skills/exec-plan/SKILL.md +267 -0
  63. package/skills/exec-plan/commit-modes.md +113 -0
  64. package/skills/exec-plan/execution-protocol.md +471 -0
  65. package/skills/git-commit/SKILL.md +35 -0
  66. package/skills/github-issues/SKILL.md +38 -0
  67. package/skills/grill-me/SKILL.md +342 -0
  68. package/skills/make-plan/SKILL.md +282 -0
  69. package/skills/make-plan/quality-checklist.md +96 -0
  70. package/skills/make-plan/templates.md +535 -0
  71. package/skills/make-plan/zero-ambiguity-gate.md +19 -0
  72. package/skills/make-requirements/SKILL.md +268 -0
  73. package/skills/make-requirements/discovery-phases.md +255 -0
  74. package/skills/make-requirements/review-and-add.md +73 -0
  75. package/skills/make-requirements/templates.md +296 -0
  76. package/skills/make-requirements/zero-ambiguity-gate.md +18 -0
  77. package/skills/outcome-review/SKILL.md +34 -0
  78. package/skills/preflight/SKILL.md +310 -0
  79. package/skills/preflight/dimensions.md +181 -0
  80. package/skills/preflight/report-format.md +300 -0
  81. package/skills/retro-requirements/SKILL.md +218 -0
  82. package/skills/retro-requirements/confidence-classification.md +45 -0
  83. package/skills/retro-requirements/phases.md +609 -0
  84. package/skills/retro-requirements/triage-gate.md +135 -0
  85. package/skills/roadmap/SKILL.md +381 -0
  86. package/skills/roadmap/stage-hooks.md +80 -0
  87. package/skills/roadmap/template.md +200 -0
  88. package/skills/setup-codeops/SKILL.md +94 -0
  89. package/skills/setup-codeops/migration.md +106 -0
  90. package/skills/setup-codeops/scaffold.md +99 -0
  91. package/skills/setup-routing/SKILL.md +102 -0
  92. package/skills/setup-routing/routing.md +44 -0
  93. package/skills/techdocs/SKILL.md +199 -0
  94. package/skills/techdocs/authoring-and-update.md +178 -0
  95. package/skills/techdocs/templates.md +655 -0
  96. package/skills/techdocs/vitepress-setup.md +143 -0
  97. package/skills/upgrade-plan/SKILL.md +75 -0
  98. package/skills/upgrade-plan/content-quality-gate.md +35 -0
  99. package/skills/upgrade-plan/upgrade-checklists.md +107 -0
  100. package/standards/coding-standards-full.md +124 -0
  101. package/standards/coding-standards.md +64 -0
  102. package/standards/output-style.md +17 -0
@@ -0,0 +1,198 @@
1
+ # CodeOps Layout Convention (shared reference)
2
+
3
+ > **CodeOps Artifact Schema**: 1
4
+
5
+ This is the **single source of truth** for where CodeOps artifacts live. It is a shared
6
+ reference document, **not a skill** — it lives at the **plugin root** in `_shared/` (deliberately
7
+ **outside** `skills/`), so the plugin loader, which treats each `skills/<dir>` as a skill, never
8
+ meets a `SKILL.md`-less directory. Every layout-aware skill (`roadmap`, `make-requirements`,
9
+ `make-plan`, `exec-plan`, `preflight`, `upgrade-plan`, `retro-requirements`, `techdocs`) **links here** (as
10
+ `../../_shared/layout-convention.md`) for path resolution and ID rules instead of hardcoding
11
+ paths. Change the layout in one place: here.
12
+
13
+ CodeOps supports two layouts. A repo is in exactly one of them, decided by a single marker file.
14
+
15
+ ---
16
+
17
+ ## Detection rule (apply this first, every time)
18
+
19
+ ```
20
+ 1. If codeops/.codeops.yml exists AND it declares `codeopsLayout: nested`
21
+ → NESTED layout. Resolve all artifact paths under codeops/features/<feature>/…
22
+ 2. Otherwise (no marker, or a malformed/incomplete marker)
23
+ → FLAT layout (the pre-3.0.0 behavior, unchanged). Surface a warning if a marker
24
+ exists but is malformed, then proceed as flat — never crash.
25
+ 3. In NESTED layout, <feature> is the target feature, which the skill ASKS the user to
26
+ confirm/choose — it never silently guesses (see "Feature targeting" below).
27
+ ```
28
+
29
+ Detection is a simple key match so it works without a YAML parser. The canonical check
30
+ (mirroring the JSON/`grep` fallback pattern in `scripts/validate.sh`):
31
+
32
+ ```bash
33
+ if [[ -f codeops/.codeops.yml ]] && grep -Eq '^codeopsLayout:[[:space:]]*nested[[:space:]]*$' codeops/.codeops.yml; then
34
+ layout=nested
35
+ else
36
+ layout=flat
37
+ fi
38
+ ```
39
+
40
+ ---
41
+
42
+ ## The path map (canonical)
43
+
44
+ | Concept | Flat layout (marker absent) | Nested layout (marker present) |
45
+ | ------- | --------------------------- | ------------------------------ |
46
+ | Requirements dir | `requirements/` | `codeops/features/<f>/requirements/` |
47
+ | RD document | `requirements/RD-NN-*.md` | `codeops/features/<f>/requirements/RD-NN-*.md` |
48
+ | Plan folder | `plans/<plan>/` | `codeops/features/<f>/plans/<plan>/` |
49
+ | Feature roadmap | `plans/00-roadmap.md` | `codeops/features/<f>/00-roadmap.md` |
50
+ | Portfolio roadmap | *(n/a)* | `codeops/00-roadmap.md` |
51
+ | Staged AGENTS.md notes | *(n/a)* | `codeops/features/<f>/AGENTS.notes.md` |
52
+ | Ambiguity register | `requirements/00-ambiguity-register.md` or `plans/<plan>/00-ambiguity-register.md` | the same file, under the feature |
53
+ | Scope-expansion register | See the target-qualified rules below | the same target-qualified filename, under the feature |
54
+ | Task mini-plan | `plans/<task-slug>/99-execution-plan.md` | `codeops/features/<f>/plans/<task-slug>/99-execution-plan.md` |
55
+ | Archive | `plans/_archive/<set>/` | `codeops/_archive/<f>/` |
56
+
57
+ In nested layout, a feature's inner directories are created **lazily** — only when that
58
+ feature's first RD, plan, or task is written. The marker and the (possibly empty) portfolio
59
+ roadmap are the only things `setup-codeops` creates up front.
60
+
61
+ ### Scope-expansion register paths
62
+
63
+ Scope-expansion register paths are collision-free because their authority is scoped to one audit or
64
+ planning target:
65
+
66
+ | Governed target | Register filename |
67
+ |---|---|
68
+ | Full requirements set, full plan, or plan execution | `00-scope-expansion-register.md` in the governed artifact directory |
69
+ | Single requirement or single plan document | `00-scope-expansion-register-<document-name>.md` in the document directory |
70
+ | Ad-hoc file | `scope-expansion-register-<artifact-name>.md` in the artifact directory |
71
+ | Ad-hoc directory | `scope-expansion-register.md` inside the governed directory |
72
+
73
+ The shared scope-expansion protocol owns register contents and lifecycle. This layout convention
74
+ owns paths and target identity. `<document-name>` and `<artifact-name>` are the complete existing
75
+ filename, including its extension; for example, `foo.md` and `foo.txt` resolve to different
76
+ registers. Use only the final filename component and reject path separators, `..`, absolute paths,
77
+ or a name that is not the governed target's exact filename. An ad-hoc directory's register lives
78
+ inside that exact directory. Two governed targets must never resolve to the same register unless
79
+ they intentionally share the full-set or full-plan scope baseline.
80
+
81
+ ---
82
+
83
+ ## ID rules
84
+
85
+ - **RD ids reset per feature.** Within `codeops/features/billing/requirements/` the ids run
86
+ `RD-01, RD-02, …` independently of every other feature. (In flat layout there is one global
87
+ RD sequence, as before.)
88
+ - **Cross-feature references are feature-qualified.** A plan's `00-index.md` declares one or more
89
+ requirements on a single line, for example `> **Implements**: billing/RD-01, billing/RD-02` in
90
+ nested layout or `> **Implements**: RD-01, RD-02` in flat layout. The roadmap matcher and plan
91
+ status parser read this line.
92
+ - **Tasks use a separate per-feature sequence** `T-01, T-02, …`, so a task id never collides
93
+ with an RD id in the same feature. See the task-lane spec for the lightweight task model.
94
+
95
+ ---
96
+
97
+ ## Feature targeting (nested layout)
98
+
99
+ When a layout-aware skill runs in a nested repo and the target feature is not already implied
100
+ by context (e.g. the plan/RD being operated on), the skill **asks the user which feature** to
101
+ work in, and **creates the feature folder lazily** if it is new. It never guesses.
102
+
103
+ A new feature folder is `codeops/features/<feature>/` where `<feature>` is a sanitized slug:
104
+ lowercase, words separated by `-`, no path separators, no `..`, never absolute. Reject or
105
+ normalize any candidate that would traverse outside `codeops/features/` (see Error handling).
106
+
107
+ ---
108
+
109
+ ## Archiving (nested layout)
110
+
111
+ Archiving is **feature-level and manual**: `git mv codeops/features/<f> codeops/_archive/<f>`.
112
+ The portfolio roadmap keeps a compact Archived section — the feature's row is **moved** there
113
+ (📦), not deleted. Never fragment a live feature (don't archive individual plans out of one).
114
+
115
+ ---
116
+
117
+ ## `codeops/.codeops.yml` marker schema
118
+
119
+ Minimal and flat so it parses trivially. **`setup-codeops` is the sole writer of this file**
120
+ (every other skill must leave it untouched; the `analyze-project` command may *read*
121
+ `integrationBranch`, but never writes here).
122
+
123
+ ```yaml
124
+ # CodeOps layout marker. Presence of this file opts the repo into the nested layout.
125
+ codeopsLayout: nested
126
+ layoutVersion: "3.0.0"
127
+ integrationBranch: master # branch where features integrate; analyze-project refreshes AGENTS.md here
128
+ conventions:
129
+ rdIdScope: per-feature # RD numbering resets per feature
130
+ taskIdPrefix: "T" # lightweight task ids: T-01, T-02 …
131
+ maintenanceFeature: _maintenance
132
+ archiveDir: codeops/_archive
133
+ ```
134
+
135
+ - Only `codeopsLayout: nested` is **required** for detection; the rest document the conventions
136
+ and let future versions evolve.
137
+ - `integrationBranch` is **optional** — the branch where feature work integrates and
138
+ `analyze-project` regenerates `AGENTS.md` (folding in any staged `AGENTS.notes.md`). When absent,
139
+ `analyze-project` falls back to the repo default branch (`origin/HEAD`, else `main`/`master`), so
140
+ existing markers keep working. `setup-codeops` should emit it.
141
+ - A committed sample marker lives at `scripts/fixtures/sample.codeops.yml` (used by
142
+ `validate.sh` ST-16 to assert the schema parses and carries `codeopsLayout`).
143
+
144
+ ---
145
+
146
+ ## Error handling
147
+
148
+ | Error case | Handling strategy |
149
+ | ---------- | ----------------- |
150
+ | `.codeops.yml` present but malformed / missing `codeopsLayout` | Treat as **flat** (safe default) and surface a warning; do not crash |
151
+ | Marker present but `codeops/features/` missing | Nested layout still selected; the feature dir is created lazily on first write |
152
+ | Feature/slug contains `..`, `/`, or is absolute | Reject — normalize and refuse path-traversal before using it as a path component |
153
+ | Skill cannot determine the target feature | Ask the user; never guess |
154
+
155
+ ---
156
+
157
+ ## Lightweight tasks (the task lane)
158
+
159
+ Ad-hoc work — a bugfix, chore, or small change — is **not a feature**. It is a lightweight
160
+ **task**, and its ceremony scales with size. This is the single source for the task model; the
161
+ skills (`roadmap`, `make-requirements`, `make-plan`, `exec-plan`) reference it. The lane exists
162
+ in **both layouts** (flat gained it in 3.2.0 — AR #6): nested tasks carry a per-feature `T-NN`
163
+ id; flat tasks are simply a mini-plan folder (a `T-NN` roadmap row too when a roadmap exists).
164
+
165
+ **Routing — feature or task?**
166
+
167
+ ```
168
+ Is this a new cohesive capability with real requirements?
169
+ ├─ yes → FEATURE: make-requirements (RD) → make-plan → exec-plan
170
+ └─ no → TASK (T-NN):
171
+ ├─ trivial → a roadmap row + the commit; NO plan document
172
+ └─ non-trivial → a single mini-plan, then exec-plan runs it
173
+ ```
174
+ If it is genuinely unclear, ask — never silently default to the heavy pipeline.
175
+
176
+ **Where a task lives**
177
+
178
+ - *Flat layout* → a mini-plan at `plans/<task-slug>/99-execution-plan.md` (plus a roadmap row if
179
+ `plans/00-roadmap.md` exists; a trivial task with a roadmap is just the row + commit).
180
+ - *Nested, belongs to a feature* (e.g. a bug in billing) → a `T-NN` row in
181
+ `codeops/features/billing/00-roadmap.md`.
182
+ - *Nested, standalone / cross-cutting* → a `T-NN` row in `codeops/features/_maintenance/00-roadmap.md`.
183
+ `_maintenance/` is a **normal feature folder** (same `00-roadmap.md` + `plans/`, rolls up into
184
+ the portfolio), created **lazily** on the first standalone task. It simply tends to hold tasks
185
+ rather than RDs.
186
+
187
+ **Ceremony by size**
188
+
189
+ - **Trivial** → just a roadmap row + the commit. No plan document.
190
+ - **Non-trivial** → a single mini-plan at the resolved task path (flat:
191
+ `plans/<task-slug>/99-execution-plan.md`; nested:
192
+ `codeops/features/<f>/plans/<task-slug>/99-execution-plan.md`) — execution doc only: objective,
193
+ a short task checklist, and a verify line. **No RD, no 00–07 doc set, no Zero-Ambiguity Gate.**
194
+
195
+ **Task lifecycle** — a compact subset of the stage machine: `⬜ Backlog → 🔄 Executing → ✅ Done`
196
+ (plus `⛔ Blocked` / `⏸️ Deferred` overlays). Tasks never use the RD/Plan-Preflight stages.
197
+ Specification-first ordering still applies *when a task warrants tests* (e.g. a bugfix gets a
198
+ regression test); a trivial doc/config tweak may not.
@@ -0,0 +1,134 @@
1
+ # Quality Profile (shared convention)
2
+
3
+ > **CodeOps Artifact Schema**: 1
4
+
5
+ This is the **single canonical definition** of the per-repo quality profile and the quality-agent
6
+ conventions built on it. It lives at the plugin root in `_shared/`; dispatching skills link here
7
+ instead of carrying copies.
8
+
9
+ ## Structured profile
10
+
11
+ Quality and routing configuration lives in `codeops/codeops.json`, validated against
12
+ `schemas/codeops-config.schema.json`. `AGENTS.md` contains project guidance, not mutable machine
13
+ configuration.
14
+
15
+ The quality section controls independent review and stop conditions. The routing section controls
16
+ optional roles, effort, model pins, sandboxes, and concurrency. Domain/risk tags in the active
17
+ specification select additional reviewer roles. Outcome metrics are disabled unless
18
+ `metrics.enabled` is explicitly `true`.
19
+
20
+ ### Parsing, absence, and ownership
21
+
22
+ - Missing quality configuration selects strict CodeOps defaults for planned complex-system work:
23
+ independent review on, at least one reviewer, and stop on major findings.
24
+ - Invalid structured configuration is a readiness blocker; do not silently discard malformed
25
+ policy and continue with unknown guarantees.
26
+ - setup-routing proposes and updates structured policy. Hand edits are valid and validated.
27
+ - A missing optional custom agent never disables review; use a complete dynamic packet or inline
28
+ fallback.
29
+
30
+ ## Taxonomies
31
+
32
+ ### Lens enum (7 — grow-only; renaming or repurposing an existing value is forbidden)
33
+
34
+ | Lens | Scope (one line) |
35
+ |------|------------------|
36
+ | `correctness` | Logic errors, broken behavior against the spec and tests. **Base.** |
37
+ | `maintainability` | Design-quality judgment calls: clarity, structure, duplication, naming. **Base.** |
38
+ | `standards` | Violations of the always-on written coding standards. **Base-only — never a valid profile add-on.** |
39
+ | `security` | Injection, authorization, secrets handling, unsafe input. Add-on. |
40
+ | `perf` | Hot paths, allocations, algorithmic complexity, blocking I/O. Add-on. |
41
+ | `api-surface` | Public interface design, compatibility, versioning. Add-on. |
42
+ | `concurrency` | Races, locking, ordering — explicitly **owns data-integrity**. Add-on. |
43
+
44
+ The base lenses `correctness` + `maintainability` + `standards` are always on for every review;
45
+ the profile's `lenses` list names **add-ons only**. Disambiguation: a violation of a written
46
+ standard is `standards`; a design-quality judgment with no written rule behind it is
47
+ `maintainability` — keeping the two distinguishable in findings and outcome evidence.
48
+
49
+ ### security_profile enum (5)
50
+
51
+ | Profile | Focus |
52
+ |---------|-------|
53
+ | `owasp-web` | Classic web-app risks: injection, XSS, CSRF, broken access control, SSRF |
54
+ | `auth-protocol` | Authentication/session flows: token handling, expiry, replay, fixation |
55
+ | `financial-integrity` | Money movement: idempotency, double-spend, rounding, audit trails |
56
+ | `tenant-isolation` | Multi-tenant boundaries: cross-tenant reads/writes, scoping, leakage |
57
+ | `mcp-agent` | Agent/MCP integrations: prompt injection, tool abuse, secret exfiltration |
58
+
59
+ The per-profile checklists live in `agent-templates/security-auditor.md`; this table is the naming
60
+ authority. Both enums are grow-only: adding a value here legalizes it everywhere (the structural
61
+ guards read the enums from this file).
62
+
63
+ ### Severity
64
+
65
+ Findings reuse the preflight severity scale **by reference** — 🔴 CRITICAL / 🟠 MAJOR /
66
+ 🟡 MINOR as defined in the preflight skill — verbatim, with no extra levels.
67
+
68
+ ### Finding prefixes
69
+
70
+ RV (phase-reviewer) · SA (security-auditor) · PA (preflight-auditor) · PE (perf-auditor), each
71
+ numbered `XX-NNN`. Every finding-producing agent reports "no findings" explicitly rather than
72
+ returning empty output.
73
+
74
+ ## Activation & supersession
75
+
76
+ | Condition | Effect |
77
+ |-----------|--------|
78
+ | No structured profile | Strict defaults: one correctness review for every non-trivial executed phase |
79
+ | `quality.independentReview: false` | Allowed only outside strict mode; announce that routine independent review is disabled. This never disables a required Complexity Escalation Gate challenger after an escalation is detected. |
80
+ | Independent review on | Post-phase quality review runs for **all executed phases and task mini-plans** (whole-task diff); trivial tasks are never reviewed |
81
+ | Docs-only diff | Phase reviewer still runs; security/perf auditors skip — the skip is logged, never silent |
82
+ | Security risk tags active | Security auditor dispatches once per phase with the union of applicable checklists and supersedes the reviewer's security lens |
83
+ | Performance-critical tag + code diff | Performance auditor dispatches and supersedes the reviewer's performance lens |
84
+
85
+ Supersession exists so the same ground is never reviewed twice at different depths: a dedicated
86
+ agent replaces the reviewer's matching add-on lens for that phase.
87
+
88
+ ## Dispatch packets & header
89
+
90
+ **Line 1 of every quality-agent dispatch prompt** is a compact scope header:
91
+
92
+ ```
93
+ [codeops-dispatch agent=<name> feature=<slug> phase=<id>]
94
+ ```
95
+
96
+ The header makes parallel results attributable and auditable. Outcome metrics never parse prompt
97
+ or response content.
98
+
99
+ | Agent | Packet contents (the agent receives nothing else and must need nothing else) |
100
+ |-------|------------------------------------------------------------------------------|
101
+ | phase-reviewer, security-auditor, perf-auditor | Phase diff (`git diff <phase-start-ref>..HEAD`), original goal + smallest viable design, relevant approved complexity AR/PF/RV excerpts, the phase's task + Deliverable lines, active lenses, scope mode (`strict` or `explore`), confirmed scope baseline, profile excerpt, verify command + last result |
102
+ | plan-task-executor, plan-task-executor-opus | Phase task + Deliverable + Verify lines, governing spec/ST/AR excerpts, original goal + smallest viable design, relevant approved complexity PF/RV excerpts, target paths, scope mode (`strict` or `explore`), confirmed scope baseline, verify command |
103
+ | spec-test-author | Spec excerpts + test cases, planned interface signatures from the plan documents, test framework/conventions, the FORBIDDEN implementation-file list, verify command (expected RED) |
104
+ | preflight-auditor | The artifact under audit + ONE assigned dimension cluster + original goal + smallest viable design + relevant approved complexity AR/PF/RV excerpts + scope mode (`strict` or `explore`) + confirmed scope baseline |
105
+ | design-challenger | Problem + candidate options, **without** the parent's preferred choice (per `_shared/recommendation-hardening.md`) |
106
+ | codebase-scout | The factual questions, search hints, and the facts-only contract |
107
+
108
+ ## Budget caps
109
+
110
+ - **Preflight fan-out:** ~5 clustered auditor dispatches — an exact partition of the preflight
111
+ skill's 13 dimensions, grouped by affinity: ① Ambiguities + Logical Contradictions +
112
+ Consistency (document soundness) · ② Implicit Assumptions + Codebase Alignment (grounding) ·
113
+ ③ Completeness Gaps + Dependency Issues + Ordering & Sequencing (delivery) · ④ Security Blind
114
+ Spots + Edge Cases + Feasibility Concerns (risk) · ⑤ Testability + Scope Creep Indicators
115
+ (fit). `--thorough` expands to one dispatch per dimension.
116
+ - **Re-review:** at most ONE re-review per phase, only after 🔴/🟠 fixes, scoped to the fix
117
+ diff — never a third pass.
118
+ - **Scout:** ≤3 codebase-scout dispatches per skill run, enforced by the dispatching parent.
119
+ - **Challenger:** caps live in `_shared/recommendation-hardening.md` and apply unchanged.
120
+
121
+ ## Model, effort, and agent resolution
122
+
123
+ Routing policy lives in `codeops/codeops.json`; see the setup-routing skill. Policy names roles and capabilities, not vendor tiers. A role may optionally declare a current OpenCode model, reasoning effort, and sandbox, but no gate depends on a particular model being available.
124
+
125
+ Resolution order is:
126
+
127
+ 1. an explicit model/effort requested for the current dispatch;
128
+ 2. a generated or hand-authored project agent in `.opencode/agents/<role>.md`;
129
+ 3. project `[agents]` defaults in `opencode.json`;
130
+ 4. the parent session's model and effort.
131
+
132
+ Use `python3 "${CODEOPS_PLUGIN_ROOT}/scripts/install_agents.py" --project . --roles ...` to create optional project agents. Generated agent files carry a CodeOps marker. The installer owns only marked files and preserves every hand-authored file. Use `--check` to detect missing or stale generated agents and `--dry-run` to preview changes.
133
+
134
+ Dynamic packets are the correctness baseline. If a named agent is missing or a model pin is unavailable, spawn a generic subagent with the complete packet or run inline. Report the fallback and preserve required reviewer independence, sandbox intent, and every ambiguity/readiness/verification gate.
@@ -0,0 +1,166 @@
1
+ # Recommendation Hardening Protocol
2
+
3
+ > Shared reference for the **Grounded Options & Recommendations** directive
4
+ > (`standards/coding-standards.md` → Working style). Linked by the recommendation-producing skills.
5
+ > **CodeOps Artifact Schema**: 1
6
+
7
+ This protocol makes recommendations trustworthy on the **first** pass. It is the standing answer to
8
+ the question *"are these your best possible recommendations?"* — asked and answered **before** you
9
+ present, every time, so the operator never has to ask it.
10
+
11
+ ## Why this exists
12
+
13
+ Left unstructured, a model **satisfices**: it returns the first adequate set of options drawn from
14
+ "what does a reasonable answer look like," not "what is the best answer after exhausting the space."
15
+ The Grounded Options *second-guess* step, run in the same forward pass that produced the options,
16
+ shares that pass's framing and blind spots — so it rarely finds the better answer that an external
17
+ challenge would.
18
+
19
+ When a human challenges the recommendation, two different things can happen, and only one is good:
20
+
21
+ 1. **Genuine deeper search** — a fresh pass takes the first pass as adversarial input, sees its gaps,
22
+ and finds a better answer. *This is the lift we want.*
23
+ 2. **Sycophantic drift** — the model produces a *different* answer because it read the challenge as
24
+ "you were wrong," not because the new answer is better. *This is noise, and it is corrosive to
25
+ trust.*
26
+
27
+ > **Convergence, not drift.** The goal of this protocol is to institutionalize (1) and design out
28
+ > (2). You do not "try harder when poked" — you run a **structured** hardening pass that *converges*
29
+ > on a verified-best recommendation. Reflexively changing an answer under perceived pressure is a
30
+ > failure of this protocol, not a success of it.
31
+
32
+ ## When it applies
33
+
34
+ Apply the full protocol to every **consequential** recommendation: code-modifying directions,
35
+ architecture and design choices, scope decisions, defect-resolution directions, requirements
36
+ choices, plan-making and plan-execution decisions, and preflight findings.
37
+
38
+ > **Proportionality.** Trivial, easily-reversible, or obvious choices get a one-line recommendation
39
+ > and skip the ceremony — drowning the operator in hardening theater wastes their time as surely as
40
+ > strawman options do. Match the ceremony to the stakes.
41
+
42
+ ## Layer 1 — Forced reframing
43
+
44
+ Before presenting a consequential recommendation, generate your candidate set, then **answer these
45
+ four prompts** (internally for ordinary decisions; surfaced briefly for high-stakes ones). They are
46
+ the forcing functions that break you out of the obvious framing:
47
+
48
+ - **10× budget:** "What would I recommend with **10× the time/budget**?"
49
+ - **Contrarian expert:** "What's the option a **contrarian senior expert** would push that I
50
+ dismissed or didn't consider?"
51
+ - **Obsolescence:** "What would make my current top pick **obsolete or wrong**?"
52
+ - **Pre-empt the challenge:** "If the operator asks *'is this your best?'* — **answer that now.**
53
+ What changes?"
54
+
55
+ Revise your candidate set with whatever these surface *before* moving on.
56
+
57
+ ## Layer 2 — Definition-of-done rubric
58
+
59
+ A consequential recommendation may **not** be presented until **all** of these hold. This is the
60
+ **definition-of-done** for a recommendation — it operationalizes "best" so it is a checklist, not a
61
+ vibe:
62
+
63
+ - [ ] **≥1 genuinely non-obvious option considered** — never a strawman. (When only one option is
64
+ genuinely viable, that is a valid outcome: present it alone, say it is the only viable one, and
65
+ name what you rejected and why.)
66
+ - [ ] **Confidence level set** — High / Med / Low, with the specific thing that would change it.
67
+ - [ ] **Strongest counter-argument named** — the best case *against* your chosen pick, stated in one
68
+ line.
69
+
70
+ ## Layer 3 — Confidence & trust disclosure (conditional)
71
+
72
+ The disclosure exists to surface **residual uncertainty** — so it appears only when there is some:
73
+
74
+ ```text
75
+ Confidence: High | Med | Low — <the specific thing that would change this>
76
+ Hardening: <what the deeper pass changed, or "no change">
77
+ ```
78
+
79
+ The disclosure is REQUIRED when any of these hold:
80
+ - confidence is **Med or Low**;
81
+ - the hardening pass **changed something** (an option added/dropped, the pick revised);
82
+ - the recommendation is **high-stakes** (see below).
83
+
84
+ Otherwise — High confidence, hardening changed nothing, not high-stakes — **omit the disclosure**:
85
+ a boilerplate "Hardening: no change" at High confidence carries no information and trains the
86
+ operator to skip the lines that matter. (The layers 1–2 work still runs; only the two-line
87
+ disclosure is conditional.)
88
+
89
+ For a **high-stakes** recommendation, also state the challenger's verdict:
90
+ `Challenger: converged` or `Challenger: diverged — <how>`.
91
+
92
+ This disclosure is **presentation-layer only**. It may appear in a saved artifact (e.g. a preflight
93
+ report), but it is **not** a newly-required, validated field — existing artifacts remain valid
94
+ without it, and no migration is implied.
95
+
96
+ ## Layer 4 — Tiered independent challenger
97
+
98
+ The independence layer is what actually closes the *trust* gap, because a self-critique in the same
99
+ context inherits the same blind spots. Run it **tiered by stakes**:
100
+
101
+ - **Always (every consequential recommendation):** Layers 1–3, in-context.
102
+ - **High-stakes only:** additionally spawn **one independent challenger** subagent.
103
+
104
+ ### Challenger budget (hard caps)
105
+
106
+ - **One challenger per preflight scan, not per finding.** A preflight run with multiple
107
+ CRITICAL/MAJOR findings spawns a SINGLE challenger that receives the whole finding batch
108
+ (each finding's statement + surviving options) **plus the scan's recon summary / Codebase
109
+ Context section**, so it challenges from evidence rather than from a cold start. It returns a
110
+ per-finding verdict list.
111
+ - **At most 2 challenger spawns per skill run**, total — e.g. one for the finding batch and one
112
+ for a late-discovered decision. Beyond the cap, proceed with Layers 1–3 and disclose
113
+ `Challenger: budget exhausted` (cap confidence at Med for those items).
114
+
115
+ ### The challenger mechanism
116
+
117
+ 1. Dispatch **one** independent `design-challenger` role in a fresh subagent context. Give it the
118
+ problem statement and surviving options — but **NOT** your chosen pick. Packet convention:
119
+ `_shared/quality-profile.md`. A generated custom agent is optional; a generic subagent with the
120
+ complete challenger contract is equivalent.
121
+ 2. Instruct it to produce **its own** best recommendation and the strongest case for it, blind to
122
+ your choice. Challenger prompt template:
123
+
124
+ > *"Here is a decision and the candidate options: <problem + options>. Independently recommend the
125
+ > single best option and give the strongest grounded case for it. You do not know my current pick.
126
+ > If a better option is missing from the list, propose it. State your confidence and the strongest
127
+ > argument against your own pick."*
128
+
129
+ 3. **Reconcile** the challenger's recommendation with yours:
130
+ - **Converged** → present with raised confidence; note `Challenger: converged`.
131
+ - **Diverged** → present **both**, with a grounded reconciliation of why you land where you do;
132
+ note `Challenger: diverged — <how>`. Never silently overwrite either side.
133
+
134
+ > **Fallback.** If the challenger subagent is unavailable or returns nothing usable, proceed with
135
+ > Layers 1–3, disclose `Challenger: unavailable`, and **cap confidence at Med**. Exception: a
136
+ > detected complexity escalation cannot proceed. The Complexity Escalation Gate requires an
137
+ > independent verdict, so it stays blocked when the challenger is unavailable or its budget is
138
+ > exhausted.
139
+
140
+ ## High-stakes definition (the escalation trigger)
141
+
142
+ A recommendation is **high-stakes** — and therefore gets the challenger — when any of these hold:
143
+
144
+ - it is a **preflight** finding at **CRITICAL/MAJOR** severity (challenged as a batch — one
145
+ challenger per preflight scan, see the budget above); or
146
+ - it is a **make-plan** (Phase 1C) or **make-requirements** (Phase 2B) gate decision tagged
147
+ **complex/sensitive** (the project routing tags); or
148
+ - it triggers the shared **Complexity Escalation Gate**, regardless of routing tag; or
149
+ - the **user explicitly requests a challenger** for a decision, in any skill.
150
+
151
+ Everything else — minor/observation findings, trivial/standard-tagged decisions, ad-hoc choices —
152
+ gets Layers 1–3 only. When a gate decision carries no routing tag, treat it as `standard` (the
153
+ documented default) → no challenger. Skills without their own trigger (grill-me, exec-plan, the
154
+ wrappers) escalate ONLY via this definition — they carry no private one.
155
+
156
+ ## Anti-patterns
157
+
158
+ - ❌ Changing your recommendation reflexively because you were questioned (drift, not convergence).
159
+ - ❌ Presenting >2 options without filtering, or padding with strawmen to manufacture a choice.
160
+ - ❌ Skipping the confidence/`Hardening:` disclosure where Layer 3 requires it (Med/Low
161
+ confidence, a changed pick, or high stakes) — or padding every answer with a contentless
162
+ "Hardening: no change" where it doesn't.
163
+ - ❌ Spawning a challenger for a trivial or easily-reversible choice (ceremony without stakes).
164
+ - ❌ Spawning per-finding challengers when the batch rule applies, or exceeding the 2-spawn cap.
165
+ - ❌ Giving the challenger your chosen pick (destroys its independence).
166
+ - ❌ Deriving "best" from how much effort you spent rather than from the definition-of-done rubric.