@mmerterden/multi-agent-pipeline 12.7.0 → 12.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +126 -0
- package/install/_common.mjs +48 -0
- package/install/_dev-only-files.mjs +125 -5
- package/install/claude.mjs +14 -8
- package/install/copilot.mjs +5 -8
- package/package.json +17 -2
- package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
- package/pipeline/lib/credential-store.sh +20 -0
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
- package/pipeline/multi-agent-refs/phases/operations.md +28 -0
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +49 -4
- package/pipeline/schemas/prefs.schema.json +6 -0
- package/pipeline/scripts/_smoke-root.sh +61 -0
- package/pipeline/scripts/audit-log.sh +25 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
- package/pipeline/eval/golden-tasks/README.md +0 -65
- package/pipeline/eval/intent-cases.json +0 -40
- package/pipeline/eval/run-metrics-fixture.json +0 -93
- package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
- package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
- package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
- package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
- package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
- package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
- package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
- package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
- package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
- package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
- package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
- package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
- package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
- package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
- package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
- package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
- package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
- package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
- package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
- package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
- package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
- package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
- package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
- package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
- package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
- package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
- package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
- package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
- package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
- package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
- package/pipeline/eval/triage/README.md +0 -54
- package/pipeline/scripts/benchmark-phase-0.sh +0 -128
- package/pipeline/scripts/check-md-links.mjs +0 -88
- package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
- package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
- package/pipeline/scripts/eval-intent.mjs +0 -107
- package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
- package/pipeline/scripts/eval-triage.mjs +0 -171
- package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
- package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
- package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
- package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
- package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
- package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
- package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
- package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
- package/pipeline/scripts/lint-skills.mjs +0 -154
- package/pipeline/scripts/run-smokes.mjs +0 -130
- package/pipeline/scripts/scorecard.mjs +0 -258
- package/pipeline/scripts/smoke-add-detail.sh +0 -137
- package/pipeline/scripts/smoke-agent-guard.sh +0 -74
- package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
- package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
- package/pipeline/scripts/smoke-ask-choice.sh +0 -42
- package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
- package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
- package/pipeline/scripts/smoke-changelog-version.sh +0 -47
- package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
- package/pipeline/scripts/smoke-channels-flow.sh +0 -130
- package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
- package/pipeline/scripts/smoke-clarify.sh +0 -148
- package/pipeline/scripts/smoke-command-inventory.sh +0 -81
- package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
- package/pipeline/scripts/smoke-community-gates.sh +0 -75
- package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
- package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
- package/pipeline/scripts/smoke-context-budget.sh +0 -72
- package/pipeline/scripts/smoke-cost-budget.sh +0 -70
- package/pipeline/scripts/smoke-cost-summary.sh +0 -139
- package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
- package/pipeline/scripts/smoke-description-tr.sh +0 -82
- package/pipeline/scripts/smoke-dev-critic.sh +0 -144
- package/pipeline/scripts/smoke-diff-explain.sh +0 -147
- package/pipeline/scripts/smoke-diff-risk.sh +0 -190
- package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
- package/pipeline/scripts/smoke-eval-live.sh +0 -136
- package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
- package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
- package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
- package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
- package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
- package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
- package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
- package/pipeline/scripts/smoke-generate-issue.sh +0 -120
- package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
- package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
- package/pipeline/scripts/smoke-install-layout.sh +0 -248
- package/pipeline/scripts/smoke-intent-guard.sh +0 -86
- package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
- package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
- package/pipeline/scripts/smoke-keychain.sh +0 -158
- package/pipeline/scripts/smoke-language-axis.sh +0 -109
- package/pipeline/scripts/smoke-learning-curve.sh +0 -61
- package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
- package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
- package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
- package/pipeline/scripts/smoke-md-links.sh +0 -8
- package/pipeline/scripts/smoke-md2confluence.sh +0 -126
- package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
- package/pipeline/scripts/smoke-migrate-state.sh +0 -102
- package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
- package/pipeline/scripts/smoke-model-fallback.sh +0 -89
- package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
- package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
- package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
- package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
- package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
- package/pipeline/scripts/smoke-pack-contents.sh +0 -140
- package/pipeline/scripts/smoke-pat-audit.sh +0 -128
- package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
- package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
- package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
- package/pipeline/scripts/smoke-phase-banner.sh +0 -101
- package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
- package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
- package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
- package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
- package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
- package/pipeline/scripts/smoke-plan-safety.sh +0 -139
- package/pipeline/scripts/smoke-plan-todos.sh +0 -196
- package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
- package/pipeline/scripts/smoke-pre-commit.sh +0 -170
- package/pipeline/scripts/smoke-pref-migration.sh +0 -226
- package/pipeline/scripts/smoke-prefs-language.sh +0 -134
- package/pipeline/scripts/smoke-progress-contract.sh +0 -127
- package/pipeline/scripts/smoke-prune-logs.sh +0 -137
- package/pipeline/scripts/smoke-purge.sh +0 -138
- package/pipeline/scripts/smoke-push-retry.sh +0 -75
- package/pipeline/scripts/smoke-repo-map.sh +0 -300
- package/pipeline/scripts/smoke-review-readiness.sh +0 -92
- package/pipeline/scripts/smoke-review-watch.sh +0 -146
- package/pipeline/scripts/smoke-routines.sh +0 -84
- package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
- package/pipeline/scripts/smoke-run-metrics.sh +0 -50
- package/pipeline/scripts/smoke-search.sh +0 -187
- package/pipeline/scripts/smoke-shadow-git.sh +0 -224
- package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
- package/pipeline/scripts/smoke-skill-language.sh +0 -83
- package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
- package/pipeline/scripts/smoke-skill-scan.sh +0 -198
- package/pipeline/scripts/smoke-source-parity.sh +0 -85
- package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
- package/pipeline/scripts/smoke-sync-parity.sh +0 -92
- package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
- package/pipeline/scripts/smoke-telemetry.sh +0 -147
- package/pipeline/scripts/smoke-test-gap.sh +0 -183
- package/pipeline/scripts/smoke-token-budget.sh +0 -67
- package/pipeline/scripts/smoke-token-preflight.sh +0 -82
- package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
- package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
- package/pipeline/scripts/smoke-triage-memory.sh +0 -174
- package/pipeline/scripts/smoke-update-check.sh +0 -135
- package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
- package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
- package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
- package/pipeline/scripts/smoke-validator-gates.sh +0 -164
- package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
- package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
- package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
- package/pipeline/scripts/smoke-work-summary.sh +0 -163
- package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
- package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
- package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
- package/pipeline/scripts/smoke-write-state.sh +0 -159
- package/pipeline/scripts/sync-parity-check.sh +0 -135
- package/pipeline/scripts/test-gap-rules/android.json +0 -25
- package/pipeline/scripts/test-gap-rules/ios.json +0 -34
- package/pipeline/scripts/test-gap-rules/node.json +0 -29
- package/pipeline/scripts/test-gap-rules/python.json +0 -25
- package/pipeline/scripts/validate-schemas.mjs +0 -88
|
@@ -1,147 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-diff-explain.sh - v7.8.0 Paket B: Phase 4 triage to diff bridge.
|
|
3
|
-
#
|
|
4
|
-
# Validates:
|
|
5
|
-
# 1. Slash command + Copilot peer present (parity smoke also catches this)
|
|
6
|
-
# 2. Empty triage produces a "no findings" stub without crashing
|
|
7
|
-
# 3. Mixed-classification fixture renders all bucket icons (✅ ⏭️ ❌)
|
|
8
|
-
# 4. Severity icons (🚫 ⚠️ 💡) render correctly
|
|
9
|
-
# 5. Hunk binding: synthetic diff + matching line → diff fence in output
|
|
10
|
-
# 6. Out-of-diff line → fallback message (no crash)
|
|
11
|
-
# 7. --help works without inputs
|
|
12
|
-
# 8. Missing inputs exit 1 with a clear error
|
|
13
|
-
# 9. Read-only contract: script does NOT call git apply / git checkout / write APIs
|
|
14
|
-
|
|
15
|
-
set -uo pipefail
|
|
16
|
-
|
|
17
|
-
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
18
|
-
SCRIPT="$REPO_ROOT/pipeline/scripts/diff-explain.mjs"
|
|
19
|
-
|
|
20
|
-
PASS=0
|
|
21
|
-
FAIL=0
|
|
22
|
-
pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
|
|
23
|
-
fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
|
|
24
|
-
|
|
25
|
-
cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
|
|
26
|
-
trap cleanup EXIT
|
|
27
|
-
TMP=$(mktemp -d)
|
|
28
|
-
|
|
29
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
30
|
-
echo "→ 1. Command + skill files exist"
|
|
31
|
-
[ -f "$REPO_ROOT/pipeline/commands/multi-agent/diff-explain/SKILL.md" ] && pass "diff-explain.md present" || fail "diff-explain.md missing"
|
|
32
|
-
[ -f "$REPO_ROOT/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md" ] && pass "multi-agent-diff-explain SKILL.md present" || fail "Copilot peer missing"
|
|
33
|
-
[ -x "$SCRIPT" ] && pass "diff-explain.mjs is executable" || fail "diff-explain.mjs not executable"
|
|
34
|
-
|
|
35
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
36
|
-
echo "→ 2. Empty findings stub"
|
|
37
|
-
cat > "$TMP/empty.json" <<'EOF'
|
|
38
|
-
{"accepted": [], "deferred": [], "rejected": [], "approved": true}
|
|
39
|
-
EOF
|
|
40
|
-
OUT=$(node "$SCRIPT" --triage "$TMP/empty.json" --diff /dev/null 2>&1)
|
|
41
|
-
if echo "$OUT" | grep -q "no findings to map"; then
|
|
42
|
-
pass "empty triage → 'no findings' stub"
|
|
43
|
-
else
|
|
44
|
-
fail "empty triage did not produce stub message"
|
|
45
|
-
fi
|
|
46
|
-
|
|
47
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
48
|
-
echo "→ 3. Bucket icons render correctly"
|
|
49
|
-
OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff /dev/null 2>&1)
|
|
50
|
-
echo "$OUT" | grep -q "✅" && pass "accepted bucket icon (✅) rendered" || fail "accepted icon missing"
|
|
51
|
-
echo "$OUT" | grep -q "⏭️" && pass "deferred bucket icon (⏭️) rendered" || fail "deferred icon missing"
|
|
52
|
-
|
|
53
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
54
|
-
echo "→ 4. Severity icons render"
|
|
55
|
-
echo "$OUT" | grep -q "🚫" && pass "blocking severity icon (🚫) rendered" || fail "blocking icon missing"
|
|
56
|
-
echo "$OUT" | grep -q "⚠️" && pass "important severity icon (⚠️) rendered" || fail "important icon missing"
|
|
57
|
-
echo "$OUT" | grep -q "💡" && pass "suggestion severity icon (💡) rendered" || fail "suggestion icon missing"
|
|
58
|
-
|
|
59
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
60
|
-
echo "→ 5. Hunk binding - synthetic diff matches finding line"
|
|
61
|
-
cat > "$TMP/syn.diff" <<'EOF'
|
|
62
|
-
diff --git a/src/api/checkout.ts b/src/api/checkout.ts
|
|
63
|
-
index a..b 100644
|
|
64
|
-
--- a/src/api/checkout.ts
|
|
65
|
-
+++ b/src/api/checkout.ts
|
|
66
|
-
@@ -98,5 +98,7 @@ ...
|
|
67
|
-
const total = sum(items);
|
|
68
|
-
+ const total2 = sum(items);
|
|
69
|
-
+ // duplicate
|
|
70
|
-
await chargeCard(total);
|
|
71
|
-
EOF
|
|
72
|
-
OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff "$TMP/syn.diff" 2>&1)
|
|
73
|
-
if echo "$OUT" | grep -q '```diff'; then
|
|
74
|
-
pass "hunk fenced diff block rendered"
|
|
75
|
-
else
|
|
76
|
-
fail "no diff code fence in output"
|
|
77
|
-
fi
|
|
78
|
-
if echo "$OUT" | grep -q "@@ -98,5 +98,7 @@"; then
|
|
79
|
-
pass "hunk header preserved in output"
|
|
80
|
-
else
|
|
81
|
-
fail "hunk header missing in output"
|
|
82
|
-
fi
|
|
83
|
-
|
|
84
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
85
|
-
echo "→ 6. Out-of-diff line fallback"
|
|
86
|
-
# checkout.ts is in the diff but its lines 145, 200 are NOT covered by the
|
|
87
|
-
# single hunk above. Should produce fallback messages, not crash.
|
|
88
|
-
fallback_count=$(echo "$OUT" | grep -c "line not in diff" || true)
|
|
89
|
-
if [ "$fallback_count" -ge 1 ]; then
|
|
90
|
-
pass "out-of-diff lines produce fallback message ($fallback_count occurrences)"
|
|
91
|
-
else
|
|
92
|
-
# Acceptable alternative: hunk renderer outputs nothing visible. Both options OK.
|
|
93
|
-
pass "out-of-diff lines handled (no crash)"
|
|
94
|
-
fi
|
|
95
|
-
|
|
96
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
97
|
-
echo "→ 7. --help works without inputs"
|
|
98
|
-
HELP=$(node "$SCRIPT" --help 2>&1)
|
|
99
|
-
if echo "$HELP" | grep -q "Usage:"; then
|
|
100
|
-
pass "--help prints usage"
|
|
101
|
-
else
|
|
102
|
-
fail "--help missing Usage block"
|
|
103
|
-
fi
|
|
104
|
-
|
|
105
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
106
|
-
echo "→ 8. Missing inputs exit non-zero"
|
|
107
|
-
set +e
|
|
108
|
-
node "$SCRIPT" 2>"$TMP/err.txt"
|
|
109
|
-
EC=$?
|
|
110
|
-
set -e
|
|
111
|
-
if [ "$EC" -ne 0 ] && grep -q "required" "$TMP/err.txt"; then
|
|
112
|
-
pass "no-input exit $EC with 'required' error"
|
|
113
|
-
else
|
|
114
|
-
fail "no-input did not error correctly (exit=$EC)"
|
|
115
|
-
fi
|
|
116
|
-
|
|
117
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
118
|
-
echo "→ 9. Read-only contract - no destructive git APIs"
|
|
119
|
-
if grep -qE "git apply|git checkout|git reset|git rebase|git push|git commit|writeFileSync" "$SCRIPT"; then
|
|
120
|
-
fail "diff-explain.mjs references destructive git API or writeFileSync (read-only contract violated)"
|
|
121
|
-
else
|
|
122
|
-
pass "no destructive git/write API references in diff-explain.mjs"
|
|
123
|
-
fi
|
|
124
|
-
|
|
125
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
126
|
-
echo "→ 10. Injection-safe - shell metacharacters in --base are not executed"
|
|
127
|
-
SENTINEL="$TMP/PWNED"
|
|
128
|
-
set +e
|
|
129
|
-
node "$SCRIPT" \
|
|
130
|
-
--triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" \
|
|
131
|
-
--base "main; touch $SENTINEL" --branch HEAD >/dev/null 2>&1
|
|
132
|
-
set -e
|
|
133
|
-
if [ -f "$SENTINEL" ]; then
|
|
134
|
-
fail "command injection: --base executed arbitrary shell (sentinel created)"
|
|
135
|
-
else
|
|
136
|
-
pass "shell metacharacters in --base not executed (execFileSync)"
|
|
137
|
-
fi
|
|
138
|
-
if grep -q "execSync" "$SCRIPT"; then
|
|
139
|
-
fail "diff-explain.mjs still uses execSync (shell interpolation risk)"
|
|
140
|
-
else
|
|
141
|
-
pass "diff-explain.mjs uses execFileSync only"
|
|
142
|
-
fi
|
|
143
|
-
|
|
144
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
145
|
-
echo ""
|
|
146
|
-
echo "══ diff-explain smoke: $PASS passed, $FAIL failed ══"
|
|
147
|
-
[ "$FAIL" -eq 0 ]
|
|
@@ -1,190 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-diff-risk.sh - v8.3.0
|
|
3
|
-
#
|
|
4
|
-
# Verifies the Phase 4 advisory diff-risk pipeline:
|
|
5
|
-
# 1. diff-risk-score.mjs runs against the iOS fixture and produces valid JSON
|
|
6
|
-
# 2. KeychainStore (security path) ranks first
|
|
7
|
-
# 3. Tests file ranks last (no security/public_api/migration signal)
|
|
8
|
-
# 4. validate-diff-risk.mjs accepts the output, rejects malformed
|
|
9
|
-
# 5. Android fixture: migration outranks AuthRepository outranks HomeScreen
|
|
10
|
-
# 6. --top truncation honored
|
|
11
|
-
# 7. Empty diff returns exit 2
|
|
12
|
-
# 8. phase-4-review.md ref doc declares Step 1.75 + diff-risk-score.mjs
|
|
13
|
-
# 9. code-reviewer.md agent template carries the priority-files placeholder
|
|
14
|
-
# 10. prefs.schema.json exposes diffRisk advisory toggle
|
|
15
|
-
# 11. test-removal fixture fires the test_lines_removed signal (v1.1.0)
|
|
16
|
-
#
|
|
17
|
-
# Exit 0 = all pass, 1 = any failure.
|
|
18
|
-
|
|
19
|
-
set -euo pipefail
|
|
20
|
-
|
|
21
|
-
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
22
|
-
SCORE="$ROOT/pipeline/scripts/diff-risk-score.mjs"
|
|
23
|
-
VALIDATE="$ROOT/pipeline/scripts/validate-diff-risk.mjs"
|
|
24
|
-
SCHEMA="$ROOT/pipeline/schemas/diff-risk.schema.json"
|
|
25
|
-
PHASE4="$ROOT/pipeline/multi-agent-refs/phases/phase-4-review.md"
|
|
26
|
-
REVIEWER="$ROOT/pipeline/agents/code-reviewer.md"
|
|
27
|
-
PREFS="$ROOT/pipeline/schemas/prefs.schema.json"
|
|
28
|
-
FIX_IOS="$ROOT/pipeline/scripts/fixtures/diff-risk-ios.diff"
|
|
29
|
-
FIX_AND="$ROOT/pipeline/scripts/fixtures/diff-risk-android.diff"
|
|
30
|
-
FIX_TESTRM="$ROOT/pipeline/scripts/fixtures/diff-risk-test-removal.diff"
|
|
31
|
-
|
|
32
|
-
pass=0
|
|
33
|
-
fail=0
|
|
34
|
-
failures=()
|
|
35
|
-
record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
|
|
36
|
-
record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
|
|
37
|
-
|
|
38
|
-
printf '→ smoke-diff-risk (v8.3.0): pre-review risk scoring contract\n'
|
|
39
|
-
|
|
40
|
-
[ -f "$SCHEMA" ] || { record_fail "schema missing: $SCHEMA"; exit 1; }
|
|
41
|
-
[ -f "$FIX_IOS" ] || { record_fail "fixture missing: $FIX_IOS"; exit 1; }
|
|
42
|
-
[ -f "$FIX_AND" ] || { record_fail "fixture missing: $FIX_AND"; exit 1; }
|
|
43
|
-
[ -f "$FIX_TESTRM" ] || { record_fail "fixture missing: $FIX_TESTRM"; exit 1; }
|
|
44
|
-
|
|
45
|
-
# --- 1: iOS fixture produces JSON ---
|
|
46
|
-
out_ios=$(node "$SCORE" --diff "$FIX_IOS" 2>/dev/null)
|
|
47
|
-
if jq -e '.schemaVersion == "1.1.0"' <<< "$out_ios" >/dev/null 2>&1; then
|
|
48
|
-
record_pass "iOS fixture renders schema-versioned JSON"
|
|
49
|
-
else
|
|
50
|
-
record_fail "iOS fixture JSON malformed or missing schemaVersion"
|
|
51
|
-
fi
|
|
52
|
-
|
|
53
|
-
# --- 2: KeychainStore ranks first ---
|
|
54
|
-
top1=$(jq -r '.files[0].path' <<< "$out_ios")
|
|
55
|
-
if [ "$top1" = "MyApp/Sources/Auth/KeychainStore.swift" ]; then
|
|
56
|
-
record_pass "iOS top-1 = KeychainStore (security_path + public_api + no_test)"
|
|
57
|
-
else
|
|
58
|
-
record_fail "iOS top-1 should be KeychainStore, got: $top1"
|
|
59
|
-
fi
|
|
60
|
-
|
|
61
|
-
# --- 3: Test file ranks last ---
|
|
62
|
-
last=$(jq -r '.files[-1].path' <<< "$out_ios")
|
|
63
|
-
if [ "$last" = "MyAppTests/SettingsViewTests.swift" ]; then
|
|
64
|
-
record_pass "iOS last = test file (no risk signals beyond loc_changed)"
|
|
65
|
-
else
|
|
66
|
-
record_fail "iOS last should be test file, got: $last"
|
|
67
|
-
fi
|
|
68
|
-
|
|
69
|
-
# --- 4: Validator accepts valid output, rejects malformed ---
|
|
70
|
-
set +e
|
|
71
|
-
echo "$out_ios" | node "$VALIDATE" - >/dev/null 2>&1
|
|
72
|
-
rc_ok=$?
|
|
73
|
-
echo '{"schemaVersion":"0.0.0","files":[]}' | node "$VALIDATE" - >/dev/null 2>&1
|
|
74
|
-
rc_bad=$?
|
|
75
|
-
set -e
|
|
76
|
-
if [ "$rc_ok" -eq 0 ]; then
|
|
77
|
-
record_pass "validator accepts valid output"
|
|
78
|
-
else
|
|
79
|
-
record_fail "validator rejected valid output (rc=$rc_ok)"
|
|
80
|
-
fi
|
|
81
|
-
if [ "$rc_bad" -ne 0 ]; then
|
|
82
|
-
record_pass "validator rejects malformed input"
|
|
83
|
-
else
|
|
84
|
-
record_fail "validator should reject malformed (got rc=$rc_bad)"
|
|
85
|
-
fi
|
|
86
|
-
|
|
87
|
-
# --- 5: Android ranking ---
|
|
88
|
-
out_and=$(node "$SCORE" --diff "$FIX_AND" 2>/dev/null)
|
|
89
|
-
top1=$(jq -r '.files[0].path' <<< "$out_and")
|
|
90
|
-
top2=$(jq -r '.files[1].path' <<< "$out_and")
|
|
91
|
-
top3=$(jq -r '.files[2].path' <<< "$out_and")
|
|
92
|
-
if [[ "$top1" == *"Migration_3_to_4.kt"* ]]; then
|
|
93
|
-
record_pass "Android top-1 = migration file"
|
|
94
|
-
else
|
|
95
|
-
record_fail "Android top-1 should be migration, got: $top1"
|
|
96
|
-
fi
|
|
97
|
-
if [[ "$top2" == *"AuthRepository.kt"* ]]; then
|
|
98
|
-
record_pass "Android top-2 = AuthRepository (security_path)"
|
|
99
|
-
else
|
|
100
|
-
record_fail "Android top-2 should be AuthRepository, got: $top2"
|
|
101
|
-
fi
|
|
102
|
-
if [[ "$top3" == *"HomeScreen.kt"* ]]; then
|
|
103
|
-
record_pass "Android top-3 = HomeScreen (ui_critical)"
|
|
104
|
-
else
|
|
105
|
-
record_fail "Android top-3 should be HomeScreen, got: $top3"
|
|
106
|
-
fi
|
|
107
|
-
|
|
108
|
-
# --- 6: --top N truncates ---
|
|
109
|
-
out_top2=$(node "$SCORE" --diff "$FIX_IOS" --top 2 2>/dev/null)
|
|
110
|
-
n=$(jq '.files | length' <<< "$out_top2")
|
|
111
|
-
if [ "$n" = "2" ]; then
|
|
112
|
-
record_pass "--top 2 truncates files array"
|
|
113
|
-
else
|
|
114
|
-
record_fail "--top 2 should keep 2 rows, got $n"
|
|
115
|
-
fi
|
|
116
|
-
|
|
117
|
-
# --- 7: Empty diff exits 2 ---
|
|
118
|
-
empty=$(mktemp)
|
|
119
|
-
set +e
|
|
120
|
-
node "$SCORE" --diff "$empty" >/dev/null 2>&1
|
|
121
|
-
rc=$?
|
|
122
|
-
set -e
|
|
123
|
-
rm -f "$empty"
|
|
124
|
-
if [ "$rc" -eq 2 ]; then
|
|
125
|
-
record_pass "empty diff exits 2"
|
|
126
|
-
else
|
|
127
|
-
record_fail "empty diff should exit 2, got $rc"
|
|
128
|
-
fi
|
|
129
|
-
|
|
130
|
-
# --- 8: phase-4-review.md declares the advisory step ---
|
|
131
|
-
if grep -qE '^####? .*Diff Risk' "$PHASE4"; then
|
|
132
|
-
record_pass "phase-4-review.md declares Diff Risk step"
|
|
133
|
-
else
|
|
134
|
-
record_fail "phase-4-review.md missing Diff Risk step heading"
|
|
135
|
-
fi
|
|
136
|
-
if grep -q 'diff-risk-score.mjs' "$PHASE4"; then
|
|
137
|
-
record_pass "phase-4-review.md references diff-risk-score.mjs"
|
|
138
|
-
else
|
|
139
|
-
record_fail "phase-4-review.md missing script reference"
|
|
140
|
-
fi
|
|
141
|
-
|
|
142
|
-
# --- 9: code-reviewer.md template carries priority placeholder ---
|
|
143
|
-
if grep -q 'Priority Files' "$REVIEWER"; then
|
|
144
|
-
record_pass "code-reviewer.md declares Priority Files section"
|
|
145
|
-
else
|
|
146
|
-
record_fail "code-reviewer.md missing Priority Files section"
|
|
147
|
-
fi
|
|
148
|
-
|
|
149
|
-
# --- 10: prefs schema exposes diffRisk advisory toggle ---
|
|
150
|
-
if jq -e '.properties.global.properties.diffRiskAdvisory' "$PREFS" >/dev/null 2>&1; then
|
|
151
|
-
record_pass "prefs schema exposes diffRiskAdvisory toggle"
|
|
152
|
-
else
|
|
153
|
-
record_fail "prefs.schema.json missing global.diffRiskAdvisory"
|
|
154
|
-
fi
|
|
155
|
-
|
|
156
|
-
# --- 11: test_lines_removed signal fires on the test-removal fixture ---
|
|
157
|
-
out_testrm=$(node "$SCORE" --diff "$FIX_TESTRM" 2>/dev/null)
|
|
158
|
-
sig_value=$(jq -r '.files[] | select(.path == "MyAppTests/LoginViewModelTests.swift")
|
|
159
|
-
| .signals[] | select(.name == "test_lines_removed") | .value' <<< "$out_testrm")
|
|
160
|
-
if [ "$sig_value" = "16" ]; then
|
|
161
|
-
record_pass "test_lines_removed fires with value=16 (18 removed - 2 added)"
|
|
162
|
-
else
|
|
163
|
-
record_fail "test_lines_removed should fire with value=16, got: ${sig_value:-missing}"
|
|
164
|
-
fi
|
|
165
|
-
sig_on_source=$(jq -r '[.files[] | select(.path == "MyApp/Sources/Auth/LoginViewModel.swift")
|
|
166
|
-
| .signals[] | select(.name == "test_lines_removed")] | length' <<< "$out_testrm")
|
|
167
|
-
if [ "$sig_on_source" = "0" ]; then
|
|
168
|
-
record_pass "test_lines_removed does not fire on source files"
|
|
169
|
-
else
|
|
170
|
-
record_fail "test_lines_removed must only fire on test-classified paths"
|
|
171
|
-
fi
|
|
172
|
-
set +e
|
|
173
|
-
echo "$out_testrm" | node "$VALIDATE" - >/dev/null 2>&1
|
|
174
|
-
rc_testrm=$?
|
|
175
|
-
set -e
|
|
176
|
-
if [ "$rc_testrm" -eq 0 ]; then
|
|
177
|
-
record_pass "validator accepts output carrying test_lines_removed"
|
|
178
|
-
else
|
|
179
|
-
record_fail "validator rejected test_lines_removed output (rc=$rc_testrm)"
|
|
180
|
-
fi
|
|
181
|
-
|
|
182
|
-
# --- Summary ---
|
|
183
|
-
total=$((pass + fail))
|
|
184
|
-
printf '\n→ smoke-diff-risk: %d/%d passed\n' "$pass" "$total"
|
|
185
|
-
if [ "$fail" -ne 0 ]; then
|
|
186
|
-
printf '\nFailures:\n'
|
|
187
|
-
for f in "${failures[@]}"; do printf ' - %s\n' "$f"; done
|
|
188
|
-
exit 1
|
|
189
|
-
fi
|
|
190
|
-
exit 0
|
|
@@ -1,160 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-dynamic-skill-loading.sh - v7.0.I
|
|
3
|
-
#
|
|
4
|
-
# Verifies the index + matcher contract:
|
|
5
|
-
# 1. build-skills-index.mjs produces valid JSON with schemaVersion + entries
|
|
6
|
-
# 2. index skill count matches the raw filesystem count
|
|
7
|
-
# 3. match-skills.mjs ranks iOS skills higher on a SwiftUI task
|
|
8
|
-
# 4. match-skills.mjs --json returns sorted matches[]
|
|
9
|
-
# 5. matcher platform boost applies when --stack matches frontmatter
|
|
10
|
-
# 6. matcher returns empty matches[] for a task with no relevant skills
|
|
11
|
-
# 7. matcher --limit honored
|
|
12
|
-
# 8. prefs schema exposes dynamicSkillLoading default false
|
|
13
|
-
#
|
|
14
|
-
# Exit 0 all pass, 1 any failure.
|
|
15
|
-
|
|
16
|
-
set -euo pipefail
|
|
17
|
-
|
|
18
|
-
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
19
|
-
BUILD="$ROOT/pipeline/scripts/build-skills-index.mjs"
|
|
20
|
-
MATCH="$ROOT/pipeline/scripts/match-skills.mjs"
|
|
21
|
-
SCHEMA="$ROOT/pipeline/schemas/prefs.schema.json"
|
|
22
|
-
|
|
23
|
-
pass=0; fail=0; failures=()
|
|
24
|
-
record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
|
|
25
|
-
record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
|
|
26
|
-
|
|
27
|
-
printf '→ smoke-dynamic-skill-loading (v7.0.I): index + matcher contract\n'
|
|
28
|
-
|
|
29
|
-
# Build into a sandbox so we don't clobber the shipping index
|
|
30
|
-
sandbox=$(mktemp -d)
|
|
31
|
-
skills="$sandbox/skills"
|
|
32
|
-
mkdir -p "$skills/core/multi-agent" "$skills/core/testing-backend" "$skills/external/swiftui-pro" "$skills/external/compose-components"
|
|
33
|
-
|
|
34
|
-
cat > "$skills/core/multi-agent/SKILL.md" <<'S'
|
|
35
|
-
---
|
|
36
|
-
name: multi-agent
|
|
37
|
-
description: Task orchestrator for the multi-agent pipeline
|
|
38
|
-
platform: generic
|
|
39
|
-
---
|
|
40
|
-
body
|
|
41
|
-
S
|
|
42
|
-
|
|
43
|
-
cat > "$skills/core/testing-backend/SKILL.md" <<'S'
|
|
44
|
-
---
|
|
45
|
-
name: testing-backend
|
|
46
|
-
description: Backend testing patterns for pytest / Jest
|
|
47
|
-
platform: backend
|
|
48
|
-
trigger-keywords: pytest, jest, integration test
|
|
49
|
-
---
|
|
50
|
-
body
|
|
51
|
-
S
|
|
52
|
-
|
|
53
|
-
cat > "$skills/external/swiftui-pro/SKILL.md" <<'S'
|
|
54
|
-
---
|
|
55
|
-
name: swiftui-pro
|
|
56
|
-
description: SwiftUI review for modern APIs, state management, performance
|
|
57
|
-
platform: ios
|
|
58
|
-
trigger-keywords: swiftui, swift, ios, swiftui performance
|
|
59
|
-
trigger-paths: *.swift, **/Sources/**/*.swift
|
|
60
|
-
---
|
|
61
|
-
body
|
|
62
|
-
S
|
|
63
|
-
|
|
64
|
-
cat > "$skills/external/compose-components/SKILL.md" <<'S'
|
|
65
|
-
---
|
|
66
|
-
name: compose-components
|
|
67
|
-
description: Material 3 Jetpack Compose components
|
|
68
|
-
platform: android
|
|
69
|
-
trigger-keywords: compose, jetpack, material 3, android
|
|
70
|
-
trigger-paths: *.kt, **/src/main/**/*.kt
|
|
71
|
-
---
|
|
72
|
-
body
|
|
73
|
-
S
|
|
74
|
-
|
|
75
|
-
# --- 1: build index ---
|
|
76
|
-
node "$BUILD" --root "$skills" >/dev/null
|
|
77
|
-
INDEX="$skills/.skills-index.json"
|
|
78
|
-
if [ -f "$INDEX" ] && jq -e '.schemaVersion and .entries and (.skillCount == 4)' "$INDEX" >/dev/null; then
|
|
79
|
-
record_pass "build-skills-index emits valid JSON with 4 entries"
|
|
80
|
-
else
|
|
81
|
-
record_fail "build-skills-index output malformed"
|
|
82
|
-
fi
|
|
83
|
-
|
|
84
|
-
# --- 2: count matches filesystem ---
|
|
85
|
-
raw_count=$(find "$skills" -name SKILL.md | wc -l | tr -d ' ')
|
|
86
|
-
index_count=$(jq -r '.skillCount' "$INDEX")
|
|
87
|
-
if [ "$raw_count" = "$index_count" ]; then
|
|
88
|
-
record_pass "index count matches filesystem ($index_count)"
|
|
89
|
-
else
|
|
90
|
-
record_fail "index count $index_count != filesystem $raw_count"
|
|
91
|
-
fi
|
|
92
|
-
|
|
93
|
-
# --- 3: iOS task ranks SwiftUI skills higher ---
|
|
94
|
-
out=$(node "$MATCH" "SwiftUI dark mode fix on LoginView" \
|
|
95
|
-
--touched-files "Sources/Auth/LoginView.swift" \
|
|
96
|
-
--stack ios --limit 4 --index "$INDEX" --json)
|
|
97
|
-
top=$(jq -r '.matches[0].name' <<< "$out")
|
|
98
|
-
if [ "$top" = "swiftui-pro" ]; then
|
|
99
|
-
record_pass "iOS task ranks swiftui-pro top ($top)"
|
|
100
|
-
else
|
|
101
|
-
record_fail "iOS task should rank swiftui-pro first, got $top"
|
|
102
|
-
fi
|
|
103
|
-
|
|
104
|
-
# --- 4: JSON output is sorted ---
|
|
105
|
-
scores=$(jq -r '.matches[].score' <<< "$out")
|
|
106
|
-
prev=9999; sorted=1
|
|
107
|
-
for s in $scores; do
|
|
108
|
-
if [ "$s" -gt "$prev" ]; then sorted=0; break; fi
|
|
109
|
-
prev=$s
|
|
110
|
-
done
|
|
111
|
-
if [ "$sorted" = "1" ]; then
|
|
112
|
-
record_pass "JSON matches[] sorted by score desc"
|
|
113
|
-
else
|
|
114
|
-
record_fail "matches[] not sorted"
|
|
115
|
-
fi
|
|
116
|
-
|
|
117
|
-
# --- 5: platform boost ---
|
|
118
|
-
# A keyword-free task with only --stack android should still boost Android skill
|
|
119
|
-
out=$(node "$MATCH" "Build the feature end to end" --stack android --limit 4 --index "$INDEX" --json)
|
|
120
|
-
rules=$(jq -r '.matches[] | select(.name=="compose-components") | .reasons[].rule' <<< "$out" 2>/dev/null || echo "")
|
|
121
|
-
if echo "$rules" | grep -q 'platform'; then
|
|
122
|
-
record_pass "platform boost reason present for matched stack"
|
|
123
|
-
else
|
|
124
|
-
record_fail "platform boost missing for --stack android"
|
|
125
|
-
fi
|
|
126
|
-
|
|
127
|
-
# --- 6: empty matches ---
|
|
128
|
-
out=$(node "$MATCH" "Totally unrelated lmn xyz qrs foobar baz" --limit 4 --index "$INDEX" --json)
|
|
129
|
-
n=$(jq '.matches | length' <<< "$out")
|
|
130
|
-
if [ "$n" = "0" ]; then
|
|
131
|
-
record_pass "empty matches[] for unrelated task"
|
|
132
|
-
else
|
|
133
|
-
record_fail "expected 0 matches for unrelated task, got $n"
|
|
134
|
-
fi
|
|
135
|
-
|
|
136
|
-
# --- 7: --limit honored ---
|
|
137
|
-
out=$(node "$MATCH" "swiftui compose test pipeline" --limit 2 --index "$INDEX" --json)
|
|
138
|
-
n=$(jq '.matches | length' <<< "$out")
|
|
139
|
-
if [ "$n" -le "2" ]; then
|
|
140
|
-
record_pass "--limit 2 returns at most 2 matches ($n)"
|
|
141
|
-
else
|
|
142
|
-
record_fail "--limit 2 returned $n"
|
|
143
|
-
fi
|
|
144
|
-
|
|
145
|
-
# --- 8: schema pref ---
|
|
146
|
-
if jq -e '.properties.global.properties.dynamicSkillLoading.default == false' "$SCHEMA" >/dev/null 2>&1; then
|
|
147
|
-
record_pass "prefs schema exposes dynamicSkillLoading default false"
|
|
148
|
-
else
|
|
149
|
-
record_fail "prefs schema missing dynamicSkillLoading or wrong default"
|
|
150
|
-
fi
|
|
151
|
-
|
|
152
|
-
rm -rf "$sandbox"
|
|
153
|
-
|
|
154
|
-
printf '\n══ dynamic-skill-loading smoke: %d passed, %d failed ══\n' "$pass" "$fail"
|
|
155
|
-
if [ "$fail" -gt 0 ]; then
|
|
156
|
-
printf '\nFailures:\n'
|
|
157
|
-
for m in "${failures[@]}"; do printf ' - %s\n' "$m"; done
|
|
158
|
-
exit 1
|
|
159
|
-
fi
|
|
160
|
-
exit 0
|
|
@@ -1,136 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env bash
|
|
2
|
-
# smoke-eval-live.sh - v7.8.0 Paket A: opt-in live golden-task runner.
|
|
3
|
-
#
|
|
4
|
-
# Validates the cost-guarding contract WITHOUT making any model calls. Live
|
|
5
|
-
# mode is gated behind two independent signals (--live flag AND
|
|
6
|
-
# MULTI_AGENT_LIVE_EVAL=1 env) so accidental cost is impossible during CI
|
|
7
|
-
# or local `npm test` runs.
|
|
8
|
-
#
|
|
9
|
-
# All assertions exercise the dry-run code path or expect early-exit before
|
|
10
|
-
# any model invocation.
|
|
11
|
-
|
|
12
|
-
set -uo pipefail
|
|
13
|
-
|
|
14
|
-
REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
|
|
15
|
-
SCRIPT="$REPO_ROOT/pipeline/scripts/eval-golden-tasks-live.mjs"
|
|
16
|
-
|
|
17
|
-
PASS=0
|
|
18
|
-
FAIL=0
|
|
19
|
-
pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
|
|
20
|
-
fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
|
|
21
|
-
|
|
22
|
-
cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
|
|
23
|
-
trap cleanup EXIT
|
|
24
|
-
TMP=$(mktemp -d)
|
|
25
|
-
|
|
26
|
-
# Strip env that would unlock live mode; smoke MUST never spend money.
|
|
27
|
-
unset MULTI_AGENT_LIVE_EVAL
|
|
28
|
-
|
|
29
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
30
|
-
echo "→ 1. Script presence + syntax"
|
|
31
|
-
[ -x "$SCRIPT" ] && pass "eval-golden-tasks-live.mjs is executable" || fail "script not executable"
|
|
32
|
-
node --check "$SCRIPT" 2>/dev/null && pass "syntax check" || fail "syntax error"
|
|
33
|
-
|
|
34
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
35
|
-
echo "→ 2. Default mode is dry-run (no --live, no env)"
|
|
36
|
-
OUT=$(node "$SCRIPT" --json 2>/dev/null)
|
|
37
|
-
if echo "$OUT" | grep -q '"mode": "dry-run"'; then
|
|
38
|
-
pass "default mode = dry-run"
|
|
39
|
-
else
|
|
40
|
-
fail "default mode not dry-run: $(echo "$OUT" | head -3)"
|
|
41
|
-
fi
|
|
42
|
-
|
|
43
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
44
|
-
echo "→ 3. Dry-run does not invoke model"
|
|
45
|
-
# All cases must end with status "skipped-dry-run" - i.e. no live call
|
|
46
|
-
SKIPPED_COUNT=$(echo "$OUT" | grep -c '"pass": "skipped-dry-run"' || true)
|
|
47
|
-
if [ "$SKIPPED_COUNT" -ge 1 ]; then
|
|
48
|
-
pass "dry-run skipped all cases ($SKIPPED_COUNT) without invoking model"
|
|
49
|
-
else
|
|
50
|
-
fail "dry-run did not skip cases properly"
|
|
51
|
-
fi
|
|
52
|
-
|
|
53
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
54
|
-
echo "→ 4. --live without env exits 2 (cost gate)"
|
|
55
|
-
node "$SCRIPT" --live 2>"$TMP/err.txt"
|
|
56
|
-
EC=$?
|
|
57
|
-
if [ "$EC" -eq 2 ] && grep -q "MULTI_AGENT_LIVE_EVAL" "$TMP/err.txt"; then
|
|
58
|
-
pass "--live without env: exit $EC + cost gate message"
|
|
59
|
-
else
|
|
60
|
-
fail "--live cost gate weak (exit=$EC, err='$(cat "$TMP/err.txt")')"
|
|
61
|
-
fi
|
|
62
|
-
|
|
63
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
64
|
-
echo "→ 5. Static check - no fall-through past cost gate"
|
|
65
|
-
# The cost gate must be the first thing checked after flag parsing. If --live
|
|
66
|
-
# is set without env, no model call should be reachable. Verify by static
|
|
67
|
-
# grep: spawnSync("claude", ...) must come AFTER the env-check block.
|
|
68
|
-
GATE_LINE=$(grep -n 'MULTI_AGENT_LIVE_EVAL' "$SCRIPT" | head -1 | cut -d: -f1)
|
|
69
|
-
CALL_LINE=$(grep -n 'spawnSync("claude"' "$SCRIPT" | head -1 | cut -d: -f1)
|
|
70
|
-
if [ -n "$GATE_LINE" ] && [ -n "$CALL_LINE" ] && [ "$GATE_LINE" -lt "$CALL_LINE" ]; then
|
|
71
|
-
pass "cost gate at line $GATE_LINE precedes claude call at line $CALL_LINE"
|
|
72
|
-
else
|
|
73
|
-
fail "cost gate ordering wrong (gate=$GATE_LINE, call=$CALL_LINE)"
|
|
74
|
-
fi
|
|
75
|
-
|
|
76
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
77
|
-
echo "→ 6. Budget cap honored - total-budget < single case budget skips all"
|
|
78
|
-
OUT=$(node "$SCRIPT" --max-cases=5 --budget=0.5 --total-budget=0.1 --json 2>/dev/null)
|
|
79
|
-
RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
|
|
80
|
-
SKIPPED=$(echo "$OUT" | grep -c '"skipped-budget-exhausted"' || true)
|
|
81
|
-
if [ "$SKIPPED" -ge 1 ]; then
|
|
82
|
-
pass "budget exhaustion produces 'skipped-budget-exhausted' result ($SKIPPED times)"
|
|
83
|
-
else
|
|
84
|
-
pass "budget very tight: cases_run=$RUN_COUNT (acceptable in dry-run mode)"
|
|
85
|
-
fi
|
|
86
|
-
|
|
87
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
88
|
-
echo "→ 7. --max-cases caps the run"
|
|
89
|
-
OUT=$(node "$SCRIPT" --max-cases=1 --json 2>/dev/null)
|
|
90
|
-
RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
|
|
91
|
-
if [ "$RUN_COUNT" = "1" ]; then
|
|
92
|
-
pass "--max-cases=1 limits cases_run to 1"
|
|
93
|
-
else
|
|
94
|
-
fail "--max-cases not honored: cases_run=$RUN_COUNT"
|
|
95
|
-
fi
|
|
96
|
-
|
|
97
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
98
|
-
echo "→ 8. --case selector"
|
|
99
|
-
OUT=$(node "$SCRIPT" --case 01-ios-bugfix-darkmode --json 2>/dev/null)
|
|
100
|
-
if echo "$OUT" | grep -q '"case": "01-ios-bugfix-darkmode"'; then
|
|
101
|
-
pass "--case selector picks the right fixture"
|
|
102
|
-
else
|
|
103
|
-
fail "--case selector did not target fixture"
|
|
104
|
-
fi
|
|
105
|
-
|
|
106
|
-
node "$SCRIPT" --case nonexistent-case 2>"$TMP/err.txt"
|
|
107
|
-
EC=$?
|
|
108
|
-
if [ "$EC" -eq 2 ] && grep -q "not found" "$TMP/err.txt"; then
|
|
109
|
-
pass "unknown --case exits 2 with 'not found'"
|
|
110
|
-
else
|
|
111
|
-
fail "unknown --case error weak (exit=$EC)"
|
|
112
|
-
fi
|
|
113
|
-
|
|
114
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
115
|
-
echo "→ 9. Help text mentions cost gate prominently"
|
|
116
|
-
HELP=$(node "$SCRIPT" --help 2>&1)
|
|
117
|
-
if echo "$HELP" | grep -q "MULTI_AGENT_LIVE_EVAL"; then
|
|
118
|
-
pass "--help documents the cost gate env var"
|
|
119
|
-
else
|
|
120
|
-
fail "--help missing cost gate documentation"
|
|
121
|
-
fi
|
|
122
|
-
|
|
123
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
124
|
-
echo "→ 10. Bad --budget exits 2"
|
|
125
|
-
node "$SCRIPT" --budget=invalid 2>/dev/null
|
|
126
|
-
EC=$?
|
|
127
|
-
if [ "$EC" -eq 2 ]; then
|
|
128
|
-
pass "invalid --budget exits 2"
|
|
129
|
-
else
|
|
130
|
-
fail "invalid --budget did not error (exit=$EC)"
|
|
131
|
-
fi
|
|
132
|
-
|
|
133
|
-
# ──────────────────────────────────────────────────────────────────────────
|
|
134
|
-
echo ""
|
|
135
|
-
echo "══ eval-live smoke: $PASS passed, $FAIL failed ══"
|
|
136
|
-
[ "$FAIL" -eq 0 ]
|