@mmerterden/multi-agent-pipeline 12.6.0 → 12.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +209 -0
- package/README.md +18 -18
- package/docs/FIGMA_PIPELINE.md +34 -34
- package/docs/adr/0001-three-model-triage.md +12 -12
- package/docs/adr/0002-instruction-driven-flag.md +5 -5
- package/docs/adr/0003-unified-shared-skills.md +5 -5
- package/docs/adr/0004-zero-dependency-philosophy.md +5 -5
- package/docs/adr/0005-lazy-phase-docs.md +2 -2
- package/docs/adr/0006-skills-core-external-split.md +6 -6
- package/docs/adr/0007-multi-tool-adapter-framework.md +19 -19
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +19 -19
- package/docs/adr/README.md +1 -1
- package/docs/best-practices.md +3 -3
- package/docs/features.md +28 -28
- package/docs/performance.md +16 -16
- package/docs/recovery-guide.md +39 -39
- package/index.js +4 -4
- package/install/_common.mjs +53 -11
- package/install/_copilot-instructions.mjs +2 -2
- package/install/_dev-only-files.mjs +126 -6
- package/install/_platform-filter.mjs +1 -1
- package/install/_telemetry.mjs +1 -1
- package/install/claude.mjs +20 -13
- package/install/copilot.mjs +11 -23
- package/install/index.mjs +7 -15
- package/install/templates/copilot-instructions.md +54 -54
- package/install.js +1 -1
- package/package.json +29 -11
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/build-optimize/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/create-jira/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev-autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev-local/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev-local-autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/finish/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/forget/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/issue/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/jira/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/kill/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/language/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/local/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/log/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/prune-logs/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/purge/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/refactor/SKILL.md +16 -8
- package/pipeline/commands/multi-agent/resume/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/review-issue/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/review-jira/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/routines/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/save/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/scan/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/search/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/stack/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/status/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/sync/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/uninstall/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/update/SKILL.md +3 -3
- package/pipeline/lib/account-resolver.sh +1 -1
- package/pipeline/lib/channels-multi-repo.sh +1 -1
- package/pipeline/lib/context-link-extractor.sh +1 -1
- package/pipeline/lib/credential-store.sh +21 -1
- package/pipeline/lib/fetch-confluence.sh +1 -1
- package/pipeline/lib/fetch-crashlytics.sh +1 -1
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/fetch-graylog.sh +1 -1
- package/pipeline/lib/fetch-swagger.sh +1 -1
- package/pipeline/lib/issue-fetcher.sh +1 -1
- package/pipeline/lib/multi-repo-pipeline.sh +1 -1
- package/pipeline/lib/repo-cache.sh +1 -1
- package/pipeline/lib/submodule-detector.sh +1 -1
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +1 -1
- package/pipeline/multi-agent-refs/component-generation.md +121 -0
- package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
- package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
- package/pipeline/multi-agent-refs/phases/operations.md +28 -0
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +50 -5
- package/pipeline/preferences-template.json +5 -11
- package/pipeline/schemas/agent-state.schema.json +39 -9
- package/pipeline/schemas/analysis-output.schema.json +18 -4
- package/pipeline/schemas/analysis-spec.schema.json +120 -32
- package/pipeline/schemas/clarify-output.schema.json +15 -5
- package/pipeline/schemas/design-check-config.schema.json +32 -11
- package/pipeline/schemas/dev-critic-output.schema.json +20 -5
- package/pipeline/schemas/figma-project-config.schema.json +42 -10
- package/pipeline/schemas/learnings-ledger.schema.json +10 -2
- package/pipeline/schemas/migrations/figma-config-1.0.0-to-2.0.0.mjs +1 -4
- package/pipeline/schemas/migrations/prefs-2.0.0-to-2.1.0.mjs +24 -7
- package/pipeline/schemas/plan-todos.schema.json +6 -3
- package/pipeline/schemas/planning-output.schema.json +5 -1
- package/pipeline/schemas/prefs.schema.json +97 -229
- package/pipeline/schemas/test-gap.schema.json +5 -5
- package/pipeline/schemas/token-budget.json +8 -8
- package/pipeline/schemas/triage-corpus.schema.json +1 -1
- package/pipeline/scripts/_smoke-root.sh +61 -0
- package/pipeline/scripts/aggregate-metrics.mjs +18 -6
- package/pipeline/scripts/audit-log.sh +25 -0
- package/pipeline/scripts/build-skills-index.mjs +6 -2
- package/pipeline/scripts/build-stack-plugins.mjs +142 -39
- package/pipeline/scripts/check-derived-drift.mjs +196 -0
- package/pipeline/scripts/classify-plan-safety.mjs +20 -7
- package/pipeline/scripts/cost-budget-check.mjs +2 -1
- package/pipeline/scripts/cost-table.json +1 -1
- package/pipeline/scripts/diff-explain.mjs +7 -3
- package/pipeline/scripts/diff-risk-score.mjs +13 -3
- package/pipeline/scripts/evidence-gate.mjs +7 -2
- package/pipeline/scripts/gen-mode-dispatch.mjs +38 -21
- package/pipeline/scripts/gen-skills-index.mjs +18 -3
- package/pipeline/scripts/learning-curve.mjs +13 -3
- package/pipeline/scripts/learnings-ledger.mjs +103 -36
- package/pipeline/scripts/localize-commands.mjs +6 -1
- package/pipeline/scripts/match-skills.mjs +15 -4
- package/pipeline/scripts/migrate-prefs.mjs +33 -16
- package/pipeline/scripts/phase-tracker.sh +3 -1
- package/pipeline/scripts/repo-map.mjs +110 -64
- package/pipeline/scripts/review-scope.mjs +7 -1
- package/pipeline/scripts/routine-registry.mjs +4 -9
- package/pipeline/scripts/run-aggregator.mjs +11 -5
- package/pipeline/scripts/run-metrics.mjs +13 -8
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
- package/pipeline/scripts/test-gap-scan.mjs +44 -12
- package/pipeline/scripts/test-integrity-gate.mjs +5 -1
- package/pipeline/scripts/token-budget-report.mjs +44 -21
- package/pipeline/scripts/triage-memory.mjs +126 -34
- package/pipeline/scripts/uninstall.mjs +74 -30
- package/pipeline/scripts/validate-analysis-doc.mjs +15 -5
- package/pipeline/scripts/validate-diff-risk.mjs +32 -18
- package/pipeline/scripts/validate-test-gap.mjs +17 -7
- package/pipeline/scripts/validate-triage.mjs +17 -5
- package/pipeline/scripts/write-state.mjs +32 -9
- package/pipeline/skills/.skills-index.json +91 -91
- package/pipeline/skills/shared/README.md +57 -57
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +1 -1
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +26 -279
- package/pipeline/skills/shared/core/multi-agent-analysis/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-analysis-resolve/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-build-optimize/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-create-jira/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-dev/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-dev-autopilot/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-dev-local/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-dev-local-autopilot/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-finish/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-forget/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-garbage-collect/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-issue/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-jira/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-kill/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-log/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-prune-logs/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-purge/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +16 -8
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-review-issue/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-review-jira/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-routines/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-save/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-search/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-stack/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-test/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-uninstall/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +1 -1
- package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +1 -1
- package/pipeline/skills/shared/external/agent-introspection-debugging/SKILL.md +4 -4
- package/pipeline/skills/shared/external/agentflow/SKILL.md +1 -1
- package/pipeline/skills/shared/external/android-jetpack-compose-expert/SKILL.md +1 -1
- package/pipeline/skills/shared/external/android_ui_verification/SKILL.md +1 -1
- package/pipeline/skills/shared/external/api-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/api-security-best-practices/SKILL.md +1 -1
- package/pipeline/skills/shared/external/app-store-changelog/SKILL.md +1 -1
- package/pipeline/skills/shared/external/backlog/BACKLOG.md +1 -1
- package/pipeline/skills/shared/external/backlog/SKILL.md +12 -12
- package/pipeline/skills/shared/external/ci-cd-pipelines/SKILL.md +1 -1
- package/pipeline/skills/shared/external/context-compression/SKILL.md +1 -1
- package/pipeline/skills/shared/external/council/SKILL.md +3 -3
- package/pipeline/skills/shared/external/css-modern/SKILL.md +1 -1
- package/pipeline/skills/shared/external/database-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +1 -1
- package/pipeline/skills/shared/external/docker-expert/SKILL.md +1 -1
- package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +1 -1
- package/pipeline/skills/shared/external/firebase/SKILL.md +1 -1
- package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +1 -1
- package/pipeline/skills/shared/external/help-skills/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-components-content/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-components-status/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-components-system/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-foundations/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-inputs/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-platforms/SKILL.md +1 -1
- package/pipeline/skills/shared/external/hig-technologies/SKILL.md +1 -1
- package/pipeline/skills/shared/external/html-semantic/SKILL.md +1 -1
- package/pipeline/skills/shared/external/humanizer/SKILL.md +1 -1
- package/pipeline/skills/shared/external/ios-debugger-agent/SKILL.md +1 -1
- package/pipeline/skills/shared/external/ios-developer/SKILL.md +1 -1
- package/pipeline/skills/shared/external/kotlin-coroutines-expert/SKILL.md +1 -1
- package/pipeline/skills/shared/external/macos-menubar-tuist-app/SKILL.md +1 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +1 -1
- package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +1 -1
- package/pipeline/skills/shared/external/nextjs-app-router/SKILL.md +1 -1
- package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/observability-engineer/SKILL.md +1 -1
- package/pipeline/skills/shared/external/python-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/react-best-practices/SKILL.md +1 -1
- package/pipeline/skills/shared/external/rest-api-design/SKILL.md +1 -1
- package/pipeline/skills/shared/external/search-first/SKILL.md +2 -2
- package/pipeline/skills/shared/external/skill-creator/SKILL.md +12 -12
- package/pipeline/skills/shared/external/skill-creator/audit.md +21 -21
- package/pipeline/skills/shared/external/skill-creator/checklist.md +3 -3
- package/pipeline/skills/shared/external/skill-creator/examples.md +10 -10
- package/pipeline/skills/shared/external/skill-creator/label-check.md +17 -17
- package/pipeline/skills/shared/external/skill-creator/scripts/audit-panel.js +86 -50
- package/pipeline/skills/shared/external/skill-creator/template.md +9 -9
- package/pipeline/skills/shared/external/swift-concurrency-expert/SKILL.md +1 -1
- package/pipeline/skills/shared/external/swiftui-performance-audit/SKILL.md +1 -1
- package/pipeline/skills/shared/external/swiftui-ui-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/swiftui-view-refactor/SKILL.md +1 -1
- package/pipeline/skills/shared/external/tailwind-css/SKILL.md +1 -1
- package/pipeline/skills/shared/external/testing-backend/SKILL.md +1 -1
- package/pipeline/skills/shared/external/typescript-patterns/SKILL.md +1 -1
- package/pipeline/skills/shared/external/vue-composition/SKILL.md +1 -1
- package/pipeline/skills/shared/external/web-accessibility/SKILL.md +1 -1
- package/pipeline/skills/shared/external/web-performance/SKILL.md +1 -1
- package/pipeline/skills/shared/external/web-testing/SKILL.md +1 -1
- package/pipeline/skills/shared/external/xcode-build-benchmark/schemas/build-benchmark.schema.json +9 -49
- package/pipeline/skills/skills-index.md +57 -57
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
- package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
- package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
- package/pipeline/eval/golden-tasks/README.md +0 -65
- package/pipeline/eval/intent-cases.json +0 -40
- package/pipeline/eval/run-metrics-fixture.json +0 -46
- package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
- package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
- package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
- package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
- package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
- package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
- package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
- package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
- package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
- package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
- package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
- package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
- package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
- package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
- package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
- package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
- package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
- package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
- package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
- package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
- package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
- package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
- package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
- package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
- package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
- package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
- package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
- package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
- package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
- package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
- package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
- package/pipeline/eval/triage/README.md +0 -54
- package/pipeline/scripts/benchmark-phase-0.sh +0 -128
- package/pipeline/scripts/check-md-links.mjs +0 -84
- package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -297
- package/pipeline/scripts/eval-golden-tasks.mjs +0 -212
- package/pipeline/scripts/eval-intent.mjs +0 -103
- package/pipeline/scripts/eval-mine-corpus.mjs +0 -201
- package/pipeline/scripts/eval-triage.mjs +0 -171
- package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
- package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
- package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
- package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
- package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
- package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
- package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
- package/pipeline/scripts/lint-mcp-refs.mjs +0 -207
- package/pipeline/scripts/lint-skills.mjs +0 -143
- package/pipeline/scripts/run-smokes.mjs +0 -76
- package/pipeline/scripts/smoke-add-detail.sh +0 -137
- package/pipeline/scripts/smoke-agent-guard.sh +0 -74
- package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
- package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
- package/pipeline/scripts/smoke-ask-choice.sh +0 -42
- package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
- package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
- package/pipeline/scripts/smoke-changelog-version.sh +0 -47
- package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
- package/pipeline/scripts/smoke-channels-flow.sh +0 -130
- package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
- package/pipeline/scripts/smoke-clarify.sh +0 -148
- package/pipeline/scripts/smoke-command-inventory.sh +0 -81
- package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
- package/pipeline/scripts/smoke-community-gates.sh +0 -75
- package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
- package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
- package/pipeline/scripts/smoke-cost-budget.sh +0 -70
- package/pipeline/scripts/smoke-cost-summary.sh +0 -139
- package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
- package/pipeline/scripts/smoke-description-tr.sh +0 -82
- package/pipeline/scripts/smoke-dev-critic.sh +0 -144
- package/pipeline/scripts/smoke-diff-explain.sh +0 -147
- package/pipeline/scripts/smoke-diff-risk.sh +0 -190
- package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
- package/pipeline/scripts/smoke-eval-live.sh +0 -136
- package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
- package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
- package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
- package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
- package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
- package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
- package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
- package/pipeline/scripts/smoke-generate-issue.sh +0 -120
- package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
- package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
- package/pipeline/scripts/smoke-install-layout.sh +0 -248
- package/pipeline/scripts/smoke-intent-guard.sh +0 -86
- package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
- package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
- package/pipeline/scripts/smoke-keychain.sh +0 -158
- package/pipeline/scripts/smoke-language-axis.sh +0 -109
- package/pipeline/scripts/smoke-learning-curve.sh +0 -61
- package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
- package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
- package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
- package/pipeline/scripts/smoke-md-links.sh +0 -8
- package/pipeline/scripts/smoke-md2confluence.sh +0 -126
- package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
- package/pipeline/scripts/smoke-migrate-state.sh +0 -102
- package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
- package/pipeline/scripts/smoke-model-fallback.sh +0 -89
- package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
- package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
- package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -115
- package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
- package/pipeline/scripts/smoke-pack-contents.sh +0 -140
- package/pipeline/scripts/smoke-pat-audit.sh +0 -128
- package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
- package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
- package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
- package/pipeline/scripts/smoke-phase-banner.sh +0 -101
- package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
- package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
- package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
- package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
- package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
- package/pipeline/scripts/smoke-plan-safety.sh +0 -139
- package/pipeline/scripts/smoke-plan-todos.sh +0 -196
- package/pipeline/scripts/smoke-plugin-validate.sh +0 -64
- package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
- package/pipeline/scripts/smoke-pre-commit.sh +0 -170
- package/pipeline/scripts/smoke-pref-migration.sh +0 -226
- package/pipeline/scripts/smoke-prefs-language.sh +0 -134
- package/pipeline/scripts/smoke-progress-contract.sh +0 -127
- package/pipeline/scripts/smoke-prune-logs.sh +0 -137
- package/pipeline/scripts/smoke-purge.sh +0 -138
- package/pipeline/scripts/smoke-push-retry.sh +0 -75
- package/pipeline/scripts/smoke-repo-map.sh +0 -300
- package/pipeline/scripts/smoke-review-readiness.sh +0 -92
- package/pipeline/scripts/smoke-review-watch.sh +0 -146
- package/pipeline/scripts/smoke-routines.sh +0 -84
- package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
- package/pipeline/scripts/smoke-run-metrics.sh +0 -50
- package/pipeline/scripts/smoke-search.sh +0 -187
- package/pipeline/scripts/smoke-shadow-git.sh +0 -224
- package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
- package/pipeline/scripts/smoke-skill-language.sh +0 -83
- package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
- package/pipeline/scripts/smoke-skill-scan.sh +0 -198
- package/pipeline/scripts/smoke-source-parity.sh +0 -85
- package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
- package/pipeline/scripts/smoke-sync-parity.sh +0 -92
- package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
- package/pipeline/scripts/smoke-telemetry.sh +0 -147
- package/pipeline/scripts/smoke-test-gap.sh +0 -183
- package/pipeline/scripts/smoke-token-budget.sh +0 -67
- package/pipeline/scripts/smoke-token-preflight.sh +0 -82
- package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
- package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
- package/pipeline/scripts/smoke-triage-memory.sh +0 -174
- package/pipeline/scripts/smoke-update-check.sh +0 -135
- package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
- package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
- package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
- package/pipeline/scripts/smoke-validator-gates.sh +0 -164
- package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
- package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
- package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
- package/pipeline/scripts/smoke-work-summary.sh +0 -163
- package/pipeline/scripts/smoke-workflow-audit.sh +0 -69
- package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
- package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
- package/pipeline/scripts/smoke-write-state.sh +0 -115
- package/pipeline/scripts/sync-parity-check.sh +0 -135
- package/pipeline/scripts/test-gap-rules/android.json +0 -25
- package/pipeline/scripts/test-gap-rules/ios.json +0 -29
- package/pipeline/scripts/test-gap-rules/node.json +0 -17
- package/pipeline/scripts/test-gap-rules/python.json +0 -19
- package/pipeline/scripts/validate-schemas.mjs +0 -88
|
@@ -1,297 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* @file eval-golden-tasks-live.mjs - v7.8.0 Paket A
|
|
5
|
-
*
|
|
6
|
-
* Opt-in live evaluation harness for golden-task fixtures. Unlike the
|
|
7
|
-
* always-on contract harness (`eval-golden-tasks.mjs`) which only validates
|
|
8
|
-
* schema shape, this harness optionally invokes a real model via the local
|
|
9
|
-
* `claude` CLI and compares the actual output against the fixture's expected
|
|
10
|
-
* scope (stack, blocker count, deferral count).
|
|
11
|
-
*
|
|
12
|
-
* **Default behavior is dry-run** - no model is invoked unless `--live` is
|
|
13
|
-
* passed AND `MULTI_AGENT_LIVE_EVAL=1` is set in the environment. This
|
|
14
|
-
* double-gate prevents accidental cost.
|
|
15
|
-
*
|
|
16
|
-
* Cost guard:
|
|
17
|
-
* - Per-case budget cap (default $1 USD) - case skipped if exceeded
|
|
18
|
-
* - Run-wide max-cases cap (default 1) - even with --live, only N cases run
|
|
19
|
-
* - Run-wide hard ceiling: budget * max_cases (default $1 total)
|
|
20
|
-
*
|
|
21
|
-
* Zero-dep: shells out to the user's installed `claude` CLI rather than the
|
|
22
|
-
* Anthropic SDK. Honors ADR-4 and uses the user's existing CLI auth.
|
|
23
|
-
*
|
|
24
|
-
* Usage:
|
|
25
|
-
* node eval-golden-tasks-live.mjs # dry-run (default)
|
|
26
|
-
* node eval-golden-tasks-live.mjs --live # blocked unless env set
|
|
27
|
-
* MULTI_AGENT_LIVE_EVAL=1 node eval-golden-tasks-live.mjs --live --case 01-...
|
|
28
|
-
* node eval-golden-tasks-live.mjs --budget=0.50 --max-cases=2
|
|
29
|
-
* node eval-golden-tasks-live.mjs --json # machine-readable summary
|
|
30
|
-
*
|
|
31
|
-
* Exit codes:
|
|
32
|
-
* 0 all run cases produced output matching expected scope
|
|
33
|
-
* 1 one or more cases mismatched expected
|
|
34
|
-
* 2 setup/usage error
|
|
35
|
-
* 3 budget exhausted before run completed
|
|
36
|
-
*
|
|
37
|
-
* @module pipeline/scripts/eval-golden-tasks-live
|
|
38
|
-
*/
|
|
39
|
-
|
|
40
|
-
import { readdirSync, readFileSync, existsSync } from "node:fs";
|
|
41
|
-
import { join, dirname, resolve } from "node:path";
|
|
42
|
-
import { fileURLToPath } from "node:url";
|
|
43
|
-
import { spawnSync } from "node:child_process";
|
|
44
|
-
|
|
45
|
-
const here = dirname(fileURLToPath(import.meta.url));
|
|
46
|
-
const root = resolve(here, "..", "..");
|
|
47
|
-
const evalDir = join(root, "pipeline", "eval", "golden-tasks");
|
|
48
|
-
|
|
49
|
-
const argv = process.argv.slice(2);
|
|
50
|
-
const flags = {};
|
|
51
|
-
for (let i = 0; i < argv.length; i++) {
|
|
52
|
-
const a = argv[i];
|
|
53
|
-
if (a.startsWith("--")) {
|
|
54
|
-
const [key, ...rest] = a.slice(2).split("=");
|
|
55
|
-
if (rest.length) flags[key] = rest.join("=");
|
|
56
|
-
else if (argv[i + 1] && !argv[i + 1].startsWith("--")) {
|
|
57
|
-
flags[key] = argv[i + 1];
|
|
58
|
-
i++;
|
|
59
|
-
} else flags[key] = true;
|
|
60
|
-
}
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
if (flags.help || flags.h) {
|
|
64
|
-
console.log(`Usage: eval-golden-tasks-live.mjs [options]
|
|
65
|
-
|
|
66
|
-
Modes:
|
|
67
|
-
(default) dry-run: list what would run, no model invocation
|
|
68
|
-
--live live: call \`claude -p\` for real (requires
|
|
69
|
-
MULTI_AGENT_LIVE_EVAL=1 in env)
|
|
70
|
-
|
|
71
|
-
Selection:
|
|
72
|
-
--case <slug> run only this case (e.g. 01-ios-bugfix-darkmode)
|
|
73
|
-
--max-cases <N> cap total cases to run (default: 1)
|
|
74
|
-
|
|
75
|
-
Cost guard:
|
|
76
|
-
--budget <usd> per-case budget cap (default: 1.00 USD)
|
|
77
|
-
--total-budget <usd> run-wide ceiling (default: budget * max-cases)
|
|
78
|
-
|
|
79
|
-
Output:
|
|
80
|
-
--json machine-readable JSON summary
|
|
81
|
-
`);
|
|
82
|
-
process.exit(0);
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
const isLive = !!flags.live;
|
|
86
|
-
const liveEnvOK = process.env.MULTI_AGENT_LIVE_EVAL === "1";
|
|
87
|
-
const budget = parseFloat(flags.budget || "1.00");
|
|
88
|
-
const maxCases = parseInt(flags["max-cases"] || "1", 10);
|
|
89
|
-
const totalBudget = parseFloat(flags["total-budget"] || String(budget * maxCases));
|
|
90
|
-
const asJson = !!flags.json;
|
|
91
|
-
|
|
92
|
-
if (isLive && !liveEnvOK) {
|
|
93
|
-
process.stderr.write(
|
|
94
|
-
"eval-golden-tasks-live: --live requires MULTI_AGENT_LIVE_EVAL=1 in env (cost gate).\n",
|
|
95
|
-
);
|
|
96
|
-
process.exit(2);
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
if (!Number.isFinite(budget) || budget <= 0) {
|
|
100
|
-
process.stderr.write(`eval-golden-tasks-live: invalid --budget ${flags.budget}\n`);
|
|
101
|
-
process.exit(2);
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
function listCases() {
|
|
105
|
-
if (!existsSync(evalDir)) return [];
|
|
106
|
-
return readdirSync(evalDir, { withFileTypes: true })
|
|
107
|
-
.filter((e) => e.isDirectory() && /^\d{2}-/.test(e.name))
|
|
108
|
-
.map((e) => e.name)
|
|
109
|
-
.sort();
|
|
110
|
-
}
|
|
111
|
-
|
|
112
|
-
function loadCase(slug) {
|
|
113
|
-
const caseDir = join(evalDir, slug);
|
|
114
|
-
const taskFile = join(caseDir, "task.json");
|
|
115
|
-
if (!existsSync(taskFile)) return null;
|
|
116
|
-
let task;
|
|
117
|
-
try {
|
|
118
|
-
task = JSON.parse(readFileSync(taskFile, "utf-8"));
|
|
119
|
-
} catch (e) {
|
|
120
|
-
process.stderr.write(`[eval-live] malformed task.json for "${slug}": ${e.message}\n`);
|
|
121
|
-
return null;
|
|
122
|
-
}
|
|
123
|
-
return { slug, dir: caseDir, task };
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
function hasClaudeCLI() {
|
|
127
|
-
const r = spawnSync("which", ["claude"], { encoding: "utf-8" });
|
|
128
|
-
return r.status === 0 && (r.stdout || "").trim().length > 0;
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
/**
|
|
132
|
-
* Live invocation - wraps the task as a prompt for `claude -p`. Returns the
|
|
133
|
-
* raw stdout. The current iteration captures only the analysis-phase output
|
|
134
|
-
* by asking for a JSON-only response per the analysis schema. Future work
|
|
135
|
-
* extends this to full Phase 1/2/4 multi-call evaluation.
|
|
136
|
-
*/
|
|
137
|
-
function invokeLive(taskJson) {
|
|
138
|
-
const prompt = [
|
|
139
|
-
"Perform Phase 1 (Analysis) of the multi-agent-pipeline for the task below.",
|
|
140
|
-
"Return ONLY the analysis JSON (matching pipeline/schemas/analysis-output.schema.json).",
|
|
141
|
-
"Do not include explanatory prose, markdown fences, or commentary.",
|
|
142
|
-
"",
|
|
143
|
-
"Task:",
|
|
144
|
-
JSON.stringify(taskJson, null, 2),
|
|
145
|
-
].join("\n");
|
|
146
|
-
|
|
147
|
-
const r = spawnSync("claude", ["-p", prompt], {
|
|
148
|
-
encoding: "utf-8",
|
|
149
|
-
timeout: 5 * 60 * 1000,
|
|
150
|
-
});
|
|
151
|
-
if (r.status !== 0) {
|
|
152
|
-
return { ok: false, error: `claude exit ${r.status}: ${r.stderr || ""}` };
|
|
153
|
-
}
|
|
154
|
-
return { ok: true, raw: r.stdout || "" };
|
|
155
|
-
}
|
|
156
|
-
|
|
157
|
-
function tryParseJson(s) {
|
|
158
|
-
try {
|
|
159
|
-
return { ok: true, value: JSON.parse(s) };
|
|
160
|
-
} catch (e) {
|
|
161
|
-
// Try to extract first {...} block
|
|
162
|
-
const m = /\{[\s\S]*\}/.exec(s);
|
|
163
|
-
if (m) {
|
|
164
|
-
try {
|
|
165
|
-
return { ok: true, value: JSON.parse(m[0]) };
|
|
166
|
-
} catch {
|
|
167
|
-
/* fallthrough */
|
|
168
|
-
}
|
|
169
|
-
}
|
|
170
|
-
return { ok: false, error: e.message };
|
|
171
|
-
}
|
|
172
|
-
}
|
|
173
|
-
|
|
174
|
-
function evaluateOne(c) {
|
|
175
|
-
const result = {
|
|
176
|
-
case: c.slug,
|
|
177
|
-
mode: isLive ? "live" : "dry-run",
|
|
178
|
-
expected: {
|
|
179
|
-
stack: c.task.expectedStack,
|
|
180
|
-
blockers: c.task.expectedBlockers,
|
|
181
|
-
deferrals: c.task.expectedDeferrals,
|
|
182
|
-
},
|
|
183
|
-
actual: null,
|
|
184
|
-
pass: null,
|
|
185
|
-
notes: [],
|
|
186
|
-
};
|
|
187
|
-
|
|
188
|
-
if (!isLive) {
|
|
189
|
-
result.pass = "skipped-dry-run";
|
|
190
|
-
result.notes.push("would invoke claude -p with task prompt");
|
|
191
|
-
return result;
|
|
192
|
-
}
|
|
193
|
-
|
|
194
|
-
if (!hasClaudeCLI()) {
|
|
195
|
-
result.pass = "skipped-no-cli";
|
|
196
|
-
result.notes.push("claude CLI not on PATH");
|
|
197
|
-
return result;
|
|
198
|
-
}
|
|
199
|
-
|
|
200
|
-
const live = invokeLive(c.task);
|
|
201
|
-
if (!live.ok) {
|
|
202
|
-
result.pass = false;
|
|
203
|
-
result.notes.push(`live invocation failed: ${live.error}`);
|
|
204
|
-
return result;
|
|
205
|
-
}
|
|
206
|
-
|
|
207
|
-
const parsed = tryParseJson(live.raw);
|
|
208
|
-
if (!parsed.ok) {
|
|
209
|
-
result.pass = false;
|
|
210
|
-
result.notes.push(`could not parse model output as JSON: ${parsed.error}`);
|
|
211
|
-
return result;
|
|
212
|
-
}
|
|
213
|
-
|
|
214
|
-
const actual = parsed.value;
|
|
215
|
-
result.actual = {
|
|
216
|
-
stack: actual?.stack?.primary ?? actual?.stack ?? null,
|
|
217
|
-
};
|
|
218
|
-
if (actual?.stack?.primary === c.task.expectedStack) {
|
|
219
|
-
result.pass = true;
|
|
220
|
-
result.notes.push(`stack match: ${actual.stack.primary}`);
|
|
221
|
-
} else {
|
|
222
|
-
result.pass = false;
|
|
223
|
-
result.notes.push(
|
|
224
|
-
`stack mismatch: expected ${c.task.expectedStack}, got ${result.actual.stack}`,
|
|
225
|
-
);
|
|
226
|
-
}
|
|
227
|
-
return result;
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
// ─────────────────────────────────────────────────────────────────────────
|
|
231
|
-
const allCases = listCases();
|
|
232
|
-
let selected = allCases;
|
|
233
|
-
if (flags.case) {
|
|
234
|
-
selected = selected.filter((s) => s === flags.case);
|
|
235
|
-
if (selected.length === 0) {
|
|
236
|
-
process.stderr.write(`eval-golden-tasks-live: case not found: ${flags.case}\n`);
|
|
237
|
-
process.exit(2);
|
|
238
|
-
}
|
|
239
|
-
}
|
|
240
|
-
selected = selected.slice(0, maxCases);
|
|
241
|
-
|
|
242
|
-
const results = [];
|
|
243
|
-
let projectedSpend = 0;
|
|
244
|
-
|
|
245
|
-
for (const slug of selected) {
|
|
246
|
-
const c = loadCase(slug);
|
|
247
|
-
if (!c) {
|
|
248
|
-
results.push({ case: slug, pass: false, notes: ["fixture missing"] });
|
|
249
|
-
continue;
|
|
250
|
-
}
|
|
251
|
-
if (projectedSpend + budget > totalBudget) {
|
|
252
|
-
results.push({
|
|
253
|
-
case: slug,
|
|
254
|
-
pass: "skipped-budget-exhausted",
|
|
255
|
-
notes: [`would exceed total-budget ${totalBudget} (already projected ${projectedSpend})`],
|
|
256
|
-
});
|
|
257
|
-
continue;
|
|
258
|
-
}
|
|
259
|
-
const r = evaluateOne(c);
|
|
260
|
-
if (r.pass !== "skipped-dry-run" && r.pass !== "skipped-no-cli") {
|
|
261
|
-
projectedSpend += budget;
|
|
262
|
-
}
|
|
263
|
-
results.push(r);
|
|
264
|
-
}
|
|
265
|
-
|
|
266
|
-
const summary = {
|
|
267
|
-
mode: isLive ? "live" : "dry-run",
|
|
268
|
-
cases_total: allCases.length,
|
|
269
|
-
cases_run: results.length,
|
|
270
|
-
cases_passed: results.filter((r) => r.pass === true).length,
|
|
271
|
-
cases_failed: results.filter((r) => r.pass === false).length,
|
|
272
|
-
cases_skipped: results.filter((r) => typeof r.pass === "string" && r.pass.startsWith("skipped")).length,
|
|
273
|
-
budget_per_case: budget,
|
|
274
|
-
total_budget: totalBudget,
|
|
275
|
-
projected_spend: projectedSpend,
|
|
276
|
-
};
|
|
277
|
-
|
|
278
|
-
if (asJson) {
|
|
279
|
-
console.log(JSON.stringify({ summary, results }, null, 2));
|
|
280
|
-
} else {
|
|
281
|
-
console.log("");
|
|
282
|
-
console.log(`eval-golden-tasks-live (${summary.mode})`);
|
|
283
|
-
console.log(` cases: ${summary.cases_run}/${summary.cases_total} run, ${summary.cases_passed} passed, ${summary.cases_failed} failed, ${summary.cases_skipped} skipped`);
|
|
284
|
-
console.log(` budget: $${budget.toFixed(2)}/case · total cap $${totalBudget.toFixed(2)} · projected spend $${projectedSpend.toFixed(2)}`);
|
|
285
|
-
console.log("");
|
|
286
|
-
for (const r of results) {
|
|
287
|
-
const icon = r.pass === true ? "✓" : r.pass === false ? "✗" : "·";
|
|
288
|
-
console.log(` ${icon} ${r.case} [${r.pass}]`);
|
|
289
|
-
for (const n of r.notes) console.log(` ${n}`);
|
|
290
|
-
}
|
|
291
|
-
console.log("");
|
|
292
|
-
}
|
|
293
|
-
|
|
294
|
-
const hasFailure = results.some((r) => r.pass === false);
|
|
295
|
-
const budgetExhausted = results.some((r) => r.pass === "skipped-budget-exhausted");
|
|
296
|
-
|
|
297
|
-
process.exit(hasFailure ? 1 : budgetExhausted ? 3 : 0);
|
|
@@ -1,212 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
// eval-golden-tasks.mjs - contract regression check for whole-pipeline fixtures.
|
|
3
|
-
//
|
|
4
|
-
// Walks pipeline/eval/golden-tasks/<NN>-<name>/ subdirectories. For each case:
|
|
5
|
-
// 1. Reads task.json (input) + expected/phase-{1,2,4-review,4-triage}.json.
|
|
6
|
-
// 2. Validates each expected/ file against its paired validator script
|
|
7
|
-
// (validate-analysis.mjs / validate-planning.mjs / validate-reviewer.mjs /
|
|
8
|
-
// validate-triage.mjs).
|
|
9
|
-
// 3. Cross-file consistency: Phase 2 tasks[].files[] ⊆ Phase 1
|
|
10
|
-
// touchedAreas[].path. Phase 4 review findings[].file referenced in
|
|
11
|
-
// Phase 2 planned files (or flagged as deferred in triage).
|
|
12
|
-
// 4. Stack match: phase-1.stack.primary === task.expectedStack.
|
|
13
|
-
// 5. Blocker count match: triage.accepted[].filter(severity=blocking).length
|
|
14
|
-
// === task.expectedBlockers. Same for deferred bucket.
|
|
15
|
-
//
|
|
16
|
-
// What this catches:
|
|
17
|
-
// - Schema drift (any of the 4 phase schemas changes in a breaking way).
|
|
18
|
-
// - Fixture-vs-spec divergence (someone updates a schema without re-syncing
|
|
19
|
-
// the golden fixtures).
|
|
20
|
-
// - Coverage gap (a triage that loses findings vs. reviewer input).
|
|
21
|
-
//
|
|
22
|
-
// What this does NOT catch:
|
|
23
|
-
// - Real model output quality. This is a CONTRACT test, not a model eval.
|
|
24
|
-
// A separate multi-model harness (with API keys, non-deterministic) is
|
|
25
|
-
// future work. See pipeline/eval/golden-tasks/README.md.
|
|
26
|
-
//
|
|
27
|
-
// Usage:
|
|
28
|
-
// node pipeline/scripts/eval-golden-tasks.mjs # all cases
|
|
29
|
-
// node pipeline/scripts/eval-golden-tasks.mjs --case 01-... # one case
|
|
30
|
-
// node pipeline/scripts/eval-golden-tasks.mjs --json # CI-friendly
|
|
31
|
-
//
|
|
32
|
-
// Exit codes: 0 all pass, 1 one or more fail, 2 usage/setup error.
|
|
33
|
-
|
|
34
|
-
import { readdirSync, readFileSync, statSync, existsSync } from "node:fs";
|
|
35
|
-
import { dirname, join, resolve } from "node:path";
|
|
36
|
-
import { fileURLToPath } from "node:url";
|
|
37
|
-
import { spawnSync } from "node:child_process";
|
|
38
|
-
|
|
39
|
-
const here = dirname(fileURLToPath(import.meta.url));
|
|
40
|
-
const root = resolve(here, "..", "..");
|
|
41
|
-
const evalDir = join(root, "pipeline", "eval", "golden-tasks");
|
|
42
|
-
|
|
43
|
-
const validators = {
|
|
44
|
-
analysis: join(root, "pipeline", "scripts", "validate-analysis.mjs"),
|
|
45
|
-
planning: join(root, "pipeline", "scripts", "validate-planning.mjs"),
|
|
46
|
-
reviewer: join(root, "pipeline", "scripts", "validate-reviewer.mjs"),
|
|
47
|
-
triage: join(root, "pipeline", "scripts", "validate-triage.mjs"),
|
|
48
|
-
};
|
|
49
|
-
|
|
50
|
-
const args = process.argv.slice(2);
|
|
51
|
-
const opts = { json: false, case: null };
|
|
52
|
-
for (let i = 0; i < args.length; i++) {
|
|
53
|
-
if (args[i] === "--json") opts.json = true;
|
|
54
|
-
else if (args[i] === "--case") opts.case = args[++i];
|
|
55
|
-
else if (args[i] === "-h" || args[i] === "--help") {
|
|
56
|
-
console.log("Usage: eval-golden-tasks.mjs [--case <name>] [--json]");
|
|
57
|
-
process.exit(0);
|
|
58
|
-
}
|
|
59
|
-
}
|
|
60
|
-
|
|
61
|
-
function listCases() {
|
|
62
|
-
if (!existsSync(evalDir)) return [];
|
|
63
|
-
return readdirSync(evalDir)
|
|
64
|
-
.filter((f) => f !== "README.md" && statSync(join(evalDir, f)).isDirectory())
|
|
65
|
-
.sort();
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
function runValidator(scriptPath, payloadPath) {
|
|
69
|
-
// validate-reviewer.mjs reads stdin with `-`; others take a path.
|
|
70
|
-
// For simplicity, all validators accept either stdin or file path. Pass path.
|
|
71
|
-
const r = spawnSync("node", [scriptPath, payloadPath], { encoding: "utf-8" });
|
|
72
|
-
return { code: r.status, stdout: r.stdout || "", stderr: r.stderr || "" };
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
function runCase(name) {
|
|
76
|
-
const dir = join(evalDir, name);
|
|
77
|
-
const errors = [];
|
|
78
|
-
|
|
79
|
-
const taskPath = join(dir, "task.json");
|
|
80
|
-
const analysisPath = join(dir, "expected", "phase-1-analysis.json");
|
|
81
|
-
const planPath = join(dir, "expected", "phase-2-plan.json");
|
|
82
|
-
const reviewPath = join(dir, "expected", "phase-4-review.json");
|
|
83
|
-
const triagePath = join(dir, "expected", "phase-4-triage.json");
|
|
84
|
-
|
|
85
|
-
for (const [label, p] of [
|
|
86
|
-
["task.json", taskPath],
|
|
87
|
-
["phase-1-analysis.json", analysisPath],
|
|
88
|
-
["phase-2-plan.json", planPath],
|
|
89
|
-
["phase-4-review.json", reviewPath],
|
|
90
|
-
["phase-4-triage.json", triagePath],
|
|
91
|
-
]) {
|
|
92
|
-
if (!existsSync(p)) {
|
|
93
|
-
errors.push(`missing ${label}`);
|
|
94
|
-
}
|
|
95
|
-
}
|
|
96
|
-
if (errors.length) return { name, ok: false, errors };
|
|
97
|
-
|
|
98
|
-
let task, analysis, plan, review, triage;
|
|
99
|
-
try {
|
|
100
|
-
task = JSON.parse(readFileSync(taskPath, "utf-8"));
|
|
101
|
-
analysis = JSON.parse(readFileSync(analysisPath, "utf-8"));
|
|
102
|
-
plan = JSON.parse(readFileSync(planPath, "utf-8"));
|
|
103
|
-
review = JSON.parse(readFileSync(reviewPath, "utf-8"));
|
|
104
|
-
triage = JSON.parse(readFileSync(triagePath, "utf-8"));
|
|
105
|
-
} catch (e) {
|
|
106
|
-
return { name, ok: false, errors: [`JSON parse: ${e.message}`] };
|
|
107
|
-
}
|
|
108
|
-
|
|
109
|
-
// --- schema checks ---
|
|
110
|
-
const a = runValidator(validators.analysis, analysisPath);
|
|
111
|
-
if (a.code !== 0) errors.push(`phase-1-analysis fails validate-analysis (exit ${a.code}): ${a.stderr.trim()}`);
|
|
112
|
-
|
|
113
|
-
const p = runValidator(validators.planning, planPath);
|
|
114
|
-
if (p.code !== 0) errors.push(`phase-2-plan fails validate-planning (exit ${p.code}): ${p.stderr.trim()}`);
|
|
115
|
-
|
|
116
|
-
// phase-4-review is an array (one entry per reviewer) - validate each
|
|
117
|
-
if (!Array.isArray(review)) {
|
|
118
|
-
errors.push("phase-4-review.json must be a JSON array (one entry per reviewer)");
|
|
119
|
-
} else {
|
|
120
|
-
review.forEach((entry, idx) => {
|
|
121
|
-
// write temp file for each reviewer entry
|
|
122
|
-
const tmpPayload = JSON.stringify(entry);
|
|
123
|
-
const r = spawnSync("node", [validators.reviewer, "-"], {
|
|
124
|
-
input: tmpPayload,
|
|
125
|
-
encoding: "utf-8",
|
|
126
|
-
});
|
|
127
|
-
if (r.status !== 0) errors.push(`phase-4-review[${idx}] fails validate-reviewer (exit ${r.status}): ${(r.stderr || "").trim()}`);
|
|
128
|
-
});
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
const t = runValidator(validators.triage, triagePath);
|
|
132
|
-
// validate-triage exit codes: 0 valid+clean, 2 contradiction, 3 correction.
|
|
133
|
-
// For fixtures we accept 0 only.
|
|
134
|
-
if (t.code !== 0) errors.push(`phase-4-triage fails validate-triage (exit ${t.code}): ${t.stderr.trim()}`);
|
|
135
|
-
|
|
136
|
-
// --- cross-file consistency ---
|
|
137
|
-
|
|
138
|
-
// 1. Stack match
|
|
139
|
-
if (analysis.stack?.primary !== task.expectedStack) {
|
|
140
|
-
errors.push(`stack mismatch: task.expectedStack=${task.expectedStack}, phase-1.stack.primary=${analysis.stack?.primary}`);
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
// 2. plan files ⊆ analysis touchedAreas paths
|
|
144
|
-
const touched = new Set((analysis.touchedAreas || []).map((a) => a.path));
|
|
145
|
-
for (const task of plan.tasks || []) {
|
|
146
|
-
for (const f of task.files || []) {
|
|
147
|
-
if (!touched.has(f)) {
|
|
148
|
-
errors.push(`plan task ${task.id} references file not in Phase 1 touchedAreas: ${f}`);
|
|
149
|
-
}
|
|
150
|
-
}
|
|
151
|
-
}
|
|
152
|
-
|
|
153
|
-
// 3. every reviewer finding traces to a planned file OR appears in triage
|
|
154
|
-
// (as accepted / deferred / rejected).
|
|
155
|
-
const plannedFiles = new Set();
|
|
156
|
-
for (const task of plan.tasks || []) for (const f of task.files || []) plannedFiles.add(f);
|
|
157
|
-
|
|
158
|
-
const triageAll = [
|
|
159
|
-
...(triage.accepted || []).map((x) => x),
|
|
160
|
-
...(triage.deferred || []).map((x) => x.finding),
|
|
161
|
-
...(triage.rejected || []).map((x) => x.finding),
|
|
162
|
-
].filter((f) => f && typeof f === "object");
|
|
163
|
-
const triageKeys = new Set(triageAll.map((f) => `${f.file}::${f.line}::${f.issue}`));
|
|
164
|
-
|
|
165
|
-
for (const rev of (Array.isArray(review) ? review : [])) {
|
|
166
|
-
for (const f of rev.findings || []) {
|
|
167
|
-
const key = `${f.file}::${f.line}::${f.issue}`;
|
|
168
|
-
const inPlan = plannedFiles.has(f.file);
|
|
169
|
-
const inTriage = triageKeys.has(key);
|
|
170
|
-
if (!inPlan && !inTriage) {
|
|
171
|
-
errors.push(`review finding not in plan+triage: ${key}`);
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
}
|
|
175
|
-
|
|
176
|
-
// 4. blocker + deferral counts match task.expected*
|
|
177
|
-
const acceptedBlockers = (triage.accepted || []).filter((x) => x.severity === "blocking").length;
|
|
178
|
-
if (typeof task.expectedBlockers === "number" && acceptedBlockers !== task.expectedBlockers) {
|
|
179
|
-
errors.push(`expectedBlockers=${task.expectedBlockers}, triage.accepted blocking count=${acceptedBlockers}`);
|
|
180
|
-
}
|
|
181
|
-
const deferredCount = (triage.deferred || []).length;
|
|
182
|
-
if (typeof task.expectedDeferrals === "number" && deferredCount !== task.expectedDeferrals) {
|
|
183
|
-
errors.push(`expectedDeferrals=${task.expectedDeferrals}, triage.deferred count=${deferredCount}`);
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
return { name, ok: errors.length === 0, errors };
|
|
187
|
-
}
|
|
188
|
-
|
|
189
|
-
// --- main ---
|
|
190
|
-
|
|
191
|
-
const cases = opts.case ? [opts.case] : listCases();
|
|
192
|
-
if (cases.length === 0) {
|
|
193
|
-
console.error("No golden-task fixtures found under pipeline/eval/golden-tasks/");
|
|
194
|
-
process.exit(2);
|
|
195
|
-
}
|
|
196
|
-
|
|
197
|
-
const results = cases.map(runCase);
|
|
198
|
-
const passed = results.filter((r) => r.ok).length;
|
|
199
|
-
const failed = results.length - passed;
|
|
200
|
-
|
|
201
|
-
if (opts.json) {
|
|
202
|
-
console.log(JSON.stringify({ passed, failed, results }, null, 2));
|
|
203
|
-
} else {
|
|
204
|
-
for (const r of results) {
|
|
205
|
-
const mark = r.ok ? "\x1b[32m✓\x1b[0m" : "\x1b[31m✗\x1b[0m";
|
|
206
|
-
console.log(`${mark} ${r.name.padEnd(38)} ${r.ok ? "" : "FAIL"}`);
|
|
207
|
-
if (!r.ok) for (const e of r.errors) console.log(` - ${e}`);
|
|
208
|
-
}
|
|
209
|
-
console.log(`\n══ eval-golden-tasks: ${passed}/${results.length} passed ══`);
|
|
210
|
-
}
|
|
211
|
-
|
|
212
|
-
process.exit(failed > 0 ? 1 : 0);
|
|
@@ -1,103 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* @file eval-intent.mjs - measured accuracy for the Phase 0 intent guard.
|
|
5
|
-
*
|
|
6
|
-
* Runs every labeled case in pipeline/eval/intent-cases.json through
|
|
7
|
-
* pipeline/lib/classify-intent.sh and reports accuracy. This turns the
|
|
8
|
-
* classifier from "a heuristic we hope works" into a number with a regression
|
|
9
|
-
* gate: a dangerous miss is a clear task read as `question` (work skipped) or a
|
|
10
|
-
* clear question read as `task` (a worktree spun up for nothing). `ambiguous`
|
|
11
|
-
* is the safe middle and is only correct when the case labels it so.
|
|
12
|
-
*
|
|
13
|
-
* Exit 0 if accuracy >= threshold (default 0.95), else 1. Override with
|
|
14
|
-
* --min <0..1>. Prints a JSON summary + the mismatches.
|
|
15
|
-
*
|
|
16
|
-
* @module pipeline/scripts/eval-intent
|
|
17
|
-
*/
|
|
18
|
-
|
|
19
|
-
import { readFileSync } from "node:fs";
|
|
20
|
-
import { execFileSync } from "node:child_process";
|
|
21
|
-
import { join, dirname } from "node:path";
|
|
22
|
-
import { fileURLToPath } from "node:url";
|
|
23
|
-
|
|
24
|
-
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
25
|
-
const ROOT = join(HERE, "..", "..");
|
|
26
|
-
const CLASSIFY = join(ROOT, "pipeline", "lib", "classify-intent.sh");
|
|
27
|
-
const CASES = join(ROOT, "pipeline", "eval", "intent-cases.json");
|
|
28
|
-
|
|
29
|
-
const argv = process.argv.slice(2);
|
|
30
|
-
let minAccuracy = 0.95;
|
|
31
|
-
for (let i = 0; i < argv.length; i++) {
|
|
32
|
-
if (argv[i] === "--min") {
|
|
33
|
-
minAccuracy = Number(argv[i + 1]);
|
|
34
|
-
// A typo'd `--min` with no/NaN value would set the threshold to NaN, and
|
|
35
|
-
// `accuracy < NaN` is always false - silently turning the gate into a
|
|
36
|
-
// no-op. Fail loudly instead.
|
|
37
|
-
if (!Number.isFinite(minAccuracy) || minAccuracy < 0 || minAccuracy > 1) {
|
|
38
|
-
console.error(`eval-intent: --min needs a number in [0,1], got: ${argv[i + 1]}`);
|
|
39
|
-
process.exit(2);
|
|
40
|
-
}
|
|
41
|
-
}
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
function classify(input) {
|
|
45
|
-
try {
|
|
46
|
-
return execFileSync("bash", [CLASSIFY, input], { encoding: "utf8" }).trim();
|
|
47
|
-
} catch {
|
|
48
|
-
return "ERROR";
|
|
49
|
-
}
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
const { cases } = JSON.parse(readFileSync(CASES, "utf8"));
|
|
53
|
-
if (!Array.isArray(cases) || cases.length === 0) {
|
|
54
|
-
console.error("eval-intent: no cases found");
|
|
55
|
-
process.exit(2);
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
// Operationally-safe match: only TWO outcomes are dangerous, and they are what
|
|
59
|
-
// the gate measures -
|
|
60
|
-
// - a `task` read as `question` -> real work silently skipped
|
|
61
|
-
// - a `question` read as `task` -> a worktree spun up for nothing
|
|
62
|
-
// `ambiguous` proceeds as a task in Phase 0, so it is SAFE for a task/ambiguous
|
|
63
|
-
// case but UNSAFE for a question case (ambiguous still avoids the worktree, so
|
|
64
|
-
// it is acceptable for a question too - only `task` is the dangerous read).
|
|
65
|
-
function isSafe(expected, got) {
|
|
66
|
-
if (got === expected) return true;
|
|
67
|
-
if (expected === "task") return got === "ambiguous"; // proceeds as task anyway
|
|
68
|
-
if (expected === "ambiguous") return got === "task"; // both proceed to dev
|
|
69
|
-
if (expected === "question") return got === "ambiguous"; // still no worktree spun up
|
|
70
|
-
return false;
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
const mismatches = []; // operationally unsafe (gate-failing)
|
|
74
|
-
const exactMisses = []; // not exact, but operationally safe (informational)
|
|
75
|
-
let safe = 0;
|
|
76
|
-
let exact = 0;
|
|
77
|
-
for (const c of cases) {
|
|
78
|
-
const got = classify(c.input);
|
|
79
|
-
if (got === c.expected) exact++;
|
|
80
|
-
else exactMisses.push({ input: c.input, expected: c.expected, got });
|
|
81
|
-
if (isSafe(c.expected, got)) safe++;
|
|
82
|
-
else mismatches.push({ input: c.input, expected: c.expected, got, danger: true });
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
const accuracy = safe / cases.length;
|
|
86
|
-
const summary = {
|
|
87
|
-
total: cases.length,
|
|
88
|
-
safeMatches: safe,
|
|
89
|
-
safeAccuracy: Math.round(accuracy * 1000) / 1000,
|
|
90
|
-
exactMatches: exact,
|
|
91
|
-
exactAccuracy: Math.round((exact / cases.length) * 1000) / 1000,
|
|
92
|
-
minAccuracy,
|
|
93
|
-
dangerousMisclassifications: mismatches,
|
|
94
|
-
safeButInexact: exactMisses,
|
|
95
|
-
};
|
|
96
|
-
console.log(JSON.stringify(summary, null, 2));
|
|
97
|
-
|
|
98
|
-
if (accuracy < minAccuracy) {
|
|
99
|
-
console.error(`\neval-intent: safe accuracy ${(accuracy * 100).toFixed(1)}% < required ${(minAccuracy * 100).toFixed(0)}% (${mismatches.length} dangerous misclassification(s))`);
|
|
100
|
-
process.exit(1);
|
|
101
|
-
}
|
|
102
|
-
console.error(`\neval-intent: safe ${(accuracy * 100).toFixed(1)}% (${safe}/${cases.length}), exact ${(exact / cases.length * 100).toFixed(1)}% - pass`);
|
|
103
|
-
process.exit(0);
|