pi-dev-team 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/PORTING.md +134 -0
- package/README.md +207 -0
- package/UPSTREAM.json +64 -0
- package/agents/Explore.md +15 -0
- package/agents/a11y-review.md +118 -0
- package/agents/adr-author.md +70 -0
- package/agents/ai-provenance-review.md +120 -0
- package/agents/angular-reactivity-review.md +95 -0
- package/agents/arch-review.md +135 -0
- package/agents/architect.md +78 -0
- package/agents/autoship-batch-proposer.md +69 -0
- package/agents/claude-setup-review.md +136 -0
- package/agents/codebase-recon.md +184 -0
- package/agents/component-architecture-review.md +119 -0
- package/agents/concurrency-review.md +109 -0
- package/agents/correctness-review.md +290 -0
- package/agents/data-flow-tracer.md +120 -0
- package/agents/doc-review.md +165 -0
- package/agents/domain-review.md +136 -0
- package/agents/general-purpose.md +10 -0
- package/agents/gherkin-quality-critic.md +113 -0
- package/agents/js-fp-review.md +114 -0
- package/agents/mutation-kill.md +684 -0
- package/agents/naming-review.md +142 -0
- package/agents/orchestrator.md +339 -0
- package/agents/performance-review.md +105 -0
- package/agents/plan-review-acceptance.md +115 -0
- package/agents/plan-review-design.md +90 -0
- package/agents/plan-review-parallelization.md +84 -0
- package/agents/plan-review-strategic.md +96 -0
- package/agents/plan-review-ux.md +110 -0
- package/agents/platform-engineer.md +64 -0
- package/agents/product-manager.md +68 -0
- package/agents/progress-guardian.md +79 -0
- package/agents/qa-engineer.md +289 -0
- package/agents/quality-reviewer.md +132 -0
- package/agents/react-reactivity-review.md +102 -0
- package/agents/refactor-opportunity-review.md +128 -0
- package/agents/security-engineer.md +60 -0
- package/agents/security-review.md +218 -0
- package/agents/session-analysis.md +95 -0
- package/agents/software-engineer.md +105 -0
- package/agents/spec-compliance-review.md +100 -0
- package/agents/spec-reviewer.md +114 -0
- package/agents/structure-review.md +146 -0
- package/agents/tech-writer.md +84 -0
- package/agents/test-review.md +246 -0
- package/agents/test-smell-review.md +188 -0
- package/agents/token-efficiency-review.md +139 -0
- package/agents/ui-ux-designer.md +54 -0
- package/agents/vue-reactivity-review.md +95 -0
- package/bin/__pycache__/claudecpython-314.pyc +0 -0
- package/bin/claude +258 -0
- package/docs/upstream/.pages +1 -0
- package/docs/upstream/CHANGELOG.md +2586 -0
- package/docs/upstream/README.md +155 -0
- package/docs/upstream/agent-architecture.md +214 -0
- package/docs/upstream/agent_info.md +187 -0
- package/docs/upstream/artifact-migration.md +124 -0
- package/docs/upstream/code-intelligence-nudge.md +149 -0
- package/docs/upstream/code-review-process.md +294 -0
- package/docs/upstream/concurrent-use.md +73 -0
- package/docs/upstream/context-management.md +111 -0
- package/docs/upstream/developer-notes.md +280 -0
- package/docs/upstream/diagrams/architecture-overview.svg +101 -0
- package/docs/upstream/diagrams/review-dispatch.svg +139 -0
- package/docs/upstream/diagrams/team-agents.svg +128 -0
- package/docs/upstream/diagrams/test-improve-flow.svg +166 -0
- package/docs/upstream/diagrams/workflow-linear.svg +66 -0
- package/docs/upstream/diagrams/workflow-three-phase.svg +200 -0
- package/docs/upstream/eval-maintenance.md +95 -0
- package/docs/upstream/eval-running-guide.md +147 -0
- package/docs/upstream/eval-system.md +291 -0
- package/docs/upstream/session-review-oss-complements.md +75 -0
- package/docs/upstream/session-review.md +212 -0
- package/docs/upstream/skills.md +188 -0
- package/docs/upstream/team-structure.md +21 -0
- package/docs/upstream/telemetry-ci-access.md +129 -0
- package/docs/upstream/telemetry-repo-security.md +120 -0
- package/docs/upstream/test-evaluation.md +277 -0
- package/docs/upstream/test-improve.md +154 -0
- package/docs/upstream/triage-workflow.md +282 -0
- package/docs/upstream/workflows.md +289 -0
- package/extensions/dev-team/index.ts +539 -0
- package/extensions/dev-team/lib/agents.ts +272 -0
- package/extensions/dev-team/lib/ai-credits.ts +92 -0
- package/extensions/dev-team/lib/autocompact.ts +81 -0
- package/extensions/dev-team/lib/child-run.ts +102 -0
- package/extensions/dev-team/lib/config.ts +236 -0
- package/extensions/dev-team/lib/gh-command.ts +103 -0
- package/extensions/dev-team/lib/github-style.ts +307 -0
- package/extensions/dev-team/lib/hooks.ts +350 -0
- package/extensions/dev-team/lib/metrics.ts +115 -0
- package/extensions/dev-team/lib/safe-read.ts +49 -0
- package/extensions/dev-team/lib/session-files.ts +57 -0
- package/extensions/dev-team/lib/session-spend.ts +123 -0
- package/extensions/dev-team/lib/shell-scan.ts +205 -0
- package/extensions/dev-team/lib/skills.ts +213 -0
- package/extensions/dev-team/lib/subagent-render.ts +245 -0
- package/extensions/dev-team/lib/subagent-types.ts +164 -0
- package/extensions/dev-team/lib/subagent.ts +596 -0
- package/extensions/dev-team/lib/terminal-text.ts +54 -0
- package/extensions/dev-team/lib/tools-misc.ts +152 -0
- package/extensions/dev-team/lib/transcript.ts +110 -0
- package/extensions/dev-team/lib/trust.ts +52 -0
- package/extensions/dev-team/lib/usage-breakdown.ts +176 -0
- package/extensions/dev-team/lib/usage-chart.ts +153 -0
- package/extensions/dev-team/lib/usage-command.ts +107 -0
- package/extensions/dev-team/lib/usage-history.ts +203 -0
- package/extensions/dev-team/lib/usage-render.ts +225 -0
- package/extensions/dev-team/lib/usage-split-bar.ts +127 -0
- package/extensions/dev-team/lib/usage-state.ts +116 -0
- package/extensions/dev-team/lib/usage-text.ts +159 -0
- package/extensions/dev-team/lib/usage-view.ts +109 -0
- package/hooks/__pycache__/refactor_test_freeze_guard.cpython-314.pyc +0 -0
- package/hooks/agent_dispatch_ledger.py +190 -0
- package/hooks/autocompact_setup_nudge.py +99 -0
- package/hooks/bash_retry_guard.py +228 -0
- package/hooks/boundary_events_write_guard.py +352 -0
- package/hooks/code_intelligence_nudge.py +293 -0
- package/hooks/code_intelligence_turn_mark.py +317 -0
- package/hooks/codegraph_bootstrap.py +139 -0
- package/hooks/contract_version_guard.py +362 -0
- package/hooks/cost_meter.py +106 -0
- package/hooks/destructive-commands.json +62 -0
- package/hooks/destructive_guard.py +477 -0
- package/hooks/eval_compliance_check.py +440 -0
- package/hooks/guards.json +17 -0
- package/hooks/hooks.json +323 -0
- package/hooks/internal_double_gate.py +296 -0
- package/hooks/js_fp_review.py +212 -0
- package/hooks/knowledge_index.py +119 -0
- package/hooks/lib/__pycache__/artifact_paths.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/atomic_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/autocompact_config.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/boundary_events.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/doc_classification.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/gh_pr_create_detect.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/git_safe_diff.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/instrument_log.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/metrics_query.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/plugin_version.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/pre_commit_doc_classifier.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_agent_registry.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_corroboration.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_hash.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_verdicts.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stdin_json.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stryker_invocation.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/telemetry_consent.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/test_file_classify.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/token_efficiency_limits.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/verify_guard_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/xunit_v3_operator_gate.cpython-314.pyc +0 -0
- package/hooks/lib/agent_skill_hints.py +74 -0
- package/hooks/lib/artifact_paths.py +263 -0
- package/hooks/lib/atomic_state.py +557 -0
- package/hooks/lib/autocompact_config.py +103 -0
- package/hooks/lib/autoship_log.py +106 -0
- package/hooks/lib/banned_scripts_policy.py +51 -0
- package/hooks/lib/boundary_events.py +436 -0
- package/hooks/lib/build_knowledge_index.py +504 -0
- package/hooks/lib/build_skills_index.py +361 -0
- package/hooks/lib/build_state.py +116 -0
- package/hooks/lib/classify_ship_outcome.py +126 -0
- package/hooks/lib/config_changelog_schema.py +115 -0
- package/hooks/lib/cost_meter.py +955 -0
- package/hooks/lib/doc_classification.py +116 -0
- package/hooks/lib/gh_pr_create_detect.py +136 -0
- package/hooks/lib/git_safe_diff.py +123 -0
- package/hooks/lib/instrument_log.py +66 -0
- package/hooks/lib/iteration_journal_gate.py +197 -0
- package/hooks/lib/knowledge_index_paths.py +88 -0
- package/hooks/lib/mcp_json_repowise.py +177 -0
- package/hooks/lib/metrics_query.py +202 -0
- package/hooks/lib/minimal_yaml.py +434 -0
- package/hooks/lib/plugin_version.py +142 -0
- package/hooks/lib/pre_commit_detect.py +537 -0
- package/hooks/lib/pre_commit_doc_classifier.py +126 -0
- package/hooks/lib/pricing.py +118 -0
- package/hooks/lib/report_pdf.py +371 -0
- package/hooks/lib/review_agent_registry.py +142 -0
- package/hooks/lib/review_dispatch_ledger.py +101 -0
- package/hooks/lib/review_gate_corroboration.py +521 -0
- package/hooks/lib/review_gate_hash.py +252 -0
- package/hooks/lib/review_gate_normalized_hash.py +1115 -0
- package/hooks/lib/review_verdicts.py +301 -0
- package/hooks/lib/run_report.py +160 -0
- package/hooks/lib/skill_categories.yaml +125 -0
- package/hooks/lib/stdin_json.py +57 -0
- package/hooks/lib/stryker_invocation.py +102 -0
- package/hooks/lib/telemetry_consent.py +41 -0
- package/hooks/lib/telemetry_report.py +108 -0
- package/hooks/lib/test_file_classify.py +160 -0
- package/hooks/lib/token_efficiency_limits.py +51 -0
- package/hooks/lib/turn_identity.py +77 -0
- package/hooks/lib/verify_guard_state.py +110 -0
- package/hooks/lib/workflow_state.py +206 -0
- package/hooks/lib/xunit_v3_operator_gate.py +596 -0
- package/hooks/mcp_json_repowise_nudge.py +74 -0
- package/hooks/mutation_adapters/__init__.py +7 -0
- package/hooks/mutation_adapters/__pycache__/__init__.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/lib.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/mutmut.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/pitest.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker_net.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/lib.py +478 -0
- package/hooks/mutation_adapters/mutmut.py +188 -0
- package/hooks/mutation_adapters/pitest.py +266 -0
- package/hooks/mutation_adapters/stryker.py +157 -0
- package/hooks/mutation_adapters/stryker_net.py +264 -0
- package/hooks/mutation_gate.py +193 -0
- package/hooks/mutation_testing_smoke_gate.py +371 -0
- package/hooks/pending_review_notify.py +121 -0
- package/hooks/phase_marker.py +138 -0
- package/hooks/post_compact_state_reinject.py +180 -0
- package/hooks/post_format.py +115 -0
- package/hooks/pre_commit_knowledge_index.py +128 -0
- package/hooks/pre_commit_review.py +66 -0
- package/hooks/pre_pr_review.py +694 -0
- package/hooks/pre_tool_guard.py +405 -0
- package/hooks/py.sh +73 -0
- package/hooks/refactor-bash-write-patterns.json +29 -0
- package/hooks/refactor_test_bash_guard.py +253 -0
- package/hooks/refactor_test_freeze_guard.py +139 -0
- package/hooks/refactor_test_revert_guard.py +186 -0
- package/hooks/repo_review_nudge.py +287 -0
- package/hooks/review_verdict_recorder.py +464 -0
- package/hooks/scan_bash_command_for_banned_scripts.py +428 -0
- package/hooks/scan_worktree_for_banned_scripts.py +238 -0
- package/hooks/session_learning_trigger.py +248 -0
- package/hooks/skills_index.py +126 -0
- package/hooks/stryker_xunit_shim_guard.py +571 -0
- package/hooks/subagent_completion_guard.py +309 -0
- package/hooks/subagent_skill_context.py +139 -0
- package/hooks/task_completion_metrics.py +216 -0
- package/hooks/tdd_guard.py +229 -0
- package/hooks/telemetry.py +341 -0
- package/hooks/token_efficiency_review.py +194 -0
- package/hooks/verify_guard.py +183 -0
- package/hooks/verify_guard_edit_marker.py +73 -0
- package/hooks/version_check.py +173 -0
- package/knowledge/accepted-risks-schema.md +98 -0
- package/knowledge/adr-decision-criteria.md +64 -0
- package/knowledge/adversarial-review-protocol.md +139 -0
- package/knowledge/agent-registry.md +228 -0
- package/knowledge/agent-review-methodology.md +80 -0
- package/knowledge/ai-friendly-repo-guidelines.md +67 -0
- package/knowledge/architecture-assessment.md +96 -0
- package/knowledge/artifact-lifecycle.md +57 -0
- package/knowledge/cd-maturity-model.md +82 -0
- package/knowledge/cd-test-architecture.md +190 -0
- package/knowledge/ci-cd-file-scope.md +24 -0
- package/knowledge/codegraph-vs-graphify.md +192 -0
- package/knowledge/component-test-patterns.md +139 -0
- package/knowledge/database-change-management.md +80 -0
- package/knowledge/database-test-patterns.md +79 -0
- package/knowledge/decision-defaults.md +88 -0
- package/knowledge/dependency-breaking-techniques.md +116 -0
- package/knowledge/deployment-pipeline.md +86 -0
- package/knowledge/design-smells.md +122 -0
- package/knowledge/directory-enumeration.md +38 -0
- package/knowledge/domain-modeling.md +123 -0
- package/knowledge/evidence-bundle.md +90 -0
- package/knowledge/exploratory-testing-field-guide.md +122 -0
- package/knowledge/failure-routing.md +28 -0
- package/knowledge/fixture-construction.md +56 -0
- package/knowledge/frontend-component-architecture.md +139 -0
- package/knowledge/gherkin-quality-review-dispatch.md +135 -0
- package/knowledge/index.json +6766 -0
- package/knowledge/internal-collaborator-doubling.md +101 -0
- package/knowledge/legacy-test-strategy.md +71 -0
- package/knowledge/long-run-waiting.md +66 -0
- package/knowledge/microservice-testing.md +71 -0
- package/knowledge/model-pricing.json +23 -0
- package/knowledge/mutation-score-formulas.md +60 -0
- package/knowledge/object-calisthenics.md +147 -0
- package/knowledge/oracle-provenance.md +94 -0
- package/knowledge/orchestrator-script-implementation.md +185 -0
- package/knowledge/owasp-detection.md +148 -0
- package/knowledge/plan-review-rubric.md +56 -0
- package/knowledge/proxy-connectivity.md +62 -0
- package/knowledge/reactive-effect-patterns.md +73 -0
- package/knowledge/recon-inventory-excludes.txt +32 -0
- package/knowledge/references/bdd-value-guide.md +61 -0
- package/knowledge/references/csharp-http-client-testing.md +264 -0
- package/knowledge/release-strategies.md +74 -0
- package/knowledge/report-output-location.md +117 -0
- package/knowledge/report-pdf-integration.md +63 -0
- package/knowledge/report-print.css +129 -0
- package/knowledge/report-template.md +114 -0
- package/knowledge/report-to-pdf.md +69 -0
- package/knowledge/request-processing-flow.md +63 -0
- package/knowledge/result-verification.md +52 -0
- package/knowledge/review-agent-output-contract.md +121 -0
- package/knowledge/review-lens-classification.md +113 -0
- package/knowledge/review-rubric.md +62 -0
- package/knowledge/review-template.md +104 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/negative.js +1 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/positive.js +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/negative.py +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/positive.py +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/positive.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/negative.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/negative.cs +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/positive.cs +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/negative.js +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/negative.java +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/positive.java +1 -0
- package/knowledge/schemas/disposition-register-v1.json +65 -0
- package/knowledge/schemas/recon-envelope-v1.json +198 -0
- package/knowledge/schemas/unified-finding-v1.json +72 -0
- package/knowledge/security-primitives-contract.md +301 -0
- package/knowledge/security-review-rule-map.yaml +107 -0
- package/knowledge/skills-registry.md +72 -0
- package/knowledge/task-size-classifier.md +103 -0
- package/knowledge/telemetry-schema.md +881 -0
- package/knowledge/test-automation-maturity.md +56 -0
- package/knowledge/test-automation-principles.md +71 -0
- package/knowledge/test-cadence-tradeoffs.md +68 -0
- package/knowledge/test-doubles.md +105 -0
- package/knowledge/test-file-indicators.md +22 -0
- package/knowledge/test-layer-gates.md +35 -0
- package/knowledge/test-matrix-examples/django-batch.md +24 -0
- package/knowledge/test-matrix-examples/dotnet-grpc-fronting-api.md +90 -0
- package/knowledge/test-matrix-examples/dotnet-http-consumer.md +131 -0
- package/knowledge/test-matrix-examples/react-node-spa.md +24 -0
- package/knowledge/test-matrix-examples/spring-boot-service.md +25 -0
- package/knowledge/test-matrix-examples/ssr-htmx.md +24 -0
- package/knowledge/test-organization.md +70 -0
- package/knowledge/test-pyramid.md +84 -0
- package/knowledge/test-refactoring.md +67 -0
- package/knowledge/test-review-division-of-labor.md +85 -0
- package/knowledge/test-smells.md +80 -0
- package/knowledge/test-stack-profiles/bdd-frameworks.md +235 -0
- package/knowledge/test-stack-profiles/django.md +13 -0
- package/knowledge/test-stack-profiles/dotnet.md +18 -0
- package/knowledge/test-stack-profiles/go.md +16 -0
- package/knowledge/test-stack-profiles/node.md +16 -0
- package/knowledge/test-stack-profiles/react.md +12 -0
- package/knowledge/test-stack-profiles/spring-boot.md +16 -0
- package/knowledge/test-stack-profiles/ssr-htmx.md +14 -0
- package/knowledge/test-stack-profiles/vue.md +12 -0
- package/knowledge/test-strategy.md +70 -0
- package/knowledge/testability-patterns.md +240 -0
- package/knowledge/testing-quadrants.md +44 -0
- package/knowledge/testing-techniques/approval.md +15 -0
- package/knowledge/testing-techniques/chaos.md +17 -0
- package/knowledge/testing-techniques/fuzz.md +15 -0
- package/knowledge/testing-techniques/property-based.md +15 -0
- package/knowledge/testing-techniques/schema-validation.md +15 -0
- package/knowledge/testing-techniques/screenshot.md +15 -0
- package/knowledge/three-phase-workflow.md +198 -0
- package/knowledge/value-patterns.md +55 -0
- package/knowledge/verification-mode.md +116 -0
- package/knowledge/virtual-service-libraries.md +75 -0
- package/knowledge/wave-consolidation-guidance.md +21 -0
- package/overrides/agents/Explore.md +15 -0
- package/overrides/agents/general-purpose.md +10 -0
- package/overrides/notes/autoship.md +6 -0
- package/overrides/notes/issues-from-assessment.md +3 -0
- package/overrides/notes/issues-from-plan.md +3 -0
- package/overrides/notes/mutation-night-watch.md +3 -0
- package/overrides/notes/mutation-testing.md +3 -0
- package/overrides/notes/pr.md +7 -0
- package/overrides/notes/project-init.md +6 -0
- package/overrides/notes/setup.md +13 -0
- package/overrides/notes/specs.md +3 -0
- package/overrides/skills/headless-run/SKILL.md +45 -0
- package/overrides/skills/upgrade/SKILL.md +30 -0
- package/overrides/skills/version/SKILL.md +25 -0
- package/package.json +36 -0
- package/scripts/authoring_digest.py +93 -0
- package/scripts/autoship_discover.py +121 -0
- package/scripts/autoship_group.py +409 -0
- package/scripts/autoship_proposals.py +494 -0
- package/scripts/autoship_queue.py +291 -0
- package/scripts/autoship_reclaim.py +495 -0
- package/scripts/build_jobs.py +108 -0
- package/scripts/build_rollback_point.py +240 -0
- package/scripts/build_slice_scope.py +157 -0
- package/scripts/build_wave.py +109 -0
- package/scripts/build_wave_reconcile.py +252 -0
- package/scripts/build_worktree_baseref.py +113 -0
- package/scripts/check_agent_scope.py +117 -0
- package/scripts/check_agent_tool_mapping.py +213 -0
- package/scripts/check_review_agent_mcp_tools.py +317 -0
- package/scripts/check_security_assessment_mcp_tools.py +165 -0
- package/scripts/checkpoint_abort.py +502 -0
- package/scripts/claude_setup_review.py +438 -0
- package/scripts/codebase_recon.py +556 -0
- package/scripts/coverage_config.py +623 -0
- package/scripts/coverage_delta_steering.py +330 -0
- package/scripts/coverage_discovery_dotnet.py +315 -0
- package/scripts/coverage_discovery_java.py +742 -0
- package/scripts/coverage_discovery_js.py +546 -0
- package/scripts/coverage_gap_ranking.py +556 -0
- package/scripts/coverage_readiness.py +455 -0
- package/scripts/coverage_report_parse.py +521 -0
- package/scripts/detect_bdd_convention.py +252 -0
- package/scripts/eval_ablation.py +376 -0
- package/scripts/gherkin_analysis_coverage_gate.py +306 -0
- package/scripts/gherkin_cross_feature_duplicate_titles_gate.py +173 -0
- package/scripts/gherkin_effectiveness_rollup.py +238 -0
- package/scripts/gherkin_failure_path_gate.py +206 -0
- package/scripts/gherkin_feature_merge.py +720 -0
- package/scripts/gherkin_stub_gate.py +163 -0
- package/scripts/gherkin_stub_merge.py +479 -0
- package/scripts/git_origin_host.py +88 -0
- package/scripts/install-java-static-analysis.py +110 -0
- package/scripts/issue_deps.py +74 -0
- package/scripts/lib/_bdd_markers.py +28 -0
- package/scripts/lib/_gherkin_text.py +93 -0
- package/scripts/lib/_vendored_tree.py +70 -0
- package/scripts/lib/autoship_state.py +397 -0
- package/scripts/lib/claude_md_guard.py +226 -0
- package/scripts/lib/deterministic_recon.py +446 -0
- package/scripts/lib/mcp_tool_grants.py +211 -0
- package/scripts/lib/plan_parse.py +386 -0
- package/scripts/lib/review_result.py +84 -0
- package/scripts/lib/review_roster.py +86 -0
- package/scripts/lib/session_log/__init__.py +34 -0
- package/scripts/lib/session_log/__pycache__/__init__.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/__pycache__/records.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/classify.py +231 -0
- package/scripts/lib/session_log/corrections.py +194 -0
- package/scripts/lib/session_log/discovery.py +108 -0
- package/scripts/lib/session_log/records.py +218 -0
- package/scripts/lib/session_log/redact.py +76 -0
- package/scripts/lib/session_log/signals.py +373 -0
- package/scripts/lib/session_report_downstream.py +614 -0
- package/scripts/lib/session_report_maintainer.py +1273 -0
- package/scripts/lib/session_report_shared.py +262 -0
- package/scripts/lib/settings_hook_guard.py +157 -0
- package/scripts/lib/slug.py +33 -0
- package/scripts/lib/stub_extractors/__init__.py +82 -0
- package/scripts/lib/stub_extractors/_common.py +328 -0
- package/scripts/lib/stub_extractors/csharp.py +19 -0
- package/scripts/lib/stub_extractors/go.py +173 -0
- package/scripts/lib/stub_extractors/java.py +18 -0
- package/scripts/lib/stub_extractors/jsts.py +126 -0
- package/scripts/mutation_stack_sections.py +149 -0
- package/scripts/mutation_yield_steering.py +345 -0
- package/scripts/orchestrator.py +895 -0
- package/scripts/plan_gherkin_export.py +227 -0
- package/scripts/plan_waves.py +208 -0
- package/scripts/pr_close_keyword_lint.py +108 -0
- package/scripts/progress_guardian.py +888 -0
- package/scripts/recon_inventory.py +273 -0
- package/scripts/review_findings_log.py +93 -0
- package/scripts/run_invariants.py +124 -0
- package/scripts/select_lenses.py +640 -0
- package/scripts/session_report.py +486 -0
- package/scripts/set_autocompact_env.py +221 -0
- package/scripts/ship_resume_guard.py +135 -0
- package/scripts/ship_review_gate.py +63 -0
- package/scripts/specs_convention_marker.py +103 -0
- package/scripts/test_improve_resume.py +277 -0
- package/scripts/test_review_mechanics.py +958 -0
- package/scripts/token_efficiency_review.py +322 -0
- package/scripts/verdict_scope.py +285 -0
- package/scripts/verify_gherkin_quality_critic_isolation.py +296 -0
- package/scripts/verify_tier.py +157 -0
- package/skills/adr-tools/SKILL.md +118 -0
- package/skills/agent-readiness/SKILL.md +105 -0
- package/skills/agent-readiness/ai_friendly_analyzers.py +326 -0
- package/skills/agent-readiness/scanner.py +441 -0
- package/skills/agent-readiness/scorecard.yaml +88 -0
- package/skills/api-design/SKILL.md +115 -0
- package/skills/apply-fixes/SKILL.md +171 -0
- package/skills/apply-test-doubles/SKILL.md +321 -0
- package/skills/artifact-lifecycle/SKILL.md +127 -0
- package/skills/autoship/SKILL.md +1124 -0
- package/skills/benchmark/SKILL.md +105 -0
- package/skills/branch-workflow/SKILL.md +89 -0
- package/skills/browse/SKILL.md +184 -0
- package/skills/browser-testing/SKILL.md +62 -0
- package/skills/browser-testing/references/playwright-patterns.md +216 -0
- package/skills/build/SKILL.md +422 -0
- package/skills/build/references/static-self-heal.md +245 -0
- package/skills/careful/SKILL.md +72 -0
- package/skills/cd-test-architecture/SKILL.md +371 -0
- package/skills/ci-debugging/SKILL.md +105 -0
- package/skills/co-evolution-audit/SKILL.md +269 -0
- package/skills/code-review/SKILL.md +1015 -0
- package/skills/code-review/examples/aggregated-sample.json +56 -0
- package/skills/code-review/examples/sample-report.md +41 -0
- package/skills/code-review/output-format.md +478 -0
- package/skills/code-review/scripts/activation.py +86 -0
- package/skills/code-review/scripts/change_impact.py +357 -0
- package/skills/code-review/scripts/change_shape.py +372 -0
- package/skills/code-review/scripts/change_size.py +212 -0
- package/skills/code-review/scripts/changed_file_list.py +141 -0
- package/skills/code-review/scripts/closing_pass.py +187 -0
- package/skills/code-review/scripts/consolidate.py +277 -0
- package/skills/code-review/scripts/contract_failure_report.py +185 -0
- package/skills/code-review/scripts/dispatch_reconcile.py +66 -0
- package/skills/code-review/scripts/dispatch_waves.py +164 -0
- package/skills/code-review/scripts/finding_signature.py +446 -0
- package/skills/code-review/scripts/ledger.py +283 -0
- package/skills/code-review/scripts/partition.py +169 -0
- package/skills/code-review/scripts/render_tiered_findings.py +274 -0
- package/skills/code-review/scripts/repo_invariants.py +1066 -0
- package/skills/code-review/scripts/review_context_pack.py +306 -0
- package/skills/code-review/scripts/review_round_log.py +345 -0
- package/skills/code-review/scripts/review_value_coverage.py +297 -0
- package/skills/code-review/scripts/validate_review_output.py +467 -0
- package/skills/code-review/sliced-mode.md +205 -0
- package/skills/competitive-analysis/SKILL.md +191 -0
- package/skills/context-loading-protocol/SKILL.md +157 -0
- package/skills/continue/SKILL.md +90 -0
- package/skills/cost-report/SKILL.md +178 -0
- package/skills/coverage-baseline/SKILL.md +335 -0
- package/skills/coverage-baseline/references/multi-project-discovery.md +202 -0
- package/skills/coverage-delta/SKILL.md +181 -0
- package/skills/coverage-delta/references/mutation-gate.md +70 -0
- package/skills/design-doc/SKILL.md +95 -0
- package/skills/design-interrogation/SKILL.md +89 -0
- package/skills/design-it-twice/SKILL.md +91 -0
- package/skills/docker-image-audit/SKILL.md +108 -0
- package/skills/docker-image-audit/references/install-guide.md +64 -0
- package/skills/docker-image-audit/references/report-template.md +73 -0
- package/skills/docker-image-create/SKILL.md +185 -0
- package/skills/domain-analysis/SKILL.md +183 -0
- package/skills/domain-driven-design/SKILL.md +194 -0
- package/skills/exploratory-testing/SKILL.md +108 -0
- package/skills/explore/SKILL.md +51 -0
- package/skills/farley-score/SKILL.md +165 -0
- package/skills/feature-file-validation/SKILL.md +78 -0
- package/skills/feature-file-validation/references/validation-rules.md +115 -0
- package/skills/feedback-learning/SKILL.md +414 -0
- package/skills/fix/SKILL.md +450 -0
- package/skills/freeze/SKILL.md +68 -0
- package/skills/frontend-architecture/SKILL.md +113 -0
- package/skills/gherkin-derive/SKILL.md +630 -0
- package/skills/gherkin-public/SKILL.md +266 -0
- package/skills/governance-compliance/SKILL.md +150 -0
- package/skills/guard/SKILL.md +75 -0
- package/skills/handoff/SKILL.md +139 -0
- package/skills/handoff/references/summary-templates.md +242 -0
- package/skills/harness-audit/SKILL.md +751 -0
- package/skills/harness-audit/scripts/lesson_validate.py +386 -0
- package/skills/harness-audit/scripts/redundancy_criterion.py +188 -0
- package/skills/headless-run/SKILL.md +45 -0
- package/skills/headless-run/scripts/isolated_dispatch.py +381 -0
- package/skills/help/SKILL.md +72 -0
- package/skills/hexagonal-architecture/SKILL.md +85 -0
- package/skills/human-oversight-protocol/SKILL.md +224 -0
- package/skills/issues-from-assessment/SKILL.md +223 -0
- package/skills/issues-from-plan/SKILL.md +133 -0
- package/skills/legacy-code/SKILL.md +132 -0
- package/skills/mermaid-diagramming/SKILL.md +120 -0
- package/skills/mutation-night-watch/SKILL.md +154 -0
- package/skills/mutation-night-watch/references/scheduling.md +135 -0
- package/skills/mutation-testing/SKILL.md +396 -0
- package/skills/mutation-testing/references/languages/csharp-stryker-net.md +676 -0
- package/skills/mutation-testing/references/languages/go-go-mutesting.md +95 -0
- package/skills/mutation-testing/references/languages/java-pitest.md +77 -0
- package/skills/mutation-testing/references/languages/javascript-stryker.md +188 -0
- package/skills/mutation-testing/references/languages/python-mutmut.md +97 -0
- package/skills/mutation-testing/references/time-estimation.md +34 -0
- package/skills/mutation-testing/references/tool-detection.md +15 -0
- package/skills/mutation-testing/references/workflow-callers.md +23 -0
- package/skills/mutation-testing/scripts/__pycache__/xunit_v3_feature_detector.cpython-314.pyc +0 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_slice_runner.py +635 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_status_loop.py +525 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_wrapper.py +681 -0
- package/skills/mutation-testing/scripts/mutation_baseline_reuse.py +292 -0
- package/skills/mutation-testing/scripts/mutation_exclude_policy.py +268 -0
- package/skills/mutation-testing/scripts/mutation_feasibility_gate.py +463 -0
- package/skills/mutation-testing/scripts/mutation_kill_headless.py +331 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert.py +199 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert_python.py +150 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop.py +869 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop_python.py +949 -0
- package/skills/mutation-testing/scripts/mutation_kill_retry.py +592 -0
- package/skills/mutation-testing/scripts/mutation_kill_shared.py +620 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch.py +462 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch_stacks.py +425 -0
- package/skills/mutation-testing/scripts/mutation_report.py +743 -0
- package/skills/mutation-testing/scripts/mutation_report_cli.py +175 -0
- package/skills/mutation-testing/scripts/mutation_safety_gate.py +69 -0
- package/skills/mutation-testing/scripts/stryker_shard_pipeline.py +847 -0
- package/skills/mutation-testing/scripts/stryker_shard_setup.py +440 -0
- package/skills/mutation-testing/scripts/stryker_timeout_retry.py +142 -0
- package/skills/mutation-testing/scripts/xunit_v3_feature_detector.py +341 -0
- package/skills/performance-benchmark/SKILL.md +174 -0
- package/skills/performance-benchmark/examples/report-format.md +43 -0
- package/skills/performance-benchmark/references/benchmark-script.md +169 -0
- package/skills/performance-metrics/SKILL.md +265 -0
- package/skills/plan/SKILL.md +199 -0
- package/skills/plan/references/gherkin-persistence.md +43 -0
- package/skills/plan/references/plan-template.md +182 -0
- package/skills/pr/SKILL.md +289 -0
- package/skills/pr/scripts/gate_retry_state.py +368 -0
- package/skills/project-init/README.md +141 -0
- package/skills/project-init/SKILL.md +1197 -0
- package/skills/project-init/evals/evals.json +200 -0
- package/skills/project-init/references/capability-tools.md +55 -0
- package/skills/project-init/references/configs.md +221 -0
- package/skills/property-based-testing/SKILL.md +121 -0
- package/skills/property-based-testing/fixtures/invariant_fixture.py +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/README.md +42 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/README.md +263 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.js +12147 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.js +12011 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/rolldown-runtime-D7D4PA-g.js +13 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/package.json +94 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/README.md +168 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.js +38 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.js +18 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.js +22 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.js +134 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.js +37 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.js +17 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.js +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.js +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.js +44 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.js +90 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.js +78 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.js +8 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.js +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.js +46 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.js +92 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.js +82 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.js +10 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.js +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/package.json +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package-lock.json +1179 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package.json +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.js +29 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.properties.test.js +16 -0
- package/skills/property-based-testing/fixtures/no_property_fixture.py +10 -0
- package/skills/property-based-testing/fixtures/roundtrip_fixture.py +16 -0
- package/skills/property-based-testing/references/languages/javascript.md +54 -0
- package/skills/property-based-testing/scripts/detect_and_dispatch.py +80 -0
- package/skills/property-based-testing/scripts/hypothesis_scaffold.py +276 -0
- package/skills/proxy-resilience/SKILL.md +84 -0
- package/skills/quality-gate-pipeline/SKILL.md +184 -0
- package/skills/quality-targets-converge/SKILL.md +254 -0
- package/skills/repo-review/SKILL.md +159 -0
- package/skills/report-pdf/SKILL.md +66 -0
- package/skills/review/SKILL.md +47 -0
- package/skills/review-agent/SKILL.md +152 -0
- package/skills/review-summary/SKILL.md +73 -0
- package/skills/run-report/SKILL.md +70 -0
- package/skills/semantic-duplication-scan/SKILL.md +337 -0
- package/skills/semantic-scan/SKILL.md +53 -0
- package/skills/semgrep-analyze/SKILL.md +139 -0
- package/skills/setup/SKILL.md +1122 -0
- package/skills/ship/SKILL.md +240 -0
- package/skills/source-verification/SKILL.md +210 -0
- package/skills/source-verification/scripts/claim_extractor.py +155 -0
- package/skills/specs/.size-baseline.json +4 -0
- package/skills/specs/SKILL.md +243 -0
- package/skills/specs/references/completeness-checklist.md +83 -0
- package/skills/specs/references/extraction.md +58 -0
- package/skills/specs/references/glossary.md +59 -0
- package/skills/specs/references/persistence.md +115 -0
- package/skills/specs/references/predictability-check.md +77 -0
- package/skills/static-analysis-integration/SKILL.md +235 -0
- package/skills/static-analysis-integration/adapters/_envelope.py +26 -0
- package/skills/static-analysis-integration/adapters/jscpd-adapter.py +66 -0
- package/skills/static-analysis-integration/adapters/lizard-adapter.py +81 -0
- package/skills/static-analysis-integration/adapters/mypy-adapter.py +50 -0
- package/skills/static-analysis-integration/adapters/mypy-src-layout.py +93 -0
- package/skills/static-analysis-integration/adapters/security-review-adapter.py +212 -0
- package/skills/static-analysis-integration/maintenance.md +23 -0
- package/skills/static-analysis-integration/references/language-setup.md +228 -0
- package/skills/static-analysis-integration/references/sarif-parser.md +124 -0
- package/skills/static-analysis-integration/references/security-review-adapter.md +118 -0
- package/skills/static-analysis-integration/references/tool-configs.md +617 -0
- package/skills/static-analysis-integration/rulesets/pmd-quickstart.xml +24 -0
- package/skills/stryker-xunit-v2-shim/SKILL.md +274 -0
- package/skills/stryker-xunit-v2-shim/references/shim-howto.md +256 -0
- package/skills/stryker-xunit-v2-shim/scripts/generate_shim.py +143 -0
- package/skills/systematic-debugging/SKILL.md +130 -0
- package/skills/telemetry/SKILL.md +75 -0
- package/skills/test-audit-disable/SKILL.md +129 -0
- package/skills/test-design/SKILL.md +177 -0
- package/skills/test-design/scripts/__pycache__/internal_double_detector.cpython-314.pyc +0 -0
- package/skills/test-design/scripts/internal_double_detector.py +631 -0
- package/skills/test-design-advisor/SKILL.md +166 -0
- package/skills/test-driven-development/SKILL.md +169 -0
- package/skills/test-health/SKILL.md +262 -0
- package/skills/test-improve/SKILL.md +239 -0
- package/skills/test-improve/references/phase-0-approach-contract.md +228 -0
- package/skills/test-improve/references/phase-1-analyze.md +131 -0
- package/skills/test-improve/references/phase-2-baseline.md +121 -0
- package/skills/test-improve/references/phase-3-derive-gherkin.md +53 -0
- package/skills/test-improve/references/phase-4-plan-fixes.md +34 -0
- package/skills/test-improve/references/phase-5-improve.md +215 -0
- package/skills/test-improve/references/phase-6-refactor-decision.md +45 -0
- package/skills/test-improve/references/phase-7-refactor.md +44 -0
- package/skills/test-improve/references/phase-8-validate.md +66 -0
- package/skills/test-improve/references/phase-9-close-out-prompt.md +11 -0
- package/skills/test-improve/references/phase-9-report.md +62 -0
- package/skills/test-improve/references/review-loop.md +92 -0
- package/skills/test-improve/templates/executive-summary.md +123 -0
- package/skills/threat-modeling/SKILL.md +108 -0
- package/skills/triage/SKILL.md +211 -0
- package/skills/ubiquitous-language/SKILL.md +192 -0
- package/skills/ubiquitous-language/scripts/collect_domain_signals.py +300 -0
- package/skills/unfreeze/SKILL.md +37 -0
- package/skills/upgrade/SKILL.md +31 -0
- package/skills/upgrade/scripts/check_version_drift.py +113 -0
- package/skills/upgrade/scripts/enable_autoupdate.py +149 -0
- package/skills/version/SKILL.md +25 -0
- package/sync/__pycache__/sync_upstream.cpython-314.pyc +0 -0
- package/sync/sync_upstream.py +293 -0
- package/templates/ACCEPTED-RISKS.md.tmpl +46 -0
- package/templates/agents/agent-template.md +151 -0
- package/templates/agents/angular-testing.md +66 -0
- package/templates/agents/csharp-quality.md +63 -0
- package/templates/agents/esm-enforcer.md +52 -0
- package/templates/agents/front-end-testing.md +65 -0
- package/templates/agents/go-quality.md +65 -0
- package/templates/agents/python-quality.md +62 -0
- package/templates/agents/react-testing.md +61 -0
- package/templates/agents/ts-enforcer.md +60 -0
- package/templates/agents/twelve-factor-audit.md +49 -0
- package/tools/entropy-check.py +250 -0
- package/tools/model-hash-verify.py +213 -0
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""detect_bdd_convention.py — detect a target project's BDD convention.
|
|
3
|
+
|
|
4
|
+
Deterministic probe `/plan` shells to at plan-creation time, reporting
|
|
5
|
+
where derived .feature files should land:
|
|
6
|
+
|
|
7
|
+
{"signal": "feature-files" | "manifest" | "none",
|
|
8
|
+
"framework": <name or null>,
|
|
9
|
+
"dir": <repo-relative destination or null>}
|
|
10
|
+
|
|
11
|
+
Precedence: existing .feature files > BDD dependency in a manifest > none.
|
|
12
|
+
Detection is conservative — a false negative (no signal, which prompts the
|
|
13
|
+
operator) is preferred over a false positive (the wrong directory), so any
|
|
14
|
+
ambiguity (multiple unrelated .feature roots) reports "none". Vendored and
|
|
15
|
+
generated trees (node_modules/, vendor/, dist/, build/, .git/, virtualenvs)
|
|
16
|
+
are never treated as a signal. A common .feature directory is only a signal
|
|
17
|
+
when at least one of its path components is a conventional BDD scenario
|
|
18
|
+
directory name (see `_CONVENTIONAL_FEATURE_DIR_NAMES`, issue #1462) — the
|
|
19
|
+
signal is then the ancestor path up to and including the deepest such
|
|
20
|
+
component (not necessarily the full common directory, which may sit deeper
|
|
21
|
+
under an incidentally-nested leaf). Otherwise it reports "none", since a
|
|
22
|
+
single incidental `.feature` root elsewhere in the repo (e.g. a
|
|
23
|
+
narrowly-scoped test fixture directory) is not evidence of a repo-wide
|
|
24
|
+
Gherkin convention.
|
|
25
|
+
|
|
26
|
+
Stdlib-only.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import argparse
|
|
32
|
+
import fnmatch
|
|
33
|
+
import json
|
|
34
|
+
import re
|
|
35
|
+
import sys
|
|
36
|
+
from collections.abc import Iterable
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import NamedTuple
|
|
39
|
+
|
|
40
|
+
_HERE = Path(__file__).resolve().parent
|
|
41
|
+
sys.path.insert(0, str(_HERE / "lib"))
|
|
42
|
+
|
|
43
|
+
from _vendored_tree import iter_files as _iter_files
|
|
44
|
+
|
|
45
|
+
# Recognized final-path-component names for a genuine BDD scenario directory
|
|
46
|
+
# (case-insensitive). Mirrors MANIFEST_RULES' own "features"/"Features"
|
|
47
|
+
# convention below. A computed common .feature directory whose last
|
|
48
|
+
# component isn't one of these is treated as no signal rather than a false
|
|
49
|
+
# positive — see issue #1462 (evals/skills/ in this repo has a narrowly-scoped
|
|
50
|
+
# documented purpose, not a general Gherkin destination, but was the only
|
|
51
|
+
# .feature root and so was wrongly returned as "the" convention).
|
|
52
|
+
_CONVENTIONAL_FEATURE_DIR_NAMES = frozenset(
|
|
53
|
+
{"features", "feature", "specs", "spec", "bdd", "acceptance"}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _common_directory(paths: Iterable[Path]) -> Path:
|
|
58
|
+
"""Longest common ancestor of relative paths ('.' when they share none)."""
|
|
59
|
+
common: list[str] = []
|
|
60
|
+
for components in zip(*(p.parts for p in paths)):
|
|
61
|
+
if len(set(components)) != 1:
|
|
62
|
+
break
|
|
63
|
+
common.append(components[0])
|
|
64
|
+
return Path(*common) if common else Path(".")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def scan_feature_dir(root: Path) -> str | None:
|
|
68
|
+
"""Repo-relative common directory of the project's .feature files.
|
|
69
|
+
|
|
70
|
+
None when no .feature file exists outside vendored trees, when the
|
|
71
|
+
files' common ancestor is the project root itself — which covers both
|
|
72
|
+
multiple unrelated roots and root-level orphans (conservative: prompt
|
|
73
|
+
rather than guess) — or when NO component of that common ancestor is a
|
|
74
|
+
conventional BDD scenario directory name (see
|
|
75
|
+
`_CONVENTIONAL_FEATURE_DIR_NAMES`, issue #1462): the only existing
|
|
76
|
+
`.feature` root in a repo is not by itself evidence of a general Gherkin
|
|
77
|
+
convention.
|
|
78
|
+
|
|
79
|
+
When a component DOES match (e.g. `features` in `features/checkout`, or
|
|
80
|
+
in `src/test/resources/features`), the signal is the ancestor path up to
|
|
81
|
+
and including the DEEPEST matching component — not necessarily the full
|
|
82
|
+
common directory. This matters when every `.feature` file happens to sit
|
|
83
|
+
under one nested leaf of an otherwise-conventional root (e.g. only
|
|
84
|
+
`features/checkout/*.feature` exists, no sibling directly under
|
|
85
|
+
`features/` itself): the common ancestor is `features/checkout`, whose
|
|
86
|
+
*final* component (`checkout`) isn't conventional, but `features` — a
|
|
87
|
+
real ancestor — is. Checking only the final component would wrongly
|
|
88
|
+
report no signal here. Taking the deepest (not shallowest) matching
|
|
89
|
+
component matters too: for `specs/features`, both `specs` and `features`
|
|
90
|
+
are individually conventional names, but the deepest match (`features`)
|
|
91
|
+
is the one that yields the existing, expected `specs/features`
|
|
92
|
+
destination rather than truncating to the shallower `specs`.
|
|
93
|
+
"""
|
|
94
|
+
parents = sorted(
|
|
95
|
+
{
|
|
96
|
+
found.parent.relative_to(root)
|
|
97
|
+
for found in _iter_files(root)
|
|
98
|
+
if found.suffix == ".feature"
|
|
99
|
+
}
|
|
100
|
+
)
|
|
101
|
+
if not parents:
|
|
102
|
+
return None
|
|
103
|
+
common = _common_directory(parents)
|
|
104
|
+
if common == Path("."):
|
|
105
|
+
return None
|
|
106
|
+
lowered_parts = [part.lower() for part in common.parts]
|
|
107
|
+
conventional_positions = [
|
|
108
|
+
index
|
|
109
|
+
for index, part in enumerate(lowered_parts)
|
|
110
|
+
if part in _CONVENTIONAL_FEATURE_DIR_NAMES
|
|
111
|
+
]
|
|
112
|
+
if not conventional_positions:
|
|
113
|
+
return None
|
|
114
|
+
return Path(*common.parts[: max(conventional_positions) + 1]).as_posix()
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class ManifestRule(NamedTuple):
|
|
118
|
+
"""One supported BDD stack: which manifest declares it, and where its
|
|
119
|
+
.feature files canonically live (repo-relative)."""
|
|
120
|
+
|
|
121
|
+
framework: str
|
|
122
|
+
manifest_globs: tuple[str, ...] # fnmatch patterns on the manifest file name
|
|
123
|
+
tokens: tuple[str, ...] # dependency tokens that signal the framework
|
|
124
|
+
destination: str | None # None => CSPROJ_FEATURES_SUBDIR under the manifest's dir
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# Community convention for Reqnroll/SpecFlow: Features/ under the test csproj.
|
|
128
|
+
CSPROJ_FEATURES_SUBDIR = "Features"
|
|
129
|
+
|
|
130
|
+
# One row per stack; keep in sync with
|
|
131
|
+
# knowledge/test-stack-profiles/bdd-frameworks.md (guarded by
|
|
132
|
+
# tests/scripts/test_detect_bdd_convention.py::TestMappingDocSync).
|
|
133
|
+
MANIFEST_RULES: tuple[ManifestRule, ...] = (
|
|
134
|
+
ManifestRule("cucumber-js", ("package.json",), ("@cucumber/cucumber",), "features"),
|
|
135
|
+
ManifestRule(
|
|
136
|
+
"pytest-bdd", ("pyproject.toml", "requirements*.txt"), ("pytest-bdd",), "features"
|
|
137
|
+
),
|
|
138
|
+
ManifestRule(
|
|
139
|
+
"behave", ("pyproject.toml", "requirements*.txt"), ("behave",), "features"
|
|
140
|
+
),
|
|
141
|
+
ManifestRule("reqnroll", ("*.csproj",), ("Reqnroll",), None),
|
|
142
|
+
ManifestRule("specflow", ("*.csproj",), ("SpecFlow",), None),
|
|
143
|
+
ManifestRule(
|
|
144
|
+
"cucumber-jvm",
|
|
145
|
+
("pom.xml", "build.gradle", "build.gradle.kts"),
|
|
146
|
+
("io.cucumber",),
|
|
147
|
+
"src/test/resources/features",
|
|
148
|
+
),
|
|
149
|
+
ManifestRule("godog", ("go.mod",), ("godog",), "features"),
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _package_json_declares(path: Path, token: str) -> bool:
|
|
154
|
+
"""True when `token` is a dependency or devDependency in a package.json."""
|
|
155
|
+
try:
|
|
156
|
+
manifest = json.loads(path.read_text())
|
|
157
|
+
except (OSError, UnicodeDecodeError, ValueError):
|
|
158
|
+
return False
|
|
159
|
+
if not isinstance(manifest, dict):
|
|
160
|
+
return False
|
|
161
|
+
for key in ("dependencies", "devDependencies"):
|
|
162
|
+
dependencies = manifest.get(key)
|
|
163
|
+
if isinstance(dependencies, dict) and token in dependencies:
|
|
164
|
+
return True
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _text_declares(path: Path, token: str) -> bool:
|
|
169
|
+
"""True when `token` appears as a standalone name in the manifest's text.
|
|
170
|
+
|
|
171
|
+
Hyphen/underscore continuations do not count: `behave-django` or
|
|
172
|
+
`pytest-bdd-ng` must not signal `behave` / `pytest-bdd`.
|
|
173
|
+
"""
|
|
174
|
+
try:
|
|
175
|
+
text = path.read_text()
|
|
176
|
+
except (OSError, UnicodeDecodeError):
|
|
177
|
+
return False
|
|
178
|
+
pattern = r"(?<![\w-])" + re.escape(token) + r"(?![\w-])"
|
|
179
|
+
return re.search(pattern, text) is not None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def scan_manifests(root: Path) -> list[tuple[str, str]]:
|
|
183
|
+
"""(framework, destination) pairs for every manifest signal under `root`."""
|
|
184
|
+
hits: list[tuple[str, str]] = []
|
|
185
|
+
for path in _iter_files(root):
|
|
186
|
+
for rule in MANIFEST_RULES:
|
|
187
|
+
if not any(fnmatch.fnmatch(path.name, glob) for glob in rule.manifest_globs):
|
|
188
|
+
continue
|
|
189
|
+
declares = (
|
|
190
|
+
_package_json_declares if path.name == "package.json" else _text_declares
|
|
191
|
+
)
|
|
192
|
+
if not any(declares(path, token) for token in rule.tokens):
|
|
193
|
+
continue
|
|
194
|
+
if rule.destination is not None:
|
|
195
|
+
destination = rule.destination
|
|
196
|
+
else:
|
|
197
|
+
destination = (
|
|
198
|
+
path.parent.relative_to(root) / CSPROJ_FEATURES_SUBDIR
|
|
199
|
+
).as_posix()
|
|
200
|
+
hits.append((rule.framework, destination))
|
|
201
|
+
return hits
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def detect(root: Path) -> dict:
|
|
205
|
+
"""Detection result {signal, framework, dir} for the project at `root`.
|
|
206
|
+
|
|
207
|
+
Precedence: existing .feature files > manifest dependency > none. Manifest
|
|
208
|
+
hits pointing at more than one destination are a conflict and report
|
|
209
|
+
"none" (conservative); several hits sharing one destination are not.
|
|
210
|
+
"""
|
|
211
|
+
feature_dir = scan_feature_dir(root)
|
|
212
|
+
if feature_dir is not None:
|
|
213
|
+
return {"signal": "feature-files", "framework": None, "dir": feature_dir}
|
|
214
|
+
|
|
215
|
+
hits = scan_manifests(root)
|
|
216
|
+
destinations = {destination for _, destination in hits}
|
|
217
|
+
if len(destinations) == 1:
|
|
218
|
+
frameworks = sorted({framework for framework, _ in hits})
|
|
219
|
+
return {
|
|
220
|
+
"signal": "manifest",
|
|
221
|
+
"framework": frameworks[0],
|
|
222
|
+
"dir": destinations.pop(),
|
|
223
|
+
}
|
|
224
|
+
return {"signal": "none", "framework": None, "dir": None}
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def main(argv: list[str] | None = None) -> int:
|
|
228
|
+
parser = argparse.ArgumentParser(
|
|
229
|
+
prog="detect_bdd_convention.py",
|
|
230
|
+
description="Detect a target project's BDD convention and print it as JSON.",
|
|
231
|
+
)
|
|
232
|
+
parser.add_argument(
|
|
233
|
+
"target",
|
|
234
|
+
nargs="?",
|
|
235
|
+
default=".",
|
|
236
|
+
help="Project root to probe (default: current directory)",
|
|
237
|
+
)
|
|
238
|
+
args = parser.parse_args(argv)
|
|
239
|
+
|
|
240
|
+
target = Path(args.target)
|
|
241
|
+
if not target.is_dir():
|
|
242
|
+
sys.stderr.write(
|
|
243
|
+
f"detect-bdd-convention: target is not a directory: {args.target}\n"
|
|
244
|
+
)
|
|
245
|
+
return 2
|
|
246
|
+
|
|
247
|
+
print(json.dumps(detect(target), sort_keys=True))
|
|
248
|
+
return 0
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
if __name__ == "__main__": # pragma: no cover
|
|
252
|
+
sys.exit(main())
|
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Knowledge ablation analysis (#107) + agent-ablation analysis (#868).
|
|
3
|
+
|
|
4
|
+
Two related, deterministic (model-free) ablation reports live here:
|
|
5
|
+
|
|
6
|
+
1. **Knowledge ablation** (original, `--mode knowledge`, the default). Does a
|
|
7
|
+
knowledge file earn its tokens? Run the eval corpus twice — once with the
|
|
8
|
+
file AVAILABLE, once ABLATED (hidden) — and diff the grades. Pairs that
|
|
9
|
+
PASSED with the knowledge but FAIL without it are the file's measured
|
|
10
|
+
*retrieval value*.
|
|
11
|
+
|
|
12
|
+
2. **Agent ablation** (`--mode agent`, #868). Does removing a review agent
|
|
13
|
+
from the orchestrator's inline-review roster change *integration-tier*
|
|
14
|
+
pipeline outcomes? `/agent-eval --ablation <agent>` dispatches
|
|
15
|
+
`scripts/run_integration_eval.py` twice per fixture (baseline: full
|
|
16
|
+
roster; ablated: roster minus the named agent), each arm run K=3 trials
|
|
17
|
+
(reusing the `eval_variance.py` pass@k machinery). This module grades the
|
|
18
|
+
two arms' recorded actuals and computes the delta across three
|
|
19
|
+
dimensions: issues caught at review checkpoints, `testCommands`
|
|
20
|
+
pass/fail, and token cost. A failing baseline blocks any drop/retain
|
|
21
|
+
verdict (the underlying signal is uninterpretable) per the acceptance
|
|
22
|
+
criteria in issue #868.
|
|
23
|
+
|
|
24
|
+
Both modes are deterministic, model-free grading over already-recorded
|
|
25
|
+
actuals; live dispatch happens elsewhere (the `/agent-eval` skill /
|
|
26
|
+
`run_integration_eval.py`), never in this module.
|
|
27
|
+
|
|
28
|
+
Inputs (knowledge mode, `--mode knowledge`, default)
|
|
29
|
+
-----------------------------------------------------
|
|
30
|
+
--expected-dir DIR Expected specs (default: evals/expected).
|
|
31
|
+
--baseline FILE Actuals from the run WITH the knowledge available.
|
|
32
|
+
--ablated FILE Actuals from the run with the knowledge ablated.
|
|
33
|
+
--knowledge NAME Label for the ablated file (reporting only).
|
|
34
|
+
--only AGENT Restrict grading to one agent (optional).
|
|
35
|
+
-o FILE Write the report (default: stdout).
|
|
36
|
+
|
|
37
|
+
Inputs (agent mode, `--mode agent`, #868)
|
|
38
|
+
------------------------------------------
|
|
39
|
+
--expected-dir DIR Expected specs (default: evals/expected).
|
|
40
|
+
--baseline-trials-dir DIR Directory of the baseline arm's per-trial actuals
|
|
41
|
+
(`trial-*.json`, each the full actuals mapping
|
|
42
|
+
`run_integration_eval.py --out` writes).
|
|
43
|
+
--ablated-trials-dir DIR Same, for the ablated arm.
|
|
44
|
+
--ablated-agent NAME The review agent excluded in the ablated arm.
|
|
45
|
+
--model NAME Model version used for both arms (REQUIRED —
|
|
46
|
+
deltas are model-dependent, per #868).
|
|
47
|
+
-o FILE Write the report (default: stdout).
|
|
48
|
+
--append LOG Append one JSONL record (schema eval-ablation/v1)
|
|
49
|
+
to LOG, e.g. metrics/eval-ablation.jsonl.
|
|
50
|
+
--find-latest AGENT Instead of computing a new report, read the
|
|
51
|
+
JSONL at --append (or -o) / a positional
|
|
52
|
+
--jsonl path and print the most recent record
|
|
53
|
+
for AGENT (or "null" if none) — used by
|
|
54
|
+
/harness-audit to cite evidence.
|
|
55
|
+
--jsonl PATH JSONL file to search for --find-latest (default:
|
|
56
|
+
metrics/eval-ablation.jsonl).
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
from __future__ import annotations
|
|
60
|
+
|
|
61
|
+
import argparse
|
|
62
|
+
import json
|
|
63
|
+
import sys
|
|
64
|
+
from datetime import datetime, timezone
|
|
65
|
+
from pathlib import Path
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
#: `eval_grade.py`/`eval_variance.py` are monorepo-dev-only eval-corpus
|
|
69
|
+
#: tooling (they read `evals/` fixtures that are never shipped), unlike this
|
|
70
|
+
#: module's own `--find-latest` mode — a generic JSONL reader with no
|
|
71
|
+
#: repo-shape assumption. #1653 moved this file into the plugin so a shipped
|
|
72
|
+
#: skill (harness-audit) can invoke `--find-latest` via `${CLAUDE_PLUGIN_ROOT}`,
|
|
73
|
+
#: but `--mode knowledge`/`--mode agent` still only make sense run from this
|
|
74
|
+
#: repo's own dev checkout, where the sibling modules below actually resolve.
|
|
75
|
+
#: Deferred (not module-level) so importing this file — or invoking
|
|
76
|
+
#: `--find-latest` — never requires them to be importable at all; a plugin
|
|
77
|
+
#: user has no `evals/` corpus and no reason to hit this path.
|
|
78
|
+
def _grading_deps():
|
|
79
|
+
this_dir = Path(__file__).resolve().parent
|
|
80
|
+
for candidate in (this_dir, _repo_root_scripts_dir(this_dir)):
|
|
81
|
+
entry = str(candidate)
|
|
82
|
+
if candidate.is_dir() and entry not in sys.path:
|
|
83
|
+
sys.path.insert(0, entry)
|
|
84
|
+
from eval_grade import run_grading
|
|
85
|
+
from eval_variance import aggregate_trials
|
|
86
|
+
|
|
87
|
+
return run_grading, aggregate_trials
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _repo_root_scripts_dir(this_dir: Path) -> Path:
|
|
91
|
+
"""The monorepo's own root `scripts/` dir, reached from this file's
|
|
92
|
+
shipped location (`plugins/dev-team/scripts/`) by climbing three levels
|
|
93
|
+
(`scripts` -> `dev-team` -> `plugins` -> repo root) then descending into
|
|
94
|
+
`scripts/` again — where `eval_grade.py`/`eval_variance.py` still live,
|
|
95
|
+
unmoved. Resolves to a real path only inside this repo's own checkout;
|
|
96
|
+
harmless (simply not a directory) anywhere else."""
|
|
97
|
+
return (this_dir / ".." / ".." / ".." / "scripts").resolve()
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _passing(expected_dir: Path, actuals: dict, only: set | None) -> set:
|
|
101
|
+
run_grading, _ = _grading_deps()
|
|
102
|
+
results, _ = run_grading(expected_dir, actuals, None, only)
|
|
103
|
+
return {pair for pair, passed, _ in results if passed}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def ablation_impact(expected_dir: Path, baseline: dict, ablated: dict,
|
|
107
|
+
knowledge: str = "", only: set | None = None) -> dict:
|
|
108
|
+
"""Pairs that pass WITH the knowledge but fail WITHOUT it = retrieval value."""
|
|
109
|
+
base_pass = _passing(expected_dir, baseline, only)
|
|
110
|
+
abl_pass = _passing(expected_dir, ablated, only)
|
|
111
|
+
dropped = sorted(base_pass - abl_pass) # depended on the knowledge
|
|
112
|
+
gained = sorted(abl_pass - base_pass) # noise: passed only without it
|
|
113
|
+
return {
|
|
114
|
+
"schema": "knowledge-ablation/v1",
|
|
115
|
+
"knowledge": knowledge,
|
|
116
|
+
"retrieval_value": len(dropped), # how many pairs the file held up
|
|
117
|
+
"dropped_pairs": dropped, # the evidence
|
|
118
|
+
"spurious_gains": gained, # should be empty; flags noise
|
|
119
|
+
"verdict": ("earns its place" if dropped else
|
|
120
|
+
"no measured impact — removal/consolidation candidate"),
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _load_trial_actuals(trials_dir: Path) -> list[dict]:
|
|
125
|
+
"""Load every `*.json` in trials_dir as one arm's per-trial actuals."""
|
|
126
|
+
out = []
|
|
127
|
+
for f in sorted(trials_dir.glob("*.json")):
|
|
128
|
+
try:
|
|
129
|
+
out.append(json.loads(f.read_text()))
|
|
130
|
+
except (OSError, json.JSONDecodeError):
|
|
131
|
+
continue
|
|
132
|
+
return out
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _flatten_integration_targets(actual: dict) -> list[tuple[str, dict]]:
|
|
136
|
+
"""Yield (fixture_stem::target, target_actual) for every integration entry
|
|
137
|
+
in one trial's actuals mapping (the shape `run_integration_eval.py --out`
|
|
138
|
+
writes: ``{stem: {"integration": {target: {"results", "tokens",
|
|
139
|
+
"issues_caught"}}}}``)."""
|
|
140
|
+
out = []
|
|
141
|
+
for stem, block in actual.items():
|
|
142
|
+
if not isinstance(block, dict):
|
|
143
|
+
continue
|
|
144
|
+
for target, tdata in block.get("integration", {}).items():
|
|
145
|
+
if isinstance(tdata, dict):
|
|
146
|
+
out.append((f"{stem}::{target}", tdata))
|
|
147
|
+
return out
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def summarize_arm(expected_dir: Path, trial_actuals: list[dict]) -> dict:
|
|
151
|
+
"""Deterministically summarize one ablation arm's K trials.
|
|
152
|
+
|
|
153
|
+
`grade` is "pass" only if every fixture::target pair graded via the
|
|
154
|
+
registered `integration` grader passed on **every** trial (pass@k ==
|
|
155
|
+
1.0) — a conservative, reproducible bar; any flakiness or failure marks
|
|
156
|
+
the arm "fail" so a shaky arm can never anchor a drop/retain verdict.
|
|
157
|
+
`issues_caught` and `tokens` are the mean across trials (rounded);
|
|
158
|
+
`test_commands` is taken from the first trial that recorded any (all
|
|
159
|
+
trials target the same fixtures, so this is representative, not
|
|
160
|
+
authoritative — the trial-level `grade` is what actually gates the
|
|
161
|
+
verdict).
|
|
162
|
+
"""
|
|
163
|
+
if not trial_actuals:
|
|
164
|
+
return {"grade": "fail", "issues_caught": 0, "test_commands": [], "tokens": 0}
|
|
165
|
+
|
|
166
|
+
_, aggregate_trials = _grading_deps()
|
|
167
|
+
variance = aggregate_trials(expected_dir, trial_actuals)
|
|
168
|
+
by_pair = variance.get("by_pair", {})
|
|
169
|
+
# Scope the pass@k bar to the integration pairs this arm actually
|
|
170
|
+
# dispatched. aggregate_trials grades the WHOLE expected corpus, so the
|
|
171
|
+
# unit-tier fixtures that an integration arm never runs would otherwise
|
|
172
|
+
# report pass_at_k 0.0 and force every arm — baseline included — to "fail",
|
|
173
|
+
# making every ablation "baseline failed — inconclusive". Only the
|
|
174
|
+
# fixture(s) under ablation gate the arm verdict.
|
|
175
|
+
scoped_pairs: set[str] = set()
|
|
176
|
+
for actual in trial_actuals:
|
|
177
|
+
for pair, _tdata in _flatten_integration_targets(actual):
|
|
178
|
+
scoped_pairs.add(pair)
|
|
179
|
+
all_pass = bool(scoped_pairs) and all(
|
|
180
|
+
pair in by_pair and by_pair[pair]["pass_at_k"] >= 1.0
|
|
181
|
+
for pair in scoped_pairs
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
issues_totals: list[int] = []
|
|
185
|
+
token_totals: list[int] = []
|
|
186
|
+
test_commands: list[dict] = []
|
|
187
|
+
for actual in trial_actuals:
|
|
188
|
+
issues_sum = 0
|
|
189
|
+
tokens_sum = 0
|
|
190
|
+
for _pair, tdata in _flatten_integration_targets(actual):
|
|
191
|
+
issues_sum += int(tdata.get("issues_caught", 0) or 0)
|
|
192
|
+
tokens_sum += int(tdata.get("tokens", 0) or 0)
|
|
193
|
+
if not test_commands:
|
|
194
|
+
test_commands.extend(
|
|
195
|
+
{"command": r.get("command"), "exit_code": r.get("exit_code")}
|
|
196
|
+
for r in tdata.get("results", []) or []
|
|
197
|
+
if isinstance(r, dict)
|
|
198
|
+
)
|
|
199
|
+
issues_totals.append(issues_sum)
|
|
200
|
+
token_totals.append(tokens_sum)
|
|
201
|
+
|
|
202
|
+
mean_issues = round(sum(issues_totals) / len(issues_totals)) if issues_totals else 0
|
|
203
|
+
mean_tokens = round(sum(token_totals) / len(token_totals)) if token_totals else 0
|
|
204
|
+
|
|
205
|
+
return {
|
|
206
|
+
"grade": "pass" if all_pass else "fail",
|
|
207
|
+
"issues_caught": mean_issues,
|
|
208
|
+
"test_commands": test_commands,
|
|
209
|
+
"tokens": mean_tokens,
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _passed_command_count(arm: dict) -> int:
|
|
214
|
+
return sum(1 for c in arm.get("test_commands", []) if c.get("exit_code") == 0)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def agent_ablation_report(
|
|
218
|
+
expected_dir: Path,
|
|
219
|
+
baseline_trials: list[dict],
|
|
220
|
+
ablated_trials: list[dict],
|
|
221
|
+
ablated_agent: str,
|
|
222
|
+
model: str,
|
|
223
|
+
fixtures: list[str] | None = None,
|
|
224
|
+
) -> dict:
|
|
225
|
+
"""Grade both arms and compute the three-dimension delta (#868).
|
|
226
|
+
|
|
227
|
+
A failing baseline blocks any drop/retain claim (acceptance criterion
|
|
228
|
+
"failing-baseline-blocks-delta-claims"): the delta is still reported for
|
|
229
|
+
visibility, but `verdict` is forced to "baseline failed — inconclusive"
|
|
230
|
+
regardless of what the ablated arm shows.
|
|
231
|
+
"""
|
|
232
|
+
baseline = summarize_arm(expected_dir, baseline_trials)
|
|
233
|
+
ablated = summarize_arm(expected_dir, ablated_trials)
|
|
234
|
+
|
|
235
|
+
delta = {
|
|
236
|
+
"issues_caught": ablated["issues_caught"] - baseline["issues_caught"],
|
|
237
|
+
"test_commands_passed": _passed_command_count(ablated)
|
|
238
|
+
- _passed_command_count(baseline),
|
|
239
|
+
"tokens": ablated["tokens"] - baseline["tokens"],
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if baseline["grade"] != "pass":
|
|
243
|
+
verdict = "baseline failed — inconclusive"
|
|
244
|
+
elif delta["issues_caught"] == 0 and delta["test_commands_passed"] == 0:
|
|
245
|
+
verdict = "no measured impact — supports drop"
|
|
246
|
+
else:
|
|
247
|
+
verdict = "agent is load-bearing — retain"
|
|
248
|
+
|
|
249
|
+
if fixtures is None:
|
|
250
|
+
stems = set()
|
|
251
|
+
for actual in baseline_trials + ablated_trials:
|
|
252
|
+
stems.update(k for k in actual if isinstance(actual.get(k), dict))
|
|
253
|
+
fixtures = sorted(stems)
|
|
254
|
+
|
|
255
|
+
return {
|
|
256
|
+
"schema": "eval-ablation/v1",
|
|
257
|
+
"recorded_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
258
|
+
"ablated_agent": ablated_agent,
|
|
259
|
+
"fixtures": fixtures,
|
|
260
|
+
"model": model,
|
|
261
|
+
"baseline": baseline,
|
|
262
|
+
"ablated": ablated,
|
|
263
|
+
"delta": delta,
|
|
264
|
+
"verdict": verdict,
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def find_latest_ablation_record(jsonl_path: Path, agent: str) -> dict | None:
|
|
269
|
+
"""Return the most recent `eval-ablation/v1` record for `agent`, or None.
|
|
270
|
+
|
|
271
|
+
Used by `/harness-audit` (Step 3) to cite causal evidence for a
|
|
272
|
+
drop-candidate agent instead of correlational usage data alone.
|
|
273
|
+
"""
|
|
274
|
+
if not jsonl_path.exists():
|
|
275
|
+
return None
|
|
276
|
+
latest: dict | None = None
|
|
277
|
+
for line in jsonl_path.read_text().splitlines():
|
|
278
|
+
line = line.strip()
|
|
279
|
+
if not line:
|
|
280
|
+
continue
|
|
281
|
+
try:
|
|
282
|
+
record = json.loads(line)
|
|
283
|
+
except json.JSONDecodeError:
|
|
284
|
+
continue
|
|
285
|
+
if record.get("ablated_agent") != agent:
|
|
286
|
+
continue
|
|
287
|
+
if latest is None or record.get("recorded_at", "") >= latest.get(
|
|
288
|
+
"recorded_at", ""
|
|
289
|
+
):
|
|
290
|
+
latest = record
|
|
291
|
+
return latest
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def main(argv=None) -> int:
|
|
295
|
+
ap = argparse.ArgumentParser(description=__doc__,
|
|
296
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
297
|
+
ap.add_argument("--mode", choices=["knowledge", "agent"], default="knowledge")
|
|
298
|
+
ap.add_argument("--expected-dir", default="evals/expected")
|
|
299
|
+
# knowledge mode
|
|
300
|
+
ap.add_argument("--baseline")
|
|
301
|
+
ap.add_argument("--ablated")
|
|
302
|
+
ap.add_argument("--knowledge", default="")
|
|
303
|
+
ap.add_argument("--only")
|
|
304
|
+
# agent mode (#868)
|
|
305
|
+
ap.add_argument("--baseline-trials-dir")
|
|
306
|
+
ap.add_argument("--ablated-trials-dir")
|
|
307
|
+
ap.add_argument("--ablated-agent")
|
|
308
|
+
ap.add_argument("--model", help="model version used for both arms (required "
|
|
309
|
+
"for --mode agent — deltas are model-dependent)")
|
|
310
|
+
ap.add_argument("--fixtures", help="comma-separated fixture stems actually "
|
|
311
|
+
"exercised (reporting only; auto-detected if omitted)")
|
|
312
|
+
ap.add_argument("--append", metavar="LOG",
|
|
313
|
+
help="append the report as one JSONL line to LOG")
|
|
314
|
+
ap.add_argument("--find-latest", metavar="AGENT",
|
|
315
|
+
help="print the most recent eval-ablation/v1 record for "
|
|
316
|
+
"AGENT from --jsonl (or 'null'); no report is computed")
|
|
317
|
+
ap.add_argument("--jsonl", default="metrics/eval-ablation.jsonl",
|
|
318
|
+
help="JSONL file for --find-latest (default: "
|
|
319
|
+
"metrics/eval-ablation.jsonl)")
|
|
320
|
+
ap.add_argument("-o", "--out")
|
|
321
|
+
args = ap.parse_args(argv)
|
|
322
|
+
|
|
323
|
+
if args.find_latest:
|
|
324
|
+
record = find_latest_ablation_record(Path(args.jsonl), args.find_latest)
|
|
325
|
+
print(json.dumps(record) if record else "null")
|
|
326
|
+
return 0
|
|
327
|
+
|
|
328
|
+
if args.mode == "agent":
|
|
329
|
+
if not args.model:
|
|
330
|
+
print("error: --mode agent requires --model (deltas are "
|
|
331
|
+
"model-dependent, #868)", file=sys.stderr)
|
|
332
|
+
return 2
|
|
333
|
+
if not (args.baseline_trials_dir and args.ablated_trials_dir
|
|
334
|
+
and args.ablated_agent):
|
|
335
|
+
print("error: --mode agent requires --baseline-trials-dir, "
|
|
336
|
+
"--ablated-trials-dir, and --ablated-agent", file=sys.stderr)
|
|
337
|
+
return 2
|
|
338
|
+
baseline_trials = _load_trial_actuals(Path(args.baseline_trials_dir))
|
|
339
|
+
ablated_trials = _load_trial_actuals(Path(args.ablated_trials_dir))
|
|
340
|
+
fixtures = (
|
|
341
|
+
[s.strip() for s in args.fixtures.split(",") if s.strip()]
|
|
342
|
+
if args.fixtures else None
|
|
343
|
+
)
|
|
344
|
+
report = agent_ablation_report(
|
|
345
|
+
Path(args.expected_dir), baseline_trials, ablated_trials,
|
|
346
|
+
args.ablated_agent, args.model, fixtures,
|
|
347
|
+
)
|
|
348
|
+
out = json.dumps(report, indent=2, sort_keys=True)
|
|
349
|
+
if args.out:
|
|
350
|
+
Path(args.out).write_text(out + "\n")
|
|
351
|
+
else:
|
|
352
|
+
print(out)
|
|
353
|
+
if args.append:
|
|
354
|
+
with open(args.append, "a") as fh:
|
|
355
|
+
fh.write(json.dumps(report, sort_keys=True) + "\n")
|
|
356
|
+
return 0
|
|
357
|
+
|
|
358
|
+
if not (args.baseline and args.ablated):
|
|
359
|
+
print("error: --mode knowledge requires --baseline and --ablated",
|
|
360
|
+
file=sys.stderr)
|
|
361
|
+
return 2
|
|
362
|
+
baseline = json.loads(Path(args.baseline).read_text())
|
|
363
|
+
ablated = json.loads(Path(args.ablated).read_text())
|
|
364
|
+
only = {args.only} if args.only else None
|
|
365
|
+
report = ablation_impact(Path(args.expected_dir), baseline, ablated,
|
|
366
|
+
args.knowledge, only)
|
|
367
|
+
out = json.dumps(report, indent=2, sort_keys=True)
|
|
368
|
+
if args.out:
|
|
369
|
+
Path(args.out).write_text(out + "\n")
|
|
370
|
+
else:
|
|
371
|
+
print(out)
|
|
372
|
+
return 0
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
if __name__ == "__main__":
|
|
376
|
+
raise SystemExit(main())
|