pi-dev-team 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/PORTING.md +134 -0
- package/README.md +207 -0
- package/UPSTREAM.json +64 -0
- package/agents/Explore.md +15 -0
- package/agents/a11y-review.md +118 -0
- package/agents/adr-author.md +70 -0
- package/agents/ai-provenance-review.md +120 -0
- package/agents/angular-reactivity-review.md +95 -0
- package/agents/arch-review.md +135 -0
- package/agents/architect.md +78 -0
- package/agents/autoship-batch-proposer.md +69 -0
- package/agents/claude-setup-review.md +136 -0
- package/agents/codebase-recon.md +184 -0
- package/agents/component-architecture-review.md +119 -0
- package/agents/concurrency-review.md +109 -0
- package/agents/correctness-review.md +290 -0
- package/agents/data-flow-tracer.md +120 -0
- package/agents/doc-review.md +165 -0
- package/agents/domain-review.md +136 -0
- package/agents/general-purpose.md +10 -0
- package/agents/gherkin-quality-critic.md +113 -0
- package/agents/js-fp-review.md +114 -0
- package/agents/mutation-kill.md +684 -0
- package/agents/naming-review.md +142 -0
- package/agents/orchestrator.md +339 -0
- package/agents/performance-review.md +105 -0
- package/agents/plan-review-acceptance.md +115 -0
- package/agents/plan-review-design.md +90 -0
- package/agents/plan-review-parallelization.md +84 -0
- package/agents/plan-review-strategic.md +96 -0
- package/agents/plan-review-ux.md +110 -0
- package/agents/platform-engineer.md +64 -0
- package/agents/product-manager.md +68 -0
- package/agents/progress-guardian.md +79 -0
- package/agents/qa-engineer.md +289 -0
- package/agents/quality-reviewer.md +132 -0
- package/agents/react-reactivity-review.md +102 -0
- package/agents/refactor-opportunity-review.md +128 -0
- package/agents/security-engineer.md +60 -0
- package/agents/security-review.md +218 -0
- package/agents/session-analysis.md +95 -0
- package/agents/software-engineer.md +105 -0
- package/agents/spec-compliance-review.md +100 -0
- package/agents/spec-reviewer.md +114 -0
- package/agents/structure-review.md +146 -0
- package/agents/tech-writer.md +84 -0
- package/agents/test-review.md +246 -0
- package/agents/test-smell-review.md +188 -0
- package/agents/token-efficiency-review.md +139 -0
- package/agents/ui-ux-designer.md +54 -0
- package/agents/vue-reactivity-review.md +95 -0
- package/bin/__pycache__/claudecpython-314.pyc +0 -0
- package/bin/claude +258 -0
- package/docs/upstream/.pages +1 -0
- package/docs/upstream/CHANGELOG.md +2586 -0
- package/docs/upstream/README.md +155 -0
- package/docs/upstream/agent-architecture.md +214 -0
- package/docs/upstream/agent_info.md +187 -0
- package/docs/upstream/artifact-migration.md +124 -0
- package/docs/upstream/code-intelligence-nudge.md +149 -0
- package/docs/upstream/code-review-process.md +294 -0
- package/docs/upstream/concurrent-use.md +73 -0
- package/docs/upstream/context-management.md +111 -0
- package/docs/upstream/developer-notes.md +280 -0
- package/docs/upstream/diagrams/architecture-overview.svg +101 -0
- package/docs/upstream/diagrams/review-dispatch.svg +139 -0
- package/docs/upstream/diagrams/team-agents.svg +128 -0
- package/docs/upstream/diagrams/test-improve-flow.svg +166 -0
- package/docs/upstream/diagrams/workflow-linear.svg +66 -0
- package/docs/upstream/diagrams/workflow-three-phase.svg +200 -0
- package/docs/upstream/eval-maintenance.md +95 -0
- package/docs/upstream/eval-running-guide.md +147 -0
- package/docs/upstream/eval-system.md +291 -0
- package/docs/upstream/session-review-oss-complements.md +75 -0
- package/docs/upstream/session-review.md +212 -0
- package/docs/upstream/skills.md +188 -0
- package/docs/upstream/team-structure.md +21 -0
- package/docs/upstream/telemetry-ci-access.md +129 -0
- package/docs/upstream/telemetry-repo-security.md +120 -0
- package/docs/upstream/test-evaluation.md +277 -0
- package/docs/upstream/test-improve.md +154 -0
- package/docs/upstream/triage-workflow.md +282 -0
- package/docs/upstream/workflows.md +289 -0
- package/extensions/dev-team/index.ts +539 -0
- package/extensions/dev-team/lib/agents.ts +272 -0
- package/extensions/dev-team/lib/ai-credits.ts +92 -0
- package/extensions/dev-team/lib/autocompact.ts +81 -0
- package/extensions/dev-team/lib/child-run.ts +102 -0
- package/extensions/dev-team/lib/config.ts +236 -0
- package/extensions/dev-team/lib/gh-command.ts +103 -0
- package/extensions/dev-team/lib/github-style.ts +307 -0
- package/extensions/dev-team/lib/hooks.ts +350 -0
- package/extensions/dev-team/lib/metrics.ts +115 -0
- package/extensions/dev-team/lib/safe-read.ts +49 -0
- package/extensions/dev-team/lib/session-files.ts +57 -0
- package/extensions/dev-team/lib/session-spend.ts +123 -0
- package/extensions/dev-team/lib/shell-scan.ts +205 -0
- package/extensions/dev-team/lib/skills.ts +213 -0
- package/extensions/dev-team/lib/subagent-render.ts +245 -0
- package/extensions/dev-team/lib/subagent-types.ts +164 -0
- package/extensions/dev-team/lib/subagent.ts +596 -0
- package/extensions/dev-team/lib/terminal-text.ts +54 -0
- package/extensions/dev-team/lib/tools-misc.ts +152 -0
- package/extensions/dev-team/lib/transcript.ts +110 -0
- package/extensions/dev-team/lib/trust.ts +52 -0
- package/extensions/dev-team/lib/usage-breakdown.ts +176 -0
- package/extensions/dev-team/lib/usage-chart.ts +153 -0
- package/extensions/dev-team/lib/usage-command.ts +107 -0
- package/extensions/dev-team/lib/usage-history.ts +203 -0
- package/extensions/dev-team/lib/usage-render.ts +225 -0
- package/extensions/dev-team/lib/usage-split-bar.ts +127 -0
- package/extensions/dev-team/lib/usage-state.ts +116 -0
- package/extensions/dev-team/lib/usage-text.ts +159 -0
- package/extensions/dev-team/lib/usage-view.ts +109 -0
- package/hooks/__pycache__/refactor_test_freeze_guard.cpython-314.pyc +0 -0
- package/hooks/agent_dispatch_ledger.py +190 -0
- package/hooks/autocompact_setup_nudge.py +99 -0
- package/hooks/bash_retry_guard.py +228 -0
- package/hooks/boundary_events_write_guard.py +352 -0
- package/hooks/code_intelligence_nudge.py +293 -0
- package/hooks/code_intelligence_turn_mark.py +317 -0
- package/hooks/codegraph_bootstrap.py +139 -0
- package/hooks/contract_version_guard.py +362 -0
- package/hooks/cost_meter.py +106 -0
- package/hooks/destructive-commands.json +62 -0
- package/hooks/destructive_guard.py +477 -0
- package/hooks/eval_compliance_check.py +440 -0
- package/hooks/guards.json +17 -0
- package/hooks/hooks.json +323 -0
- package/hooks/internal_double_gate.py +296 -0
- package/hooks/js_fp_review.py +212 -0
- package/hooks/knowledge_index.py +119 -0
- package/hooks/lib/__pycache__/artifact_paths.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/atomic_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/autocompact_config.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/boundary_events.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/doc_classification.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/gh_pr_create_detect.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/git_safe_diff.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/instrument_log.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/metrics_query.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/plugin_version.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/pre_commit_doc_classifier.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_agent_registry.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_corroboration.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_hash.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_verdicts.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stdin_json.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stryker_invocation.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/telemetry_consent.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/test_file_classify.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/token_efficiency_limits.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/verify_guard_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/xunit_v3_operator_gate.cpython-314.pyc +0 -0
- package/hooks/lib/agent_skill_hints.py +74 -0
- package/hooks/lib/artifact_paths.py +263 -0
- package/hooks/lib/atomic_state.py +557 -0
- package/hooks/lib/autocompact_config.py +103 -0
- package/hooks/lib/autoship_log.py +106 -0
- package/hooks/lib/banned_scripts_policy.py +51 -0
- package/hooks/lib/boundary_events.py +436 -0
- package/hooks/lib/build_knowledge_index.py +504 -0
- package/hooks/lib/build_skills_index.py +361 -0
- package/hooks/lib/build_state.py +116 -0
- package/hooks/lib/classify_ship_outcome.py +126 -0
- package/hooks/lib/config_changelog_schema.py +115 -0
- package/hooks/lib/cost_meter.py +955 -0
- package/hooks/lib/doc_classification.py +116 -0
- package/hooks/lib/gh_pr_create_detect.py +136 -0
- package/hooks/lib/git_safe_diff.py +123 -0
- package/hooks/lib/instrument_log.py +66 -0
- package/hooks/lib/iteration_journal_gate.py +197 -0
- package/hooks/lib/knowledge_index_paths.py +88 -0
- package/hooks/lib/mcp_json_repowise.py +177 -0
- package/hooks/lib/metrics_query.py +202 -0
- package/hooks/lib/minimal_yaml.py +434 -0
- package/hooks/lib/plugin_version.py +142 -0
- package/hooks/lib/pre_commit_detect.py +537 -0
- package/hooks/lib/pre_commit_doc_classifier.py +126 -0
- package/hooks/lib/pricing.py +118 -0
- package/hooks/lib/report_pdf.py +371 -0
- package/hooks/lib/review_agent_registry.py +142 -0
- package/hooks/lib/review_dispatch_ledger.py +101 -0
- package/hooks/lib/review_gate_corroboration.py +521 -0
- package/hooks/lib/review_gate_hash.py +252 -0
- package/hooks/lib/review_gate_normalized_hash.py +1115 -0
- package/hooks/lib/review_verdicts.py +301 -0
- package/hooks/lib/run_report.py +160 -0
- package/hooks/lib/skill_categories.yaml +125 -0
- package/hooks/lib/stdin_json.py +57 -0
- package/hooks/lib/stryker_invocation.py +102 -0
- package/hooks/lib/telemetry_consent.py +41 -0
- package/hooks/lib/telemetry_report.py +108 -0
- package/hooks/lib/test_file_classify.py +160 -0
- package/hooks/lib/token_efficiency_limits.py +51 -0
- package/hooks/lib/turn_identity.py +77 -0
- package/hooks/lib/verify_guard_state.py +110 -0
- package/hooks/lib/workflow_state.py +206 -0
- package/hooks/lib/xunit_v3_operator_gate.py +596 -0
- package/hooks/mcp_json_repowise_nudge.py +74 -0
- package/hooks/mutation_adapters/__init__.py +7 -0
- package/hooks/mutation_adapters/__pycache__/__init__.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/lib.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/mutmut.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/pitest.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker_net.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/lib.py +478 -0
- package/hooks/mutation_adapters/mutmut.py +188 -0
- package/hooks/mutation_adapters/pitest.py +266 -0
- package/hooks/mutation_adapters/stryker.py +157 -0
- package/hooks/mutation_adapters/stryker_net.py +264 -0
- package/hooks/mutation_gate.py +193 -0
- package/hooks/mutation_testing_smoke_gate.py +371 -0
- package/hooks/pending_review_notify.py +121 -0
- package/hooks/phase_marker.py +138 -0
- package/hooks/post_compact_state_reinject.py +180 -0
- package/hooks/post_format.py +115 -0
- package/hooks/pre_commit_knowledge_index.py +128 -0
- package/hooks/pre_commit_review.py +66 -0
- package/hooks/pre_pr_review.py +694 -0
- package/hooks/pre_tool_guard.py +405 -0
- package/hooks/py.sh +73 -0
- package/hooks/refactor-bash-write-patterns.json +29 -0
- package/hooks/refactor_test_bash_guard.py +253 -0
- package/hooks/refactor_test_freeze_guard.py +139 -0
- package/hooks/refactor_test_revert_guard.py +186 -0
- package/hooks/repo_review_nudge.py +287 -0
- package/hooks/review_verdict_recorder.py +464 -0
- package/hooks/scan_bash_command_for_banned_scripts.py +428 -0
- package/hooks/scan_worktree_for_banned_scripts.py +238 -0
- package/hooks/session_learning_trigger.py +248 -0
- package/hooks/skills_index.py +126 -0
- package/hooks/stryker_xunit_shim_guard.py +571 -0
- package/hooks/subagent_completion_guard.py +309 -0
- package/hooks/subagent_skill_context.py +139 -0
- package/hooks/task_completion_metrics.py +216 -0
- package/hooks/tdd_guard.py +229 -0
- package/hooks/telemetry.py +341 -0
- package/hooks/token_efficiency_review.py +194 -0
- package/hooks/verify_guard.py +183 -0
- package/hooks/verify_guard_edit_marker.py +73 -0
- package/hooks/version_check.py +173 -0
- package/knowledge/accepted-risks-schema.md +98 -0
- package/knowledge/adr-decision-criteria.md +64 -0
- package/knowledge/adversarial-review-protocol.md +139 -0
- package/knowledge/agent-registry.md +228 -0
- package/knowledge/agent-review-methodology.md +80 -0
- package/knowledge/ai-friendly-repo-guidelines.md +67 -0
- package/knowledge/architecture-assessment.md +96 -0
- package/knowledge/artifact-lifecycle.md +57 -0
- package/knowledge/cd-maturity-model.md +82 -0
- package/knowledge/cd-test-architecture.md +190 -0
- package/knowledge/ci-cd-file-scope.md +24 -0
- package/knowledge/codegraph-vs-graphify.md +192 -0
- package/knowledge/component-test-patterns.md +139 -0
- package/knowledge/database-change-management.md +80 -0
- package/knowledge/database-test-patterns.md +79 -0
- package/knowledge/decision-defaults.md +88 -0
- package/knowledge/dependency-breaking-techniques.md +116 -0
- package/knowledge/deployment-pipeline.md +86 -0
- package/knowledge/design-smells.md +122 -0
- package/knowledge/directory-enumeration.md +38 -0
- package/knowledge/domain-modeling.md +123 -0
- package/knowledge/evidence-bundle.md +90 -0
- package/knowledge/exploratory-testing-field-guide.md +122 -0
- package/knowledge/failure-routing.md +28 -0
- package/knowledge/fixture-construction.md +56 -0
- package/knowledge/frontend-component-architecture.md +139 -0
- package/knowledge/gherkin-quality-review-dispatch.md +135 -0
- package/knowledge/index.json +6766 -0
- package/knowledge/internal-collaborator-doubling.md +101 -0
- package/knowledge/legacy-test-strategy.md +71 -0
- package/knowledge/long-run-waiting.md +66 -0
- package/knowledge/microservice-testing.md +71 -0
- package/knowledge/model-pricing.json +23 -0
- package/knowledge/mutation-score-formulas.md +60 -0
- package/knowledge/object-calisthenics.md +147 -0
- package/knowledge/oracle-provenance.md +94 -0
- package/knowledge/orchestrator-script-implementation.md +185 -0
- package/knowledge/owasp-detection.md +148 -0
- package/knowledge/plan-review-rubric.md +56 -0
- package/knowledge/proxy-connectivity.md +62 -0
- package/knowledge/reactive-effect-patterns.md +73 -0
- package/knowledge/recon-inventory-excludes.txt +32 -0
- package/knowledge/references/bdd-value-guide.md +61 -0
- package/knowledge/references/csharp-http-client-testing.md +264 -0
- package/knowledge/release-strategies.md +74 -0
- package/knowledge/report-output-location.md +117 -0
- package/knowledge/report-pdf-integration.md +63 -0
- package/knowledge/report-print.css +129 -0
- package/knowledge/report-template.md +114 -0
- package/knowledge/report-to-pdf.md +69 -0
- package/knowledge/request-processing-flow.md +63 -0
- package/knowledge/result-verification.md +52 -0
- package/knowledge/review-agent-output-contract.md +121 -0
- package/knowledge/review-lens-classification.md +113 -0
- package/knowledge/review-rubric.md +62 -0
- package/knowledge/review-template.md +104 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/negative.js +1 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/positive.js +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/negative.py +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/positive.py +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/positive.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/negative.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/negative.cs +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/positive.cs +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/negative.js +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/negative.java +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/positive.java +1 -0
- package/knowledge/schemas/disposition-register-v1.json +65 -0
- package/knowledge/schemas/recon-envelope-v1.json +198 -0
- package/knowledge/schemas/unified-finding-v1.json +72 -0
- package/knowledge/security-primitives-contract.md +301 -0
- package/knowledge/security-review-rule-map.yaml +107 -0
- package/knowledge/skills-registry.md +72 -0
- package/knowledge/task-size-classifier.md +103 -0
- package/knowledge/telemetry-schema.md +881 -0
- package/knowledge/test-automation-maturity.md +56 -0
- package/knowledge/test-automation-principles.md +71 -0
- package/knowledge/test-cadence-tradeoffs.md +68 -0
- package/knowledge/test-doubles.md +105 -0
- package/knowledge/test-file-indicators.md +22 -0
- package/knowledge/test-layer-gates.md +35 -0
- package/knowledge/test-matrix-examples/django-batch.md +24 -0
- package/knowledge/test-matrix-examples/dotnet-grpc-fronting-api.md +90 -0
- package/knowledge/test-matrix-examples/dotnet-http-consumer.md +131 -0
- package/knowledge/test-matrix-examples/react-node-spa.md +24 -0
- package/knowledge/test-matrix-examples/spring-boot-service.md +25 -0
- package/knowledge/test-matrix-examples/ssr-htmx.md +24 -0
- package/knowledge/test-organization.md +70 -0
- package/knowledge/test-pyramid.md +84 -0
- package/knowledge/test-refactoring.md +67 -0
- package/knowledge/test-review-division-of-labor.md +85 -0
- package/knowledge/test-smells.md +80 -0
- package/knowledge/test-stack-profiles/bdd-frameworks.md +235 -0
- package/knowledge/test-stack-profiles/django.md +13 -0
- package/knowledge/test-stack-profiles/dotnet.md +18 -0
- package/knowledge/test-stack-profiles/go.md +16 -0
- package/knowledge/test-stack-profiles/node.md +16 -0
- package/knowledge/test-stack-profiles/react.md +12 -0
- package/knowledge/test-stack-profiles/spring-boot.md +16 -0
- package/knowledge/test-stack-profiles/ssr-htmx.md +14 -0
- package/knowledge/test-stack-profiles/vue.md +12 -0
- package/knowledge/test-strategy.md +70 -0
- package/knowledge/testability-patterns.md +240 -0
- package/knowledge/testing-quadrants.md +44 -0
- package/knowledge/testing-techniques/approval.md +15 -0
- package/knowledge/testing-techniques/chaos.md +17 -0
- package/knowledge/testing-techniques/fuzz.md +15 -0
- package/knowledge/testing-techniques/property-based.md +15 -0
- package/knowledge/testing-techniques/schema-validation.md +15 -0
- package/knowledge/testing-techniques/screenshot.md +15 -0
- package/knowledge/three-phase-workflow.md +198 -0
- package/knowledge/value-patterns.md +55 -0
- package/knowledge/verification-mode.md +116 -0
- package/knowledge/virtual-service-libraries.md +75 -0
- package/knowledge/wave-consolidation-guidance.md +21 -0
- package/overrides/agents/Explore.md +15 -0
- package/overrides/agents/general-purpose.md +10 -0
- package/overrides/notes/autoship.md +6 -0
- package/overrides/notes/issues-from-assessment.md +3 -0
- package/overrides/notes/issues-from-plan.md +3 -0
- package/overrides/notes/mutation-night-watch.md +3 -0
- package/overrides/notes/mutation-testing.md +3 -0
- package/overrides/notes/pr.md +7 -0
- package/overrides/notes/project-init.md +6 -0
- package/overrides/notes/setup.md +13 -0
- package/overrides/notes/specs.md +3 -0
- package/overrides/skills/headless-run/SKILL.md +45 -0
- package/overrides/skills/upgrade/SKILL.md +30 -0
- package/overrides/skills/version/SKILL.md +25 -0
- package/package.json +36 -0
- package/scripts/authoring_digest.py +93 -0
- package/scripts/autoship_discover.py +121 -0
- package/scripts/autoship_group.py +409 -0
- package/scripts/autoship_proposals.py +494 -0
- package/scripts/autoship_queue.py +291 -0
- package/scripts/autoship_reclaim.py +495 -0
- package/scripts/build_jobs.py +108 -0
- package/scripts/build_rollback_point.py +240 -0
- package/scripts/build_slice_scope.py +157 -0
- package/scripts/build_wave.py +109 -0
- package/scripts/build_wave_reconcile.py +252 -0
- package/scripts/build_worktree_baseref.py +113 -0
- package/scripts/check_agent_scope.py +117 -0
- package/scripts/check_agent_tool_mapping.py +213 -0
- package/scripts/check_review_agent_mcp_tools.py +317 -0
- package/scripts/check_security_assessment_mcp_tools.py +165 -0
- package/scripts/checkpoint_abort.py +502 -0
- package/scripts/claude_setup_review.py +438 -0
- package/scripts/codebase_recon.py +556 -0
- package/scripts/coverage_config.py +623 -0
- package/scripts/coverage_delta_steering.py +330 -0
- package/scripts/coverage_discovery_dotnet.py +315 -0
- package/scripts/coverage_discovery_java.py +742 -0
- package/scripts/coverage_discovery_js.py +546 -0
- package/scripts/coverage_gap_ranking.py +556 -0
- package/scripts/coverage_readiness.py +455 -0
- package/scripts/coverage_report_parse.py +521 -0
- package/scripts/detect_bdd_convention.py +252 -0
- package/scripts/eval_ablation.py +376 -0
- package/scripts/gherkin_analysis_coverage_gate.py +306 -0
- package/scripts/gherkin_cross_feature_duplicate_titles_gate.py +173 -0
- package/scripts/gherkin_effectiveness_rollup.py +238 -0
- package/scripts/gherkin_failure_path_gate.py +206 -0
- package/scripts/gherkin_feature_merge.py +720 -0
- package/scripts/gherkin_stub_gate.py +163 -0
- package/scripts/gherkin_stub_merge.py +479 -0
- package/scripts/git_origin_host.py +88 -0
- package/scripts/install-java-static-analysis.py +110 -0
- package/scripts/issue_deps.py +74 -0
- package/scripts/lib/_bdd_markers.py +28 -0
- package/scripts/lib/_gherkin_text.py +93 -0
- package/scripts/lib/_vendored_tree.py +70 -0
- package/scripts/lib/autoship_state.py +397 -0
- package/scripts/lib/claude_md_guard.py +226 -0
- package/scripts/lib/deterministic_recon.py +446 -0
- package/scripts/lib/mcp_tool_grants.py +211 -0
- package/scripts/lib/plan_parse.py +386 -0
- package/scripts/lib/review_result.py +84 -0
- package/scripts/lib/review_roster.py +86 -0
- package/scripts/lib/session_log/__init__.py +34 -0
- package/scripts/lib/session_log/__pycache__/__init__.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/__pycache__/records.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/classify.py +231 -0
- package/scripts/lib/session_log/corrections.py +194 -0
- package/scripts/lib/session_log/discovery.py +108 -0
- package/scripts/lib/session_log/records.py +218 -0
- package/scripts/lib/session_log/redact.py +76 -0
- package/scripts/lib/session_log/signals.py +373 -0
- package/scripts/lib/session_report_downstream.py +614 -0
- package/scripts/lib/session_report_maintainer.py +1273 -0
- package/scripts/lib/session_report_shared.py +262 -0
- package/scripts/lib/settings_hook_guard.py +157 -0
- package/scripts/lib/slug.py +33 -0
- package/scripts/lib/stub_extractors/__init__.py +82 -0
- package/scripts/lib/stub_extractors/_common.py +328 -0
- package/scripts/lib/stub_extractors/csharp.py +19 -0
- package/scripts/lib/stub_extractors/go.py +173 -0
- package/scripts/lib/stub_extractors/java.py +18 -0
- package/scripts/lib/stub_extractors/jsts.py +126 -0
- package/scripts/mutation_stack_sections.py +149 -0
- package/scripts/mutation_yield_steering.py +345 -0
- package/scripts/orchestrator.py +895 -0
- package/scripts/plan_gherkin_export.py +227 -0
- package/scripts/plan_waves.py +208 -0
- package/scripts/pr_close_keyword_lint.py +108 -0
- package/scripts/progress_guardian.py +888 -0
- package/scripts/recon_inventory.py +273 -0
- package/scripts/review_findings_log.py +93 -0
- package/scripts/run_invariants.py +124 -0
- package/scripts/select_lenses.py +640 -0
- package/scripts/session_report.py +486 -0
- package/scripts/set_autocompact_env.py +221 -0
- package/scripts/ship_resume_guard.py +135 -0
- package/scripts/ship_review_gate.py +63 -0
- package/scripts/specs_convention_marker.py +103 -0
- package/scripts/test_improve_resume.py +277 -0
- package/scripts/test_review_mechanics.py +958 -0
- package/scripts/token_efficiency_review.py +322 -0
- package/scripts/verdict_scope.py +285 -0
- package/scripts/verify_gherkin_quality_critic_isolation.py +296 -0
- package/scripts/verify_tier.py +157 -0
- package/skills/adr-tools/SKILL.md +118 -0
- package/skills/agent-readiness/SKILL.md +105 -0
- package/skills/agent-readiness/ai_friendly_analyzers.py +326 -0
- package/skills/agent-readiness/scanner.py +441 -0
- package/skills/agent-readiness/scorecard.yaml +88 -0
- package/skills/api-design/SKILL.md +115 -0
- package/skills/apply-fixes/SKILL.md +171 -0
- package/skills/apply-test-doubles/SKILL.md +321 -0
- package/skills/artifact-lifecycle/SKILL.md +127 -0
- package/skills/autoship/SKILL.md +1124 -0
- package/skills/benchmark/SKILL.md +105 -0
- package/skills/branch-workflow/SKILL.md +89 -0
- package/skills/browse/SKILL.md +184 -0
- package/skills/browser-testing/SKILL.md +62 -0
- package/skills/browser-testing/references/playwright-patterns.md +216 -0
- package/skills/build/SKILL.md +422 -0
- package/skills/build/references/static-self-heal.md +245 -0
- package/skills/careful/SKILL.md +72 -0
- package/skills/cd-test-architecture/SKILL.md +371 -0
- package/skills/ci-debugging/SKILL.md +105 -0
- package/skills/co-evolution-audit/SKILL.md +269 -0
- package/skills/code-review/SKILL.md +1015 -0
- package/skills/code-review/examples/aggregated-sample.json +56 -0
- package/skills/code-review/examples/sample-report.md +41 -0
- package/skills/code-review/output-format.md +478 -0
- package/skills/code-review/scripts/activation.py +86 -0
- package/skills/code-review/scripts/change_impact.py +357 -0
- package/skills/code-review/scripts/change_shape.py +372 -0
- package/skills/code-review/scripts/change_size.py +212 -0
- package/skills/code-review/scripts/changed_file_list.py +141 -0
- package/skills/code-review/scripts/closing_pass.py +187 -0
- package/skills/code-review/scripts/consolidate.py +277 -0
- package/skills/code-review/scripts/contract_failure_report.py +185 -0
- package/skills/code-review/scripts/dispatch_reconcile.py +66 -0
- package/skills/code-review/scripts/dispatch_waves.py +164 -0
- package/skills/code-review/scripts/finding_signature.py +446 -0
- package/skills/code-review/scripts/ledger.py +283 -0
- package/skills/code-review/scripts/partition.py +169 -0
- package/skills/code-review/scripts/render_tiered_findings.py +274 -0
- package/skills/code-review/scripts/repo_invariants.py +1066 -0
- package/skills/code-review/scripts/review_context_pack.py +306 -0
- package/skills/code-review/scripts/review_round_log.py +345 -0
- package/skills/code-review/scripts/review_value_coverage.py +297 -0
- package/skills/code-review/scripts/validate_review_output.py +467 -0
- package/skills/code-review/sliced-mode.md +205 -0
- package/skills/competitive-analysis/SKILL.md +191 -0
- package/skills/context-loading-protocol/SKILL.md +157 -0
- package/skills/continue/SKILL.md +90 -0
- package/skills/cost-report/SKILL.md +178 -0
- package/skills/coverage-baseline/SKILL.md +335 -0
- package/skills/coverage-baseline/references/multi-project-discovery.md +202 -0
- package/skills/coverage-delta/SKILL.md +181 -0
- package/skills/coverage-delta/references/mutation-gate.md +70 -0
- package/skills/design-doc/SKILL.md +95 -0
- package/skills/design-interrogation/SKILL.md +89 -0
- package/skills/design-it-twice/SKILL.md +91 -0
- package/skills/docker-image-audit/SKILL.md +108 -0
- package/skills/docker-image-audit/references/install-guide.md +64 -0
- package/skills/docker-image-audit/references/report-template.md +73 -0
- package/skills/docker-image-create/SKILL.md +185 -0
- package/skills/domain-analysis/SKILL.md +183 -0
- package/skills/domain-driven-design/SKILL.md +194 -0
- package/skills/exploratory-testing/SKILL.md +108 -0
- package/skills/explore/SKILL.md +51 -0
- package/skills/farley-score/SKILL.md +165 -0
- package/skills/feature-file-validation/SKILL.md +78 -0
- package/skills/feature-file-validation/references/validation-rules.md +115 -0
- package/skills/feedback-learning/SKILL.md +414 -0
- package/skills/fix/SKILL.md +450 -0
- package/skills/freeze/SKILL.md +68 -0
- package/skills/frontend-architecture/SKILL.md +113 -0
- package/skills/gherkin-derive/SKILL.md +630 -0
- package/skills/gherkin-public/SKILL.md +266 -0
- package/skills/governance-compliance/SKILL.md +150 -0
- package/skills/guard/SKILL.md +75 -0
- package/skills/handoff/SKILL.md +139 -0
- package/skills/handoff/references/summary-templates.md +242 -0
- package/skills/harness-audit/SKILL.md +751 -0
- package/skills/harness-audit/scripts/lesson_validate.py +386 -0
- package/skills/harness-audit/scripts/redundancy_criterion.py +188 -0
- package/skills/headless-run/SKILL.md +45 -0
- package/skills/headless-run/scripts/isolated_dispatch.py +381 -0
- package/skills/help/SKILL.md +72 -0
- package/skills/hexagonal-architecture/SKILL.md +85 -0
- package/skills/human-oversight-protocol/SKILL.md +224 -0
- package/skills/issues-from-assessment/SKILL.md +223 -0
- package/skills/issues-from-plan/SKILL.md +133 -0
- package/skills/legacy-code/SKILL.md +132 -0
- package/skills/mermaid-diagramming/SKILL.md +120 -0
- package/skills/mutation-night-watch/SKILL.md +154 -0
- package/skills/mutation-night-watch/references/scheduling.md +135 -0
- package/skills/mutation-testing/SKILL.md +396 -0
- package/skills/mutation-testing/references/languages/csharp-stryker-net.md +676 -0
- package/skills/mutation-testing/references/languages/go-go-mutesting.md +95 -0
- package/skills/mutation-testing/references/languages/java-pitest.md +77 -0
- package/skills/mutation-testing/references/languages/javascript-stryker.md +188 -0
- package/skills/mutation-testing/references/languages/python-mutmut.md +97 -0
- package/skills/mutation-testing/references/time-estimation.md +34 -0
- package/skills/mutation-testing/references/tool-detection.md +15 -0
- package/skills/mutation-testing/references/workflow-callers.md +23 -0
- package/skills/mutation-testing/scripts/__pycache__/xunit_v3_feature_detector.cpython-314.pyc +0 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_slice_runner.py +635 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_status_loop.py +525 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_wrapper.py +681 -0
- package/skills/mutation-testing/scripts/mutation_baseline_reuse.py +292 -0
- package/skills/mutation-testing/scripts/mutation_exclude_policy.py +268 -0
- package/skills/mutation-testing/scripts/mutation_feasibility_gate.py +463 -0
- package/skills/mutation-testing/scripts/mutation_kill_headless.py +331 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert.py +199 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert_python.py +150 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop.py +869 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop_python.py +949 -0
- package/skills/mutation-testing/scripts/mutation_kill_retry.py +592 -0
- package/skills/mutation-testing/scripts/mutation_kill_shared.py +620 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch.py +462 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch_stacks.py +425 -0
- package/skills/mutation-testing/scripts/mutation_report.py +743 -0
- package/skills/mutation-testing/scripts/mutation_report_cli.py +175 -0
- package/skills/mutation-testing/scripts/mutation_safety_gate.py +69 -0
- package/skills/mutation-testing/scripts/stryker_shard_pipeline.py +847 -0
- package/skills/mutation-testing/scripts/stryker_shard_setup.py +440 -0
- package/skills/mutation-testing/scripts/stryker_timeout_retry.py +142 -0
- package/skills/mutation-testing/scripts/xunit_v3_feature_detector.py +341 -0
- package/skills/performance-benchmark/SKILL.md +174 -0
- package/skills/performance-benchmark/examples/report-format.md +43 -0
- package/skills/performance-benchmark/references/benchmark-script.md +169 -0
- package/skills/performance-metrics/SKILL.md +265 -0
- package/skills/plan/SKILL.md +199 -0
- package/skills/plan/references/gherkin-persistence.md +43 -0
- package/skills/plan/references/plan-template.md +182 -0
- package/skills/pr/SKILL.md +289 -0
- package/skills/pr/scripts/gate_retry_state.py +368 -0
- package/skills/project-init/README.md +141 -0
- package/skills/project-init/SKILL.md +1197 -0
- package/skills/project-init/evals/evals.json +200 -0
- package/skills/project-init/references/capability-tools.md +55 -0
- package/skills/project-init/references/configs.md +221 -0
- package/skills/property-based-testing/SKILL.md +121 -0
- package/skills/property-based-testing/fixtures/invariant_fixture.py +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/README.md +42 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/README.md +263 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.js +12147 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.js +12011 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/rolldown-runtime-D7D4PA-g.js +13 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/package.json +94 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/README.md +168 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.js +38 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.js +18 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.js +22 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.js +134 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.js +37 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.js +17 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.js +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.js +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.js +44 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.js +90 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.js +78 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.js +8 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.js +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.js +46 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.js +92 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.js +82 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.js +10 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.js +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/package.json +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package-lock.json +1179 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package.json +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.js +29 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.properties.test.js +16 -0
- package/skills/property-based-testing/fixtures/no_property_fixture.py +10 -0
- package/skills/property-based-testing/fixtures/roundtrip_fixture.py +16 -0
- package/skills/property-based-testing/references/languages/javascript.md +54 -0
- package/skills/property-based-testing/scripts/detect_and_dispatch.py +80 -0
- package/skills/property-based-testing/scripts/hypothesis_scaffold.py +276 -0
- package/skills/proxy-resilience/SKILL.md +84 -0
- package/skills/quality-gate-pipeline/SKILL.md +184 -0
- package/skills/quality-targets-converge/SKILL.md +254 -0
- package/skills/repo-review/SKILL.md +159 -0
- package/skills/report-pdf/SKILL.md +66 -0
- package/skills/review/SKILL.md +47 -0
- package/skills/review-agent/SKILL.md +152 -0
- package/skills/review-summary/SKILL.md +73 -0
- package/skills/run-report/SKILL.md +70 -0
- package/skills/semantic-duplication-scan/SKILL.md +337 -0
- package/skills/semantic-scan/SKILL.md +53 -0
- package/skills/semgrep-analyze/SKILL.md +139 -0
- package/skills/setup/SKILL.md +1122 -0
- package/skills/ship/SKILL.md +240 -0
- package/skills/source-verification/SKILL.md +210 -0
- package/skills/source-verification/scripts/claim_extractor.py +155 -0
- package/skills/specs/.size-baseline.json +4 -0
- package/skills/specs/SKILL.md +243 -0
- package/skills/specs/references/completeness-checklist.md +83 -0
- package/skills/specs/references/extraction.md +58 -0
- package/skills/specs/references/glossary.md +59 -0
- package/skills/specs/references/persistence.md +115 -0
- package/skills/specs/references/predictability-check.md +77 -0
- package/skills/static-analysis-integration/SKILL.md +235 -0
- package/skills/static-analysis-integration/adapters/_envelope.py +26 -0
- package/skills/static-analysis-integration/adapters/jscpd-adapter.py +66 -0
- package/skills/static-analysis-integration/adapters/lizard-adapter.py +81 -0
- package/skills/static-analysis-integration/adapters/mypy-adapter.py +50 -0
- package/skills/static-analysis-integration/adapters/mypy-src-layout.py +93 -0
- package/skills/static-analysis-integration/adapters/security-review-adapter.py +212 -0
- package/skills/static-analysis-integration/maintenance.md +23 -0
- package/skills/static-analysis-integration/references/language-setup.md +228 -0
- package/skills/static-analysis-integration/references/sarif-parser.md +124 -0
- package/skills/static-analysis-integration/references/security-review-adapter.md +118 -0
- package/skills/static-analysis-integration/references/tool-configs.md +617 -0
- package/skills/static-analysis-integration/rulesets/pmd-quickstart.xml +24 -0
- package/skills/stryker-xunit-v2-shim/SKILL.md +274 -0
- package/skills/stryker-xunit-v2-shim/references/shim-howto.md +256 -0
- package/skills/stryker-xunit-v2-shim/scripts/generate_shim.py +143 -0
- package/skills/systematic-debugging/SKILL.md +130 -0
- package/skills/telemetry/SKILL.md +75 -0
- package/skills/test-audit-disable/SKILL.md +129 -0
- package/skills/test-design/SKILL.md +177 -0
- package/skills/test-design/scripts/__pycache__/internal_double_detector.cpython-314.pyc +0 -0
- package/skills/test-design/scripts/internal_double_detector.py +631 -0
- package/skills/test-design-advisor/SKILL.md +166 -0
- package/skills/test-driven-development/SKILL.md +169 -0
- package/skills/test-health/SKILL.md +262 -0
- package/skills/test-improve/SKILL.md +239 -0
- package/skills/test-improve/references/phase-0-approach-contract.md +228 -0
- package/skills/test-improve/references/phase-1-analyze.md +131 -0
- package/skills/test-improve/references/phase-2-baseline.md +121 -0
- package/skills/test-improve/references/phase-3-derive-gherkin.md +53 -0
- package/skills/test-improve/references/phase-4-plan-fixes.md +34 -0
- package/skills/test-improve/references/phase-5-improve.md +215 -0
- package/skills/test-improve/references/phase-6-refactor-decision.md +45 -0
- package/skills/test-improve/references/phase-7-refactor.md +44 -0
- package/skills/test-improve/references/phase-8-validate.md +66 -0
- package/skills/test-improve/references/phase-9-close-out-prompt.md +11 -0
- package/skills/test-improve/references/phase-9-report.md +62 -0
- package/skills/test-improve/references/review-loop.md +92 -0
- package/skills/test-improve/templates/executive-summary.md +123 -0
- package/skills/threat-modeling/SKILL.md +108 -0
- package/skills/triage/SKILL.md +211 -0
- package/skills/ubiquitous-language/SKILL.md +192 -0
- package/skills/ubiquitous-language/scripts/collect_domain_signals.py +300 -0
- package/skills/unfreeze/SKILL.md +37 -0
- package/skills/upgrade/SKILL.md +31 -0
- package/skills/upgrade/scripts/check_version_drift.py +113 -0
- package/skills/upgrade/scripts/enable_autoupdate.py +149 -0
- package/skills/version/SKILL.md +25 -0
- package/sync/__pycache__/sync_upstream.cpython-314.pyc +0 -0
- package/sync/sync_upstream.py +293 -0
- package/templates/ACCEPTED-RISKS.md.tmpl +46 -0
- package/templates/agents/agent-template.md +151 -0
- package/templates/agents/angular-testing.md +66 -0
- package/templates/agents/csharp-quality.md +63 -0
- package/templates/agents/esm-enforcer.md +52 -0
- package/templates/agents/front-end-testing.md +65 -0
- package/templates/agents/go-quality.md +65 -0
- package/templates/agents/python-quality.md +62 -0
- package/templates/agents/react-testing.md +61 -0
- package/templates/agents/ts-enforcer.md +60 -0
- package/templates/agents/twelve-factor-audit.md +49 -0
- package/tools/entropy-check.py +250 -0
- package/tools/model-hash-verify.py +213 -0
|
@@ -0,0 +1,1066 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Deterministic pre-pass for repo-specific "every X has a Y" invariants (#1608).
|
|
3
|
+
|
|
4
|
+
`/code-review`'s final panel round on PR #1600 independently rediscovered the
|
|
5
|
+
same mechanically-checkable fact in four separate agent dispatches
|
|
6
|
+
(`doc-review`, `structure-review`, `ai-provenance-review`, `test-review`):
|
|
7
|
+
a newly-added script module under a `scripts/` directory had no corresponding
|
|
8
|
+
row in its skill's own documentation. That shape — "every module under
|
|
9
|
+
SCRIPTS_DIR should be named at least once in its skill's docs" — is a glob
|
|
10
|
+
check, not a semantic judgment call, but nothing stopped the full panel from
|
|
11
|
+
re-deriving it once per agent per round.
|
|
12
|
+
|
|
13
|
+
This module is a small, growable registry of such checks. Each check takes an
|
|
14
|
+
optional `changed_files` list (repo-relative paths for this review's
|
|
15
|
+
changeset, or `None` for "check everything") and returns a list of finding
|
|
16
|
+
dicts:
|
|
17
|
+
|
|
18
|
+
{"invariant": <str>, "file": <repo-relative str>, "message": <str>}
|
|
19
|
+
|
|
20
|
+
`changed_files` exists because some invariants are **required going forward
|
|
21
|
+
but explicitly not retrofitted** — the `_calibration` convention in
|
|
22
|
+
`evals/README.md` is the motivating case ("Required for every NEW fixture
|
|
23
|
+
going forward. Do not retrofit the ~140 existing fixtures — that is pure
|
|
24
|
+
churn with no discovered provenance to record"). Scoping such a check to the
|
|
25
|
+
changeset enforces it on exactly the fixtures being authored right now, which
|
|
26
|
+
is also where #1629 wants it: the author gets the finding **before** round 1
|
|
27
|
+
instead of from it. A check that applies corpus-wide simply ignores the
|
|
28
|
+
argument.
|
|
29
|
+
|
|
30
|
+
**Not Python-specific — a check operates on whatever file types the
|
|
31
|
+
invariant it's proving is about.** The one shipped here happens to walk a
|
|
32
|
+
directory that is entirely `.py` today only because this repo's own shipped
|
|
33
|
+
scripts are Python-only by convention (ADR 0014/0015); the glob itself
|
|
34
|
+
matches every file in that directory regardless of extension, and a future
|
|
35
|
+
check is free to target `.ts`/`.cs`/`.java`/`.go`/anything else this repo or
|
|
36
|
+
a downstream project's own conventions call for — `/code-review` runs
|
|
37
|
+
against projects in every language this plugin supports (see
|
|
38
|
+
`skills/static-analysis-integration/references/tool-configs.md`'s
|
|
39
|
+
per-language tool tiers), so new checks should not assume a Python target
|
|
40
|
+
just because the first one did. Add new checks by writing a function and
|
|
41
|
+
appending it to `CHECKS` below. Start narrow — this ships with exactly one
|
|
42
|
+
check (mutation-testing scripts documented) — and expand opportunistically
|
|
43
|
+
as more "N agents rediscovered the same mechanical fact" cases turn up (see
|
|
44
|
+
the issue for the intended pattern).
|
|
45
|
+
|
|
46
|
+
Wired into `/code-review` step 2b (see `skills/code-review/SKILL.md`):
|
|
47
|
+
findings are injected into agent context the same way static-analysis
|
|
48
|
+
findings already are — "detected by static analysis, do not re-report,
|
|
49
|
+
focus on semantic concerns" — so agents stop spending tokens re-deriving
|
|
50
|
+
facts this script already proved.
|
|
51
|
+
|
|
52
|
+
Stdlib-only. See docs/python-hook-contract.md.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
from __future__ import annotations
|
|
56
|
+
|
|
57
|
+
import argparse
|
|
58
|
+
import json
|
|
59
|
+
import re
|
|
60
|
+
import subprocess
|
|
61
|
+
import sys
|
|
62
|
+
from pathlib import Path
|
|
63
|
+
|
|
64
|
+
# skills/code-review/scripts -> skills/code-review -> skills -> plugin root
|
|
65
|
+
_PLUGIN_ROOT = Path(__file__).resolve().parents[3]
|
|
66
|
+
|
|
67
|
+
# The `agents/` directory root and its *-review.md glob are the shared,
|
|
68
|
+
# resolved single source of truth in hooks/lib (#1904 item 3) — scripts/ ->
|
|
69
|
+
# hooks/lib/ is the correct dependency direction (see
|
|
70
|
+
# review_agent_registry.py's own docstring). Import rather than re-deriving
|
|
71
|
+
# `_PLUGIN_ROOT / "agents"` locally.
|
|
72
|
+
sys.path.insert(0, str(_PLUGIN_ROOT / "hooks" / "lib"))
|
|
73
|
+
import boundary_events
|
|
74
|
+
import review_dispatch_ledger
|
|
75
|
+
from review_agent_registry import (
|
|
76
|
+
default_agents_dir,
|
|
77
|
+
find_review_agent_files,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _read_text(path: Path) -> str:
|
|
82
|
+
try:
|
|
83
|
+
return path.read_text(encoding="utf-8")
|
|
84
|
+
except OSError:
|
|
85
|
+
return ""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
_IGNORED_SCRIPT_NAMES = frozenset({"__init__.py", "__pycache__"})
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
#: Skills whose `scripts/` modules must each be named in that skill's own
|
|
92
|
+
#: documentation. A registry rather than one function per skill: the second
|
|
93
|
+
#: skill needing this check arrived (#1981) and copying the first would have
|
|
94
|
+
#: reproduced the duplication the review lenses exist to flag.
|
|
95
|
+
#:
|
|
96
|
+
#: Each entry: (skill name, extra doc paths relative to the plugin root). The
|
|
97
|
+
#: skill's own `SKILL.md` and every `references/**/*.md` under it are always
|
|
98
|
+
#: part of the doc set; `extra` is for docs that live outside the skill dir.
|
|
99
|
+
_DOCUMENTED_SCRIPT_SKILLS = (
|
|
100
|
+
("mutation-testing", ("agents/mutation-kill.md",), "mutation-kill-scripts-documented"),
|
|
101
|
+
("code-review", (), "code-review-scripts-documented"),
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _skill_doc_text(skill: str, extra: tuple) -> str:
|
|
106
|
+
skill_dir = _PLUGIN_ROOT / "skills" / skill
|
|
107
|
+
doc_files = [skill_dir / "SKILL.md"]
|
|
108
|
+
doc_files.extend(_PLUGIN_ROOT / rel for rel in extra)
|
|
109
|
+
for sub in ("references", ""):
|
|
110
|
+
target = skill_dir / sub if sub else skill_dir
|
|
111
|
+
if target.is_dir():
|
|
112
|
+
doc_files.extend(sorted(target.rglob("*.md")))
|
|
113
|
+
return "\n".join(_read_text(p) for p in doc_files)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def check_skill_scripts_documented(changed_files=None) -> list[dict]:
|
|
117
|
+
"""Every module under a registered skill's scripts/ dir must be named at
|
|
118
|
+
least once in that skill's own documentation, so a reviewer can find a
|
|
119
|
+
script's purpose without re-deriving it from source.
|
|
120
|
+
|
|
121
|
+
This is the invariant that created this module: on PR #1600 four separate
|
|
122
|
+
agents (`doc-review`, `structure-review`, `ai-provenance-review`,
|
|
123
|
+
`test-review`) independently rediscovered that a newly-added script had no
|
|
124
|
+
documentation row. It shipped covering `mutation-testing` alone.
|
|
125
|
+
|
|
126
|
+
#1981 is the second report of the same class, in a different directory:
|
|
127
|
+
`skills/code-review/scripts/` had gained `review_value_coverage.py` (from
|
|
128
|
+
#2020) with no mention in its own skill's docs. Under this repo's ratchet
|
|
129
|
+
rule a twice-reported mechanical class becomes a check, so the hardcoded
|
|
130
|
+
single-skill form became this registry.
|
|
131
|
+
|
|
132
|
+
Matches every file regardless of extension — these directories happen to
|
|
133
|
+
be all-Python today (ADR 0014/0015), but "every module is documented" is
|
|
134
|
+
language-agnostic and must keep holding if a differently-extensioned file
|
|
135
|
+
lands.
|
|
136
|
+
|
|
137
|
+
Corpus-wide by design: `changed_files` is ignored. An undocumented script
|
|
138
|
+
is a standing gap whether or not this changeset touched it, and the whole
|
|
139
|
+
point is that the panel is told about it once instead of N agents each
|
|
140
|
+
finding it.
|
|
141
|
+
"""
|
|
142
|
+
findings = []
|
|
143
|
+
for skill, extra, invariant in _DOCUMENTED_SCRIPT_SKILLS:
|
|
144
|
+
scripts_dir = _PLUGIN_ROOT / "skills" / skill / "scripts"
|
|
145
|
+
if not scripts_dir.is_dir():
|
|
146
|
+
continue
|
|
147
|
+
combined = _skill_doc_text(skill, extra)
|
|
148
|
+
for script in sorted(scripts_dir.iterdir()):
|
|
149
|
+
if not script.is_file() or script.name in _IGNORED_SCRIPT_NAMES:
|
|
150
|
+
continue
|
|
151
|
+
if script.name in combined:
|
|
152
|
+
continue
|
|
153
|
+
findings.append(
|
|
154
|
+
{
|
|
155
|
+
"invariant": invariant,
|
|
156
|
+
"file": str(script.relative_to(_PLUGIN_ROOT)),
|
|
157
|
+
"message": (
|
|
158
|
+
f"{script.name} is not named anywhere in the {skill} "
|
|
159
|
+
"skill's own documentation set (its SKILL.md or "
|
|
160
|
+
"references/**/*.md). Add a mention so reviewers "
|
|
161
|
+
"don't have to re-derive its purpose from source."
|
|
162
|
+
),
|
|
163
|
+
}
|
|
164
|
+
)
|
|
165
|
+
return findings
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
# --- #1629: churn generators observed in PR #1619 -------------------------
|
|
169
|
+
#
|
|
170
|
+
# Of #1619's 8 follow-up review rounds, at least 4 were triggered by defect
|
|
171
|
+
# classes that never needed an opus reviewer to catch. Each check below
|
|
172
|
+
# encodes one of them, so the author sees it at edit time and the panel gets
|
|
173
|
+
# "already detected — do not re-report" framing instead of N agents
|
|
174
|
+
# rediscovering the same mechanical fact.
|
|
175
|
+
|
|
176
|
+
_REPO_ROOT = _PLUGIN_ROOT.parents[1]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _repo_relative(path: Path) -> str:
|
|
180
|
+
"""Repo-relative path as a forward-slash string, matching `_changed_set`'s
|
|
181
|
+
own normalization. On Windows, `str(Path(...))` renders native
|
|
182
|
+
backslashes — comparing that directly against `_changed_set`'s
|
|
183
|
+
forward-slash-normalized entries (`rel not in changed`, used by every
|
|
184
|
+
changed-file-scoped check below) never matches, silently emptying every
|
|
185
|
+
finding on Windows regardless of what actually changed."""
|
|
186
|
+
try:
|
|
187
|
+
rel = str(path.relative_to(_REPO_ROOT))
|
|
188
|
+
except ValueError:
|
|
189
|
+
rel = str(path)
|
|
190
|
+
return rel.replace("\\", "/")
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _changed_set(changed_files):
|
|
194
|
+
if changed_files is None:
|
|
195
|
+
return None
|
|
196
|
+
out = set()
|
|
197
|
+
for raw in changed_files:
|
|
198
|
+
name = str(raw or "").strip().replace("\\", "/")
|
|
199
|
+
while name.startswith("./"):
|
|
200
|
+
name = name[2:]
|
|
201
|
+
if name:
|
|
202
|
+
out.add(name)
|
|
203
|
+
return out
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _load_json(path: Path):
|
|
207
|
+
try:
|
|
208
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
209
|
+
except (OSError, ValueError):
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _expectation_entries(spec: dict):
|
|
214
|
+
"""Yield `(target_name, entry_dict)` for every agent/skill expectation."""
|
|
215
|
+
for section in ("agents", "skills"):
|
|
216
|
+
block = spec.get(section)
|
|
217
|
+
if isinstance(block, dict):
|
|
218
|
+
for name, entry in block.items():
|
|
219
|
+
if isinstance(entry, dict):
|
|
220
|
+
yield name, entry
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _declares_tolerance_window(entry: dict) -> bool:
|
|
224
|
+
"""True when this expectation carries a `min`/`max` tolerance window —
|
|
225
|
+
the precondition `evals/README.md` attaches the `_calibration`
|
|
226
|
+
requirement to. An expectation with only `expectedStatus` and keyword
|
|
227
|
+
lists has no bounds whose provenance could be recorded."""
|
|
228
|
+
count = entry.get("issueCount")
|
|
229
|
+
if isinstance(count, dict) and ("min" in count or "max" in count):
|
|
230
|
+
return True
|
|
231
|
+
severities = entry.get("severities")
|
|
232
|
+
if isinstance(severities, dict):
|
|
233
|
+
for bounds in severities.values():
|
|
234
|
+
if isinstance(bounds, dict) and ("min" in bounds or "max" in bounds):
|
|
235
|
+
return True
|
|
236
|
+
return False
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def check_eval_calibration_blocks(changed_files=None) -> list[dict]:
|
|
240
|
+
"""Every NEW `evals/expected/*.json` expectation that declares a `min`/
|
|
241
|
+
`max` tolerance window must carry the `_calibration` block
|
|
242
|
+
`evals/README.md` requires.
|
|
243
|
+
|
|
244
|
+
#1619's round 1 lost a full `correctness-review` dispatch to exactly this
|
|
245
|
+
— a mechanically checkable convention miss.
|
|
246
|
+
|
|
247
|
+
**Scoped to `changed_files` when given.** The README explicitly forbids
|
|
248
|
+
retrofitting the ~140 pre-existing fixtures, so a corpus-wide sweep would
|
|
249
|
+
emit ~140 findings the convention says not to act on. With no changeset
|
|
250
|
+
the check reports nothing rather than every legacy fixture.
|
|
251
|
+
"""
|
|
252
|
+
changed = _changed_set(changed_files)
|
|
253
|
+
if changed is None:
|
|
254
|
+
return []
|
|
255
|
+
|
|
256
|
+
expected_dir = _REPO_ROOT / "evals" / "expected"
|
|
257
|
+
findings = []
|
|
258
|
+
for name in sorted(changed):
|
|
259
|
+
if not (name.startswith("evals/expected/") and name.endswith(".json")):
|
|
260
|
+
continue
|
|
261
|
+
path = _REPO_ROOT / name
|
|
262
|
+
if not path.is_file():
|
|
263
|
+
continue
|
|
264
|
+
spec = _load_json(path)
|
|
265
|
+
if not isinstance(spec, dict):
|
|
266
|
+
continue
|
|
267
|
+
for target, entry in _expectation_entries(spec):
|
|
268
|
+
if not _declares_tolerance_window(entry):
|
|
269
|
+
continue
|
|
270
|
+
calibration = entry.get("_calibration")
|
|
271
|
+
if isinstance(calibration, dict) and calibration.get("source"):
|
|
272
|
+
continue
|
|
273
|
+
findings.append(
|
|
274
|
+
{
|
|
275
|
+
"invariant": "eval-calibration-block-required",
|
|
276
|
+
"file": _repo_relative(path),
|
|
277
|
+
"message": (
|
|
278
|
+
f"expectation for {target!r} declares a min/max tolerance "
|
|
279
|
+
"window but carries no `_calibration` block. "
|
|
280
|
+
"evals/README.md requires one on every new fixture with "
|
|
281
|
+
"bounds: {\"source\": \"measured\"|\"estimated-by-analogy\", "
|
|
282
|
+
'"note": "<one-line rationale>"}. Without it, a later '
|
|
283
|
+
"\"tidy up the ranges\" pass can silently widen a bound that "
|
|
284
|
+
"was tuned against this fixture's own measured behavior."
|
|
285
|
+
),
|
|
286
|
+
}
|
|
287
|
+
)
|
|
288
|
+
_ = expected_dir # documented location; findings are keyed off changed paths
|
|
289
|
+
return findings
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def check_must_not_mention_terms_appear_in_fixture(changed_files=None) -> list[dict]:
|
|
293
|
+
"""Every `mustNotMention` term should actually appear somewhere in its
|
|
294
|
+
paired fixture — otherwise the guard is vacuous.
|
|
295
|
+
|
|
296
|
+
`mustNotMention` is an all-of "none of these may appear in the agent's
|
|
297
|
+
output" assertion. If a forbidden term does not occur in the fixture at
|
|
298
|
+
all, the agent had no reason to emit it and the assertion passes
|
|
299
|
+
trivially, proving nothing while reading like coverage. This is the
|
|
300
|
+
corpus-level half of the negation-blindness trap `docs/eval-maintenance.md`
|
|
301
|
+
documents (#1622).
|
|
302
|
+
|
|
303
|
+
Coordinates with, rather than duplicates, `scripts/eval_grade.py
|
|
304
|
+
--check-corpus`: if that gate grows this rule, delete this check and let
|
|
305
|
+
the pre-pass call the grader instead. Today `--check-corpus` has no
|
|
306
|
+
`mustNotMention` rule, so this is the only place it is enforced.
|
|
307
|
+
|
|
308
|
+
**Scoped to `changed_files`**, like the calibration check and for the same
|
|
309
|
+
reason: the corpus carries ~31 pre-existing hits (measured 2026-07-31),
|
|
310
|
+
some of which are deliberate — `security-review`'s hardcoded-secrets
|
|
311
|
+
fixture forbids "environment variable" to stop the agent recommending a
|
|
312
|
+
weak fix, and that term legitimately isn't in the fixture. Sweeping
|
|
313
|
+
corpus-wide inside every `/code-review` pre-pass would bury each panel in
|
|
314
|
+
unrelated legacy findings, the exact opposite of this slice's purpose.
|
|
315
|
+
Run with `--all` to triage that backlog deliberately.
|
|
316
|
+
"""
|
|
317
|
+
changed = _changed_set(changed_files)
|
|
318
|
+
if changed is None:
|
|
319
|
+
return []
|
|
320
|
+
expected_dir = _REPO_ROOT / "evals" / "expected"
|
|
321
|
+
fixtures_dir = _REPO_ROOT / "evals" / "fixtures"
|
|
322
|
+
if not expected_dir.is_dir() or not fixtures_dir.is_dir():
|
|
323
|
+
return []
|
|
324
|
+
|
|
325
|
+
fixture_text_by_stem = {}
|
|
326
|
+
for path in fixtures_dir.iterdir():
|
|
327
|
+
stem = path.name if path.is_dir() else path.stem
|
|
328
|
+
if path.is_file():
|
|
329
|
+
fixture_text_by_stem[stem] = _read_text(path).lower()
|
|
330
|
+
|
|
331
|
+
findings = []
|
|
332
|
+
for path in sorted(expected_dir.glob("*.json")):
|
|
333
|
+
rel = _repo_relative(path)
|
|
334
|
+
if changed is not None and rel not in changed:
|
|
335
|
+
continue
|
|
336
|
+
spec = _load_json(path)
|
|
337
|
+
if not isinstance(spec, dict):
|
|
338
|
+
continue
|
|
339
|
+
fixture_text = fixture_text_by_stem.get(path.stem)
|
|
340
|
+
if fixture_text is None:
|
|
341
|
+
# No readable paired fixture (a directory fixture, or a missing
|
|
342
|
+
# one --check-corpus already warns about). Nothing to prove here.
|
|
343
|
+
continue
|
|
344
|
+
for target, entry in _expectation_entries(spec):
|
|
345
|
+
for term in entry.get("mustNotMention") or []:
|
|
346
|
+
if not isinstance(term, str) or not term.strip():
|
|
347
|
+
continue
|
|
348
|
+
if term.lower() in fixture_text:
|
|
349
|
+
continue
|
|
350
|
+
findings.append(
|
|
351
|
+
{
|
|
352
|
+
"invariant": "must-not-mention-term-absent-from-fixture",
|
|
353
|
+
"file": rel,
|
|
354
|
+
"message": (
|
|
355
|
+
f"{target!r} forbids {term!r} via mustNotMention, but that "
|
|
356
|
+
"string never appears in the paired fixture — the agent had "
|
|
357
|
+
"no reason to emit it, so the assertion passes trivially and "
|
|
358
|
+
"proves nothing. Either drop the term (see "
|
|
359
|
+
"docs/eval-maintenance.md's negation-blindness trap) or point "
|
|
360
|
+
"it at something the fixture actually contains."
|
|
361
|
+
),
|
|
362
|
+
}
|
|
363
|
+
)
|
|
364
|
+
return findings
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
#: `Scope:` glob extensions vs. the prose Skip rule's extension list. A
|
|
368
|
+
#: review agent that declares `**/*.{js,mjs,cjs,ts}` in `Scope:` but whose
|
|
369
|
+
#: Skip section only names `.js`/`.ts` self-skips on files the resolver
|
|
370
|
+
#: correctly routed to it — the `.mjs`/`.cjs` mismatch class from #1622.
|
|
371
|
+
_SCOPE_BLOCK_RE = re.compile(r"^\s*Scope\s*:\s*(.*)$", re.MULTILINE)
|
|
372
|
+
_EXT_RE = re.compile(r"\.([a-z0-9]{1,6})\b", re.IGNORECASE)
|
|
373
|
+
_BRACE_RE = re.compile(r"\{([^}]*)\}")
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _scope_extensions(body: str) -> set:
|
|
377
|
+
"""Extensions named by an agent's `Scope:` glob list, expanding brace
|
|
378
|
+
alternation (`*.{js,mjs}`)."""
|
|
379
|
+
match = _SCOPE_BLOCK_RE.search(body)
|
|
380
|
+
if not match:
|
|
381
|
+
return set()
|
|
382
|
+
# A Scope: block may be a scalar on one line or a following YAML-ish list.
|
|
383
|
+
start = match.end()
|
|
384
|
+
lines = [match.group(1)]
|
|
385
|
+
for line in body[start:].splitlines():
|
|
386
|
+
if line.startswith(("-", " ", "\t")) and line.strip():
|
|
387
|
+
lines.append(line)
|
|
388
|
+
elif line.strip():
|
|
389
|
+
break
|
|
390
|
+
text = "\n".join(lines)
|
|
391
|
+
exts = set()
|
|
392
|
+
for group in _BRACE_RE.findall(text):
|
|
393
|
+
for part in group.split(","):
|
|
394
|
+
part = part.strip().lstrip(".")
|
|
395
|
+
if part and re.fullmatch(r"[a-z0-9]{1,6}", part, re.IGNORECASE):
|
|
396
|
+
exts.add("." + part.lower())
|
|
397
|
+
text = _BRACE_RE.sub(" ", text)
|
|
398
|
+
exts.update(m.group(0).lower() for m in _EXT_RE.finditer(text))
|
|
399
|
+
return exts
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def _skip_section_extensions(body: str) -> set:
|
|
403
|
+
"""Extensions named in the agent's prose `## Skip` section."""
|
|
404
|
+
match = re.search(r"^##\s+Skip\s*$", body, re.MULTILINE)
|
|
405
|
+
if not match:
|
|
406
|
+
return set()
|
|
407
|
+
rest = body[match.end() :]
|
|
408
|
+
end = re.search(r"^##\s+", rest, re.MULTILINE)
|
|
409
|
+
section = rest[: end.start()] if end else rest
|
|
410
|
+
return {m.group(0).lower() for m in _EXT_RE.finditer(section)}
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def check_scope_glob_matches_skip_prose(changed_files=None) -> list[dict]:
|
|
414
|
+
"""A review agent's `Scope:` globs and its prose `## Skip` rule must name
|
|
415
|
+
the same file extensions.
|
|
416
|
+
|
|
417
|
+
When `Scope:` routes `.mjs`/`.cjs` to an agent whose Skip section only
|
|
418
|
+
lists `.js`/`.ts`, the agent self-skips files the resolver deliberately
|
|
419
|
+
sent it — a silent coverage hole no runtime check catches, and the
|
|
420
|
+
mismatch class #1622 found. Only extensions the Skip section could
|
|
421
|
+
plausibly be enumerating are compared: a Skip section naming no
|
|
422
|
+
extensions at all is not making a claim about file types.
|
|
423
|
+
|
|
424
|
+
**Scoped to `changed_files`** for consistency with the two checks above.
|
|
425
|
+
Three pre-existing hits exist as of 2026-07-31 — `js-fp-review`
|
|
426
|
+
(`.mjs`/`.cjs`, the original #1622 case), `angular-reactivity-review`,
|
|
427
|
+
and `component-architecture-review` — all real, none fixed by this slice,
|
|
428
|
+
which adds the detector rather than the corrections. `--all` surfaces
|
|
429
|
+
them for deliberate triage.
|
|
430
|
+
"""
|
|
431
|
+
changed = _changed_set(changed_files)
|
|
432
|
+
if changed is None:
|
|
433
|
+
return []
|
|
434
|
+
agents_dir = default_agents_dir()
|
|
435
|
+
if not agents_dir.is_dir():
|
|
436
|
+
return []
|
|
437
|
+
|
|
438
|
+
findings = []
|
|
439
|
+
for path in find_review_agent_files(agents_dir):
|
|
440
|
+
rel = _repo_relative(path)
|
|
441
|
+
if changed is not None and rel not in changed:
|
|
442
|
+
continue
|
|
443
|
+
body = _read_text(path)
|
|
444
|
+
scope_exts = _scope_extensions(body)
|
|
445
|
+
skip_exts = _skip_section_extensions(body)
|
|
446
|
+
if not scope_exts or not skip_exts:
|
|
447
|
+
continue
|
|
448
|
+
missing = sorted(scope_exts - skip_exts)
|
|
449
|
+
if not missing:
|
|
450
|
+
continue
|
|
451
|
+
findings.append(
|
|
452
|
+
{
|
|
453
|
+
"invariant": "scope-glob-skip-prose-extension-drift",
|
|
454
|
+
"file": rel,
|
|
455
|
+
"message": (
|
|
456
|
+
f"Scope: routes {', '.join(missing)} to this agent, but its "
|
|
457
|
+
"## Skip section never names those extensions. The agent will "
|
|
458
|
+
"self-skip files the resolver deliberately sent it — a silent "
|
|
459
|
+
"coverage hole. Add them to the Skip prose, or narrow the "
|
|
460
|
+
"Scope: globs so the two agree."
|
|
461
|
+
),
|
|
462
|
+
}
|
|
463
|
+
)
|
|
464
|
+
return findings
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def check_contract_failure_shapes_documented(changed_files=None) -> list[dict]:
|
|
468
|
+
"""`telemetry-schema.md`'s `contract-failures.jsonl` `shape` table row
|
|
469
|
+
must enumerate exactly `validate_review_output.FAILURE_SHAPES` (#1998).
|
|
470
|
+
|
|
471
|
+
Two independent review agents (doc-review in wave 1, domain-review in
|
|
472
|
+
wave 2 — same PR) rediscovered the same drift: prose enumerating the
|
|
473
|
+
loggable failure shapes disagreeing with the module's actual behavior.
|
|
474
|
+
`SKILL.md` no longer re-enumerates the set itself (it now points at
|
|
475
|
+
`telemetry-schema.md` instead), so only one doc can drift from the code
|
|
476
|
+
now — this check pins that one doc to the module's exported set rather
|
|
477
|
+
than trusting prose to stay in sync by hand.
|
|
478
|
+
"""
|
|
479
|
+
telemetry_schema = _PLUGIN_ROOT / "knowledge" / "telemetry-schema.md"
|
|
480
|
+
text = _read_text(telemetry_schema)
|
|
481
|
+
match = re.search(r"\|\s*`shape`\s*\|\s*string enum\s*\|([^\n]*)", text)
|
|
482
|
+
if not match:
|
|
483
|
+
return [
|
|
484
|
+
{
|
|
485
|
+
"invariant": "contract-failure-shapes-documented",
|
|
486
|
+
"file": "knowledge/telemetry-schema.md",
|
|
487
|
+
"message": (
|
|
488
|
+
"Could not find the contract-failures.jsonl `shape` table row to "
|
|
489
|
+
"check against validate_review_output.FAILURE_SHAPES — has the "
|
|
490
|
+
"table row been reworded or removed?"
|
|
491
|
+
),
|
|
492
|
+
}
|
|
493
|
+
]
|
|
494
|
+
cell = match.group(1).split(" — ", 1)[0]
|
|
495
|
+
documented = frozenset(re.findall(r"`([a-z-]+)`", cell))
|
|
496
|
+
|
|
497
|
+
from validate_review_output import FAILURE_SHAPES
|
|
498
|
+
|
|
499
|
+
if documented == FAILURE_SHAPES:
|
|
500
|
+
return []
|
|
501
|
+
return [
|
|
502
|
+
{
|
|
503
|
+
"invariant": "contract-failure-shapes-documented",
|
|
504
|
+
"file": "knowledge/telemetry-schema.md",
|
|
505
|
+
"message": (
|
|
506
|
+
f"telemetry-schema.md's contract-failures.jsonl `shape` row documents "
|
|
507
|
+
f"{sorted(documented)}, but validate_review_output.FAILURE_SHAPES is "
|
|
508
|
+
f"{sorted(FAILURE_SHAPES)} — keep the table row and the module's "
|
|
509
|
+
"exported set in sync."
|
|
510
|
+
),
|
|
511
|
+
}
|
|
512
|
+
]
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
# --- #2048: a second transcript parser must not reappear -------------------
|
|
516
|
+
#
|
|
517
|
+
# ADR 0036 records that both `structure-review` and `arch-review` raised the
|
|
518
|
+
# same duplication independently while reviewing #1991 -- two scripts each
|
|
519
|
+
# carrying their own copy of transcript-record and usage-block parsing,
|
|
520
|
+
# already drifted twice on the same defect class (#1990/#1991/#1994).
|
|
521
|
+
# Epic #2040 unified both into `session_log/` + one entry point
|
|
522
|
+
# (`session_report.py`). This check is the ratchet that keeps a THIRD
|
|
523
|
+
# independent copy from growing back once the two originals are gone.
|
|
524
|
+
|
|
525
|
+
#: Any file outside `plugins/dev-team/scripts/lib/session_log/` that
|
|
526
|
+
#: references one of these identifiers is either genuinely parsing a raw
|
|
527
|
+
#: transcript record / usage block (a real second implementation) or reading
|
|
528
|
+
#: an already-extracted usage dict's known numeric fields (not the same
|
|
529
|
+
#: failure mode) -- `_TRANSCRIPT_PARSING_ALLOWLIST` below distinguishes the
|
|
530
|
+
#: two per-file, with a reason each. `cache_creation_input_tokens` and
|
|
531
|
+
#: `cache_read_input_tokens` are usage-block field names; `isSidechain` and
|
|
532
|
+
#: `attributionAgent` are raw-record top-level fields no usage-only consumer
|
|
533
|
+
#: would ever need, so their presence is the stronger of the two signals.
|
|
534
|
+
_TRANSCRIPT_FIELD_RE = re.compile(
|
|
535
|
+
r"\b(cache_creation_input_tokens|cache_read_input_tokens|isSidechain|attributionAgent)\b"
|
|
536
|
+
)
|
|
537
|
+
|
|
538
|
+
#: Directories (repo-root-relative) this check scans -- everywhere a
|
|
539
|
+
#: transcript-parsing module has actually turned up historically (the
|
|
540
|
+
#: shipped plugin tree, and the monorepo's own repo-root `scripts/`, where
|
|
541
|
+
#: `measure_full_file_duplication.py` and the eval/experiment harnesses
|
|
542
|
+
#: live). Repo-root `tests/` is deliberately not a scan root: fixture
|
|
543
|
+
#: literals constructing synthetic usage dicts are not "parsing," and
|
|
544
|
+
#: scanning them would bury the real findings in noise.
|
|
545
|
+
_TRANSCRIPT_SCAN_ROOTS = ("plugins/dev-team", "scripts")
|
|
546
|
+
|
|
547
|
+
#: Shrink-only (docs/adr/0032's "mechanically, not by comment alone"
|
|
548
|
+
#: pattern, mirrored here for #2048's invariant). Each entry states WHY the
|
|
549
|
+
#: match is not the failure mode this check targets. Two shapes:
|
|
550
|
+
#: - "the sanctioned entry point" -- composes session_log's own shared
|
|
551
|
+
#: primitives rather than reimplementing them independently;
|
|
552
|
+
#: - "reads a pre-extracted usage dict" -- consumes a `usage` mapping a
|
|
553
|
+
#: caller already extracted (a handful of known numeric field names),
|
|
554
|
+
#: never a raw transcript record (no isSidechain/attributionAgent
|
|
555
|
+
#: access) -- a materially different, much narrower concern than
|
|
556
|
+
#: parsing a transcript.
|
|
557
|
+
#: A "migrated in #2050" entry is a real, temporary exception: that slice's
|
|
558
|
+
#: job is to fold it onto session_log.records and delete it from this list.
|
|
559
|
+
_TRANSCRIPT_PARSING_ALLOWLIST = {
|
|
560
|
+
"plugins/dev-team/scripts/lib/session_report_maintainer.py": (
|
|
561
|
+
"the maintainer half of the one sanctioned entry point over "
|
|
562
|
+
"session_log (#2046/#2047), relocated from session_report.py "
|
|
563
|
+
"itself in issue #2098's layering split -- not a second, "
|
|
564
|
+
"independently-drifting implementation; it composes session_log's "
|
|
565
|
+
"own shared primitives (records.usage_of/usage_fields, "
|
|
566
|
+
"signals.CONTEXT_TOKEN_FIELDS, discovery.*) for its remaining "
|
|
567
|
+
"attribution/threading logic, mirroring exactly what its two "
|
|
568
|
+
"now-retired predecessors did"
|
|
569
|
+
),
|
|
570
|
+
"plugins/dev-team/scripts/lib/session_report_downstream.py": (
|
|
571
|
+
"the downstream half of the one sanctioned entry point over "
|
|
572
|
+
"session_log (#2046/#2047), relocated from session_report.py "
|
|
573
|
+
"itself in issue #2098's layering split -- same rationale as "
|
|
574
|
+
"session_report_maintainer.py's entry above"
|
|
575
|
+
),
|
|
576
|
+
"plugins/dev-team/hooks/lib/cost_meter.py": (
|
|
577
|
+
"migrated onto session_log.records in #2050 (join-map, sidechain, "
|
|
578
|
+
"and attribution logic all now delegate to _records.*); the four "
|
|
579
|
+
"identifiers remain on this file's own module docstring and inline "
|
|
580
|
+
"comments, which document the harness fields this hook's DECISIONS "
|
|
581
|
+
"are still based on -- prose, not a second parsing implementation"
|
|
582
|
+
),
|
|
583
|
+
"scripts/measure_full_file_duplication.py": (
|
|
584
|
+
"migrated onto session_log.records in #2050 -- the join-map "
|
|
585
|
+
"algorithm this file's own docstring once conceded duplicating is "
|
|
586
|
+
"now imported (session_log_records.join_dispatch_agent_ids), and "
|
|
587
|
+
"the local _usage_from_record/_join_dispatch_agent_ids copies are "
|
|
588
|
+
"deleted; the identifiers remain in this file's module docstring "
|
|
589
|
+
"(Privacy boundary section) describing which fields it reads"
|
|
590
|
+
),
|
|
591
|
+
"plugins/dev-team/hooks/subagent_completion_guard.py": (
|
|
592
|
+
"reads only message.stop_reason and message.content off the LAST "
|
|
593
|
+
"JSON row of a subagent's own transcript file -- never "
|
|
594
|
+
"isSidechain/attributionAgent/cache_*_input_tokens; the two "
|
|
595
|
+
"'isSidechain' occurrences are both in this file's own module "
|
|
596
|
+
"docstring, recording issue #2188's Step 2.1a research finding "
|
|
597
|
+
"(a subagent's own transcript file is isSidechain:true by "
|
|
598
|
+
"construction, so no sidechain filtering is needed). This file "
|
|
599
|
+
"DOES carry its own small last-row reader (_tail_lines/_last_row) "
|
|
600
|
+
"rather than session_log.records.iter_file_records -- deliberately: "
|
|
601
|
+
"iter_file_records silently skips an undecodable line and "
|
|
602
|
+
"continues, while this hook needs 'the trailing line is malformed "
|
|
603
|
+
"JSON' to classify as its own distinct outcome ('unreadable'), "
|
|
604
|
+
"which a streaming skip-and-continue reader cannot express. A "
|
|
605
|
+
"narrower concern than the four-identifier duplication this "
|
|
606
|
+
"invariant targets, not zero"
|
|
607
|
+
),
|
|
608
|
+
"plugins/dev-team/hooks/review_verdict_recorder.py": (
|
|
609
|
+
"attributionAgent is read only through session_log.records "
|
|
610
|
+
"(attribution_agent_of); the 'attributionAgent' occurrences this "
|
|
611
|
+
"check's own regex matches are all in this file's own module "
|
|
612
|
+
"docstring, recording #2166 Step 2.3's own pre-implementation spike "
|
|
613
|
+
"finding against 124 real subagent transcripts (mirrors "
|
|
614
|
+
"cost_meter.py's entry above: prose documenting the harness field "
|
|
615
|
+
"this hook's decisions are based on, not a second parsing "
|
|
616
|
+
"implementation). `agentId` is NOT one of this check's own scanned "
|
|
617
|
+
"identifiers (see _TRANSCRIPT_FIELD_RE above) and is read directly "
|
|
618
|
+
"as a plain top-level field (`_own_agent_id`'s `rec.get(\"agentId\")`) "
|
|
619
|
+
"rather than through session_log -- correcting this entry's prior, "
|
|
620
|
+
"inaccurate 'never a raw field access' claim about it (#2166 Fix "
|
|
621
|
+
"#9, correctness review). This file's own transcript reader, "
|
|
622
|
+
"`_read_transcript_records`, delegates to "
|
|
623
|
+
"session_log.records.iter_file_records (Fix #9) rather than "
|
|
624
|
+
"carrying a second whole-file reader"
|
|
625
|
+
),
|
|
626
|
+
"plugins/dev-team/hooks/lib/pricing.py": (
|
|
627
|
+
"reads a pre-extracted usage dict's known numeric fields "
|
|
628
|
+
"(cache_creation_input_tokens/cache_read_input_tokens) for cost "
|
|
629
|
+
"computation -- never a raw transcript record (no isSidechain/"
|
|
630
|
+
"attributionAgent access), so this is not the duplication this "
|
|
631
|
+
"invariant targets"
|
|
632
|
+
),
|
|
633
|
+
"plugins/dev-team/skills/headless-run/scripts/isolated_dispatch.py": (
|
|
634
|
+
"reads a pre-extracted usage dict's cache-token fields for its own "
|
|
635
|
+
"cost estimate -- never a raw transcript record"
|
|
636
|
+
),
|
|
637
|
+
"scripts/run_integration_eval.py": (
|
|
638
|
+
"reads a pre-extracted usage dict's known token fields to sum "
|
|
639
|
+
"eval-harness cost -- never a raw transcript record"
|
|
640
|
+
),
|
|
641
|
+
"scripts/run_refactor_experiment.py": (
|
|
642
|
+
"reads a pre-extracted usage dict's known token fields to sum "
|
|
643
|
+
"experiment-harness cost -- never a raw transcript record"
|
|
644
|
+
),
|
|
645
|
+
"scripts/run_tdd_experiment.py": (
|
|
646
|
+
"reads a pre-extracted usage dict's known token fields to sum "
|
|
647
|
+
"experiment-harness cost -- never a raw transcript record"
|
|
648
|
+
),
|
|
649
|
+
"scripts/measure_rereview_duplication.py": (
|
|
650
|
+
"#2165's empirical leg composes measure_full_file_duplication.py's "
|
|
651
|
+
"own sanctioned collect_agent_dispatches/filter_since (already "
|
|
652
|
+
"migrated onto session_log.records in #2050) rather than "
|
|
653
|
+
"reimplementing transcript parsing -- never a raw transcript "
|
|
654
|
+
"record read directly by this file; the four identifiers remain "
|
|
655
|
+
"in this file's own module docstring (Privacy boundary section) "
|
|
656
|
+
"documenting which fields the empirical leg's output is limited to"
|
|
657
|
+
),
|
|
658
|
+
}
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def check_transcript_parsing_confined_to_session_log(changed_files=None) -> list[dict]:
|
|
662
|
+
"""No module outside `plugins/dev-team/scripts/lib/session_log/` may
|
|
663
|
+
parse a transcript record or a usage block (#2048).
|
|
664
|
+
|
|
665
|
+
ADR 0036: both `structure-review` and `arch-review` independently
|
|
666
|
+
raised the same duplication while reviewing #1991 -- that is the
|
|
667
|
+
"second report" this repo's own ratchet rule converts into a check.
|
|
668
|
+
Without it, nothing stops a third independent transcript parser from
|
|
669
|
+
growing back the moment the two the epic just unified are gone.
|
|
670
|
+
|
|
671
|
+
Corpus-wide by design, like `check_skill_scripts_documented`:
|
|
672
|
+
`changed_files` is ignored. A stray transcript-parsing module is a
|
|
673
|
+
standing architectural gap whether or not this changeset touched it.
|
|
674
|
+
"""
|
|
675
|
+
findings = []
|
|
676
|
+
session_log_prefix = "plugins/dev-team/scripts/lib/session_log/"
|
|
677
|
+
self_rel = _repo_relative(Path(__file__).resolve())
|
|
678
|
+
for root_name in _TRANSCRIPT_SCAN_ROOTS:
|
|
679
|
+
root = _REPO_ROOT / root_name
|
|
680
|
+
if not root.is_dir():
|
|
681
|
+
continue
|
|
682
|
+
for path in sorted(root.rglob("*.py")):
|
|
683
|
+
rel = _repo_relative(path)
|
|
684
|
+
if rel.startswith(session_log_prefix):
|
|
685
|
+
continue
|
|
686
|
+
if "/tests/" in f"/{rel}" or Path(rel).name.startswith("test_"):
|
|
687
|
+
continue
|
|
688
|
+
# This module's own source names the four identifiers to detect
|
|
689
|
+
# them and to explain each allowlist entry's reason -- that is
|
|
690
|
+
# the check's own text, not a transcript parser.
|
|
691
|
+
if rel == self_rel:
|
|
692
|
+
continue
|
|
693
|
+
if rel in _TRANSCRIPT_PARSING_ALLOWLIST:
|
|
694
|
+
continue
|
|
695
|
+
if _TRANSCRIPT_FIELD_RE.search(_read_text(path)):
|
|
696
|
+
findings.append(
|
|
697
|
+
{
|
|
698
|
+
"invariant": "transcript-parsing-confined-to-session-log",
|
|
699
|
+
"file": rel,
|
|
700
|
+
"message": (
|
|
701
|
+
f"{rel} references a raw transcript-record or "
|
|
702
|
+
"usage-block field (cache_creation_input_tokens / "
|
|
703
|
+
"cache_read_input_tokens / isSidechain / "
|
|
704
|
+
"attributionAgent) outside "
|
|
705
|
+
"plugins/dev-team/scripts/lib/session_log/. Move "
|
|
706
|
+
"the parsing onto session_log's shared "
|
|
707
|
+
"primitives, or add this path to "
|
|
708
|
+
"repo_invariants._TRANSCRIPT_PARSING_ALLOWLIST "
|
|
709
|
+
"with a stated reason (see ADR 0036 / issue #2048)."
|
|
710
|
+
),
|
|
711
|
+
}
|
|
712
|
+
)
|
|
713
|
+
return findings
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
# --- #2108: churn_recurrence.py / churn_coupling_report.py render_text()
|
|
717
|
+
# `report["window"]` access must stay safe -----------------------------------
|
|
718
|
+
#
|
|
719
|
+
# Session-digest churn analysis (issue #2108) traced repeated review-round
|
|
720
|
+
# rework on the #2085 PR (bash-failure taxonomy + churn baselines slice) to
|
|
721
|
+
# the SAME bug recurring in two structurally-parallel renderers. Round 2's
|
|
722
|
+
# review fixed churn_recurrence.py's render_text(): it unconditionally read
|
|
723
|
+
# report["window"], a key only churn_coupling_report.py's CLI caller
|
|
724
|
+
# injects, so a caller rendering rank_all_files()'s own output directly hit
|
|
725
|
+
# a raw KeyError; fixed via report.get("window", "unknown"). Round 3's very
|
|
726
|
+
# next review pass found the IDENTICAL bug in churn_coupling_report.py's own
|
|
727
|
+
# render_text() -- same key, same failure mode, in the sibling file --
|
|
728
|
+
# because nothing pinned "these two report shapes agree on how this
|
|
729
|
+
# caller-optional key is read." A second occurrence of the same
|
|
730
|
+
# mechanically-checkable fact is this repo's own trigger to ratchet it into
|
|
731
|
+
# a check, applied one round late.
|
|
732
|
+
|
|
733
|
+
_CHURN_REPORT_WINDOW_KEY_FILES = (
|
|
734
|
+
"scripts/lib/churn_recurrence.py",
|
|
735
|
+
"scripts/churn_coupling_report.py",
|
|
736
|
+
)
|
|
737
|
+
_WINDOW_KEY_RE = re.compile(r"""report\s*\[\s*['"]window['"]\s*\]""")
|
|
738
|
+
_WINDOW_KEY_ASSIGNMENT_RE = re.compile(
|
|
739
|
+
r"""report\s*\[\s*['"]window['"]\s*\]\s*=(?!=)"""
|
|
740
|
+
)
|
|
741
|
+
|
|
742
|
+
|
|
743
|
+
def check_churn_report_window_key_safe_access(changed_files=None) -> list[dict]:
|
|
744
|
+
"""`churn_recurrence.py` and `churn_coupling_report.py`'s `render_text()`
|
|
745
|
+
must read `report["window"]` via `.get("window", "unknown")`, never a
|
|
746
|
+
bare index — see the section comment above for the twice-recurring bug
|
|
747
|
+
this pins (#2108). Assignment (`report["window"] = ...`, the CLI
|
|
748
|
+
caller's own injection site) is a different operation and is not
|
|
749
|
+
flagged. Corpus-wide by design, like the checks above: a regression on
|
|
750
|
+
either file is a standing gap whether or not this changeset touched it.
|
|
751
|
+
"""
|
|
752
|
+
findings = []
|
|
753
|
+
for rel in _CHURN_REPORT_WINDOW_KEY_FILES:
|
|
754
|
+
text = _read_text(_REPO_ROOT / rel)
|
|
755
|
+
if not text:
|
|
756
|
+
continue
|
|
757
|
+
for lineno, line in enumerate(text.splitlines(), start=1):
|
|
758
|
+
if not _WINDOW_KEY_RE.search(line):
|
|
759
|
+
continue
|
|
760
|
+
if _WINDOW_KEY_ASSIGNMENT_RE.search(line):
|
|
761
|
+
continue # a write (the CLI's own injection site), not a read
|
|
762
|
+
if ".get(" in line:
|
|
763
|
+
continue # already safe
|
|
764
|
+
findings.append(
|
|
765
|
+
{
|
|
766
|
+
"invariant": "churn-report-window-key-safe-access",
|
|
767
|
+
"file": rel,
|
|
768
|
+
"message": (
|
|
769
|
+
f"{rel}:{lineno} reads report['window'] via a bare "
|
|
770
|
+
"index. A caller rendering a report shape that never "
|
|
771
|
+
"sets this key (e.g. rank_all_files()'s own output) "
|
|
772
|
+
"raises KeyError — this exact bug already recurred "
|
|
773
|
+
"once, in the sibling renderer (#2085 round 2, then "
|
|
774
|
+
"round 3). Use report.get('window', 'unknown') "
|
|
775
|
+
"instead."
|
|
776
|
+
),
|
|
777
|
+
}
|
|
778
|
+
)
|
|
779
|
+
return findings
|
|
780
|
+
|
|
781
|
+
|
|
782
|
+
# --- #2126: internal-collaborator-doubling.md must stay single-sourced ------
|
|
783
|
+
#
|
|
784
|
+
# Epic #2123's own design note: "cite, don't restate" cannot be an
|
|
785
|
+
# acceptance criterion someone eyeballs; it needs a mechanism. Distinctive
|
|
786
|
+
# fragments copied verbatim from the normative file — chosen for
|
|
787
|
+
# distinctiveness (a short generic phrase like "setup is easier" would be a
|
|
788
|
+
# plausible false positive anywhere in testability prose) rather than
|
|
789
|
+
# derived programmatically, since the source is prose, not a table this
|
|
790
|
+
# script can parse structurally like `check_contract_failure_shapes_documented`
|
|
791
|
+
# does. A dedicated staleness self-check (test_repo_invariants.py) keeps
|
|
792
|
+
# this list honest against the home file it was copied from.
|
|
793
|
+
|
|
794
|
+
_NORMATIVE_CONTENT_HOME_FILE = "plugins/dev-team/knowledge/internal-collaborator-doubling.md"
|
|
795
|
+
|
|
796
|
+
_NORMATIVE_CONTENT_FRAGMENTS = (
|
|
797
|
+
"Out-of-process handle",
|
|
798
|
+
"Prohibitive real cost",
|
|
799
|
+
"the project's own first-party source stays real",
|
|
800
|
+
"\"It's an injected interface\" — and the type's name",
|
|
801
|
+
)
|
|
802
|
+
# Deliberately excluded, per correctness-review (#2126):
|
|
803
|
+
# - "Ambient state" -- too generic (plausible in unrelated testability prose,
|
|
804
|
+
# contradicting this list's own distinctiveness rule).
|
|
805
|
+
# - "double-waiver: B" -- not restated *content*, it's the syntax convention
|
|
806
|
+
# every consumer is instructed to demonstrate; flagging it would punish
|
|
807
|
+
# agents/skills for correctly teaching the waiver marker, and "paraphrase
|
|
808
|
+
# the quote" makes no sense for a literal required syntax string.
|
|
809
|
+
|
|
810
|
+
#: Directories (plugin-root-relative) a citing consumer could plausibly live
|
|
811
|
+
#: in — matches where #2124/#2125's own citations actually landed
|
|
812
|
+
#: (knowledge/, agents/, skills/). Anything outside these (plans/, docs/,
|
|
813
|
+
#: tests/) is out of scope by design: this check targets consumers of the
|
|
814
|
+
#: rule, not every place its literal words could theoretically appear.
|
|
815
|
+
_NORMATIVE_CONTENT_SCAN_DIRS = ("knowledge", "agents", "skills")
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def check_normative_content_single_sourced(changed_files=None) -> list[dict]:
|
|
819
|
+
"""`internal-collaborator-doubling.md` (#2124) is the single normative
|
|
820
|
+
source for the internal-collaborator doubling rule; every consumer must
|
|
821
|
+
cite it by path rather than restate its content (#2126).
|
|
822
|
+
|
|
823
|
+
Corpus-wide by design, like `check_transcript_parsing_confined_to_session_log`:
|
|
824
|
+
a restatement is a standing gap whether or not this changeset touched it.
|
|
825
|
+
|
|
826
|
+
A match is a match regardless of attribution — quoting a fragment
|
|
827
|
+
verbatim next to a citation to the home file still creates a second copy
|
|
828
|
+
that can drift from the original once either side is edited. The fix for
|
|
829
|
+
a consumer that needs to reference specific wording is to paraphrase or
|
|
830
|
+
drop the verbatim quote, not to attribute it.
|
|
831
|
+
"""
|
|
832
|
+
home_path = _REPO_ROOT / _NORMATIVE_CONTENT_HOME_FILE
|
|
833
|
+
findings = []
|
|
834
|
+
for dirname in _NORMATIVE_CONTENT_SCAN_DIRS:
|
|
835
|
+
scan_root = _PLUGIN_ROOT / dirname
|
|
836
|
+
if not scan_root.is_dir():
|
|
837
|
+
continue
|
|
838
|
+
for path in sorted(scan_root.rglob("*.md")):
|
|
839
|
+
if path.resolve() == home_path.resolve():
|
|
840
|
+
continue
|
|
841
|
+
text = _read_text(path)
|
|
842
|
+
if not text:
|
|
843
|
+
continue
|
|
844
|
+
rel = _repo_relative(path)
|
|
845
|
+
for fragment in _NORMATIVE_CONTENT_FRAGMENTS:
|
|
846
|
+
if fragment in text:
|
|
847
|
+
findings.append(
|
|
848
|
+
{
|
|
849
|
+
"invariant": "normative-content-single-sourced",
|
|
850
|
+
"file": rel,
|
|
851
|
+
"message": (
|
|
852
|
+
f"{rel} restates a fragment of "
|
|
853
|
+
f"{_NORMATIVE_CONTENT_HOME_FILE} verbatim "
|
|
854
|
+
f"({fragment!r}) instead of citing it by path. "
|
|
855
|
+
"This creates a second copy that can silently "
|
|
856
|
+
"drift from the normative source — replace the "
|
|
857
|
+
"restatement with a pointer."
|
|
858
|
+
),
|
|
859
|
+
}
|
|
860
|
+
)
|
|
861
|
+
return findings
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
def check_ledger_filename_single_sourced(changed_files=None) -> list[dict]:
|
|
865
|
+
"""`boundary-events.jsonl`'s filename must stay single-sourced from
|
|
866
|
+
`hooks/lib/boundary_events.LOG_NAME` — the module that actually writes
|
|
867
|
+
the ledger and therefore owns its name.
|
|
868
|
+
|
|
869
|
+
Backstop review against #2166 + #2171 found this filename independently
|
|
870
|
+
hand-rolled in three homes (`boundary_events._LOG_NAME`,
|
|
871
|
+
`review_dispatch_ledger.LEDGER_STREAM`, and
|
|
872
|
+
`boundary_events_write_guard.py`'s own import), reported by 4 of 8
|
|
873
|
+
reviewers (arch, domain, naming, structure) — clearing this repo's own
|
|
874
|
+
ratchet rule ("a mechanical finding reported twice becomes a check")
|
|
875
|
+
decisively. `review_dispatch_ledger.LEDGER_STREAM` is now an alias of
|
|
876
|
+
`boundary_events.LOG_NAME`, not a fresh literal — this check asserts
|
|
877
|
+
that stays an *identity*, not just an equal value, so a future edit
|
|
878
|
+
can't silently reintroduce a fourth independent copy with a green test
|
|
879
|
+
suite.
|
|
880
|
+
|
|
881
|
+
Corpus-wide by design: this is a standing structural invariant, not a
|
|
882
|
+
changeset-scoped one.
|
|
883
|
+
"""
|
|
884
|
+
if review_dispatch_ledger.LEDGER_STREAM is not boundary_events.LOG_NAME:
|
|
885
|
+
return [
|
|
886
|
+
{
|
|
887
|
+
"invariant": "ledger-filename-single-sourced",
|
|
888
|
+
"file": "plugins/dev-team/hooks/lib/review_dispatch_ledger.py",
|
|
889
|
+
"message": (
|
|
890
|
+
"review_dispatch_ledger.LEDGER_STREAM must remain the "
|
|
891
|
+
"identical object as boundary_events.LOG_NAME (an alias, "
|
|
892
|
+
"not a fresh literal) — boundary_events.py is the module "
|
|
893
|
+
"that actually writes the boundary-events ledger and "
|
|
894
|
+
"owns its filename."
|
|
895
|
+
),
|
|
896
|
+
}
|
|
897
|
+
]
|
|
898
|
+
return []
|
|
899
|
+
|
|
900
|
+
|
|
901
|
+
# --- #2177: the context-ceiling hook and its report script are gone ----------
|
|
902
|
+
#
|
|
903
|
+
# ADR 0043 replaced the forced-handoff hook with harness autocompact and
|
|
904
|
+
# deleted the hook, its report script and the docs around them. A stale
|
|
905
|
+
# reference to either name is a dangling pointer to a file that no longer
|
|
906
|
+
# exists. History (the changelog, superseded ADRs) legitimately keeps the
|
|
907
|
+
# names; tests that assert the files stay gone must spell them out.
|
|
908
|
+
|
|
909
|
+
_CEILING_REF_RE = re.compile(
|
|
910
|
+
rb"context_ceiling_(?:guard|report)|context-ceiling-validation|"
|
|
911
|
+
rb"DEV_TEAM_CONTEXT_ABS_CEILING"
|
|
912
|
+
)
|
|
913
|
+
_CEILING_REF_EXEMPT_PREFIXES = ("docs/adr/",)
|
|
914
|
+
_CEILING_REF_EXEMPT_FILES = frozenset(
|
|
915
|
+
{
|
|
916
|
+
"plugins/dev-team/CHANGELOG.md",
|
|
917
|
+
# Tests that pin the removal, so they name what must stay absent.
|
|
918
|
+
"tests/hooks/test_autocompact_hook_registration.py",
|
|
919
|
+
"tests/scripts/test_no_ceiling_event_consumers.py",
|
|
920
|
+
"tests/repo/test_no_live_ceiling_refs.py",
|
|
921
|
+
"tests/skills/test_handoff_manual_only.py",
|
|
922
|
+
}
|
|
923
|
+
)
|
|
924
|
+
_CEILING_REF_MAX_BYTES = 2_000_000
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _tracked_files() -> list[str] | None:
|
|
928
|
+
"""Repo-relative tracked paths via `git ls-files`, or None when git is
|
|
929
|
+
unavailable or fails (the caller then skips rather than walking the tree)."""
|
|
930
|
+
try:
|
|
931
|
+
out = subprocess.run(
|
|
932
|
+
["git", "ls-files", "-z"],
|
|
933
|
+
cwd=_REPO_ROOT,
|
|
934
|
+
capture_output=True,
|
|
935
|
+
check=True,
|
|
936
|
+
timeout=30,
|
|
937
|
+
).stdout.decode("utf-8", "replace")
|
|
938
|
+
except (OSError, subprocess.SubprocessError):
|
|
939
|
+
return None
|
|
940
|
+
return [p for p in out.split("\0") if p]
|
|
941
|
+
|
|
942
|
+
|
|
943
|
+
def _is_marketplace_checkout() -> bool:
|
|
944
|
+
"""True only in this repo's own checkout. The check ships in the plugin,
|
|
945
|
+
so downstream it runs from the plugin cache, where `_REPO_ROOT` is not a
|
|
946
|
+
repo this invariant governs."""
|
|
947
|
+
return (_REPO_ROOT / ".claude-plugin" / "marketplace.json").is_file()
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def check_no_live_ceiling_refs(changed_files=None) -> list[dict]:
|
|
951
|
+
"""No tracked file outside history may reference the removed context-
|
|
952
|
+
ceiling hook or report script by name (#2177, ADR 0043).
|
|
953
|
+
|
|
954
|
+
Corpus-wide by design: a dangling pointer is wrong whether or not this
|
|
955
|
+
changeset touched it, so `changed_files` is ignored. Returns [] outside
|
|
956
|
+
this repo's own checkout and when git cannot list files.
|
|
957
|
+
"""
|
|
958
|
+
if not _is_marketplace_checkout():
|
|
959
|
+
return []
|
|
960
|
+
tracked = _tracked_files()
|
|
961
|
+
if tracked is None:
|
|
962
|
+
return []
|
|
963
|
+
findings = []
|
|
964
|
+
self_rel = _repo_relative(Path(__file__).resolve())
|
|
965
|
+
for rel in sorted(tracked):
|
|
966
|
+
if (
|
|
967
|
+
rel == self_rel
|
|
968
|
+
or rel in _CEILING_REF_EXEMPT_FILES
|
|
969
|
+
or rel.startswith(_CEILING_REF_EXEMPT_PREFIXES)
|
|
970
|
+
):
|
|
971
|
+
continue
|
|
972
|
+
path = _REPO_ROOT / rel
|
|
973
|
+
try:
|
|
974
|
+
if not path.is_file() or path.stat().st_size > _CEILING_REF_MAX_BYTES:
|
|
975
|
+
continue
|
|
976
|
+
data = path.read_bytes() # bytes: tracked binaries are not UTF-8
|
|
977
|
+
except OSError:
|
|
978
|
+
continue
|
|
979
|
+
if _CEILING_REF_RE.search(data):
|
|
980
|
+
findings.append(
|
|
981
|
+
{
|
|
982
|
+
"invariant": "no-live-ceiling-refs",
|
|
983
|
+
"file": rel,
|
|
984
|
+
"message": (
|
|
985
|
+
f"{rel} references the removed context-ceiling hook, "
|
|
986
|
+
"report script, validation doc or env var. All were "
|
|
987
|
+
"removed by #2177 (ADR 0043); point at "
|
|
988
|
+
"docs/adr/0043-replace-the-context-ceiling-guard-with-"
|
|
989
|
+
"harness-autocompact.md or drop the reference."
|
|
990
|
+
),
|
|
991
|
+
}
|
|
992
|
+
)
|
|
993
|
+
return findings
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
# Registered checks. Each entry takes an optional `changed_files` list and
|
|
997
|
+
# returns findings. See the module docstring for why that argument exists.
|
|
998
|
+
CHECKS = [
|
|
999
|
+
check_skill_scripts_documented,
|
|
1000
|
+
check_eval_calibration_blocks,
|
|
1001
|
+
check_must_not_mention_terms_appear_in_fixture,
|
|
1002
|
+
check_scope_glob_matches_skip_prose,
|
|
1003
|
+
check_contract_failure_shapes_documented,
|
|
1004
|
+
check_transcript_parsing_confined_to_session_log,
|
|
1005
|
+
check_churn_report_window_key_safe_access,
|
|
1006
|
+
check_normative_content_single_sourced,
|
|
1007
|
+
check_ledger_filename_single_sourced,
|
|
1008
|
+
check_no_live_ceiling_refs,
|
|
1009
|
+
]
|
|
1010
|
+
|
|
1011
|
+
|
|
1012
|
+
def run_all(changed_files=None) -> list[dict]:
|
|
1013
|
+
findings = []
|
|
1014
|
+
for check in CHECKS:
|
|
1015
|
+
findings.extend(check(changed_files))
|
|
1016
|
+
return findings
|
|
1017
|
+
|
|
1018
|
+
|
|
1019
|
+
def _sweep_all() -> list[dict]:
|
|
1020
|
+
"""Backlog triage (`--all`): re-run the changeset-scoped checks against
|
|
1021
|
+
every file they could apply to, so a maintainer can see the pre-existing
|
|
1022
|
+
findings each convention chose not to retrofit. Never used by the
|
|
1023
|
+
`/code-review` pre-pass."""
|
|
1024
|
+
every_expected = [
|
|
1025
|
+
_repo_relative(p) for p in sorted((_REPO_ROOT / "evals" / "expected").glob("*.json"))
|
|
1026
|
+
]
|
|
1027
|
+
every_agent = [
|
|
1028
|
+
_repo_relative(p) for p in find_review_agent_files(default_agents_dir())
|
|
1029
|
+
]
|
|
1030
|
+
return run_all(every_expected + every_agent)
|
|
1031
|
+
|
|
1032
|
+
|
|
1033
|
+
def main(argv=None) -> int:
|
|
1034
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
1035
|
+
parser.add_argument(
|
|
1036
|
+
"--files",
|
|
1037
|
+
nargs="*",
|
|
1038
|
+
default=None,
|
|
1039
|
+
help=(
|
|
1040
|
+
"Repo-relative changed files for this review. Checks scoped to new "
|
|
1041
|
+
"content (e.g. the _calibration convention, which evals/README.md "
|
|
1042
|
+
"forbids retrofitting) only fire for these paths."
|
|
1043
|
+
),
|
|
1044
|
+
)
|
|
1045
|
+
parser.add_argument(
|
|
1046
|
+
"--all",
|
|
1047
|
+
action="store_true",
|
|
1048
|
+
dest="sweep_all",
|
|
1049
|
+
help=(
|
|
1050
|
+
"Deliberate backlog triage: run every changeset-scoped check across "
|
|
1051
|
+
"the whole corpus. Not for the /code-review pre-pass — it surfaces "
|
|
1052
|
+
"pre-existing findings the convention that introduced each check "
|
|
1053
|
+
"explicitly does not require retrofitting."
|
|
1054
|
+
),
|
|
1055
|
+
)
|
|
1056
|
+
args = parser.parse_args(argv)
|
|
1057
|
+
if args.sweep_all:
|
|
1058
|
+
findings = _sweep_all()
|
|
1059
|
+
else:
|
|
1060
|
+
findings = run_all(args.files)
|
|
1061
|
+
print(json.dumps({"findings": findings}, sort_keys=True))
|
|
1062
|
+
return 0
|
|
1063
|
+
|
|
1064
|
+
|
|
1065
|
+
if __name__ == "__main__":
|
|
1066
|
+
raise SystemExit(main(sys.argv[1:]))
|