pi-dev-team 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/PORTING.md +134 -0
- package/README.md +207 -0
- package/UPSTREAM.json +64 -0
- package/agents/Explore.md +15 -0
- package/agents/a11y-review.md +118 -0
- package/agents/adr-author.md +70 -0
- package/agents/ai-provenance-review.md +120 -0
- package/agents/angular-reactivity-review.md +95 -0
- package/agents/arch-review.md +135 -0
- package/agents/architect.md +78 -0
- package/agents/autoship-batch-proposer.md +69 -0
- package/agents/claude-setup-review.md +136 -0
- package/agents/codebase-recon.md +184 -0
- package/agents/component-architecture-review.md +119 -0
- package/agents/concurrency-review.md +109 -0
- package/agents/correctness-review.md +290 -0
- package/agents/data-flow-tracer.md +120 -0
- package/agents/doc-review.md +165 -0
- package/agents/domain-review.md +136 -0
- package/agents/general-purpose.md +10 -0
- package/agents/gherkin-quality-critic.md +113 -0
- package/agents/js-fp-review.md +114 -0
- package/agents/mutation-kill.md +684 -0
- package/agents/naming-review.md +142 -0
- package/agents/orchestrator.md +339 -0
- package/agents/performance-review.md +105 -0
- package/agents/plan-review-acceptance.md +115 -0
- package/agents/plan-review-design.md +90 -0
- package/agents/plan-review-parallelization.md +84 -0
- package/agents/plan-review-strategic.md +96 -0
- package/agents/plan-review-ux.md +110 -0
- package/agents/platform-engineer.md +64 -0
- package/agents/product-manager.md +68 -0
- package/agents/progress-guardian.md +79 -0
- package/agents/qa-engineer.md +289 -0
- package/agents/quality-reviewer.md +132 -0
- package/agents/react-reactivity-review.md +102 -0
- package/agents/refactor-opportunity-review.md +128 -0
- package/agents/security-engineer.md +60 -0
- package/agents/security-review.md +218 -0
- package/agents/session-analysis.md +95 -0
- package/agents/software-engineer.md +105 -0
- package/agents/spec-compliance-review.md +100 -0
- package/agents/spec-reviewer.md +114 -0
- package/agents/structure-review.md +146 -0
- package/agents/tech-writer.md +84 -0
- package/agents/test-review.md +246 -0
- package/agents/test-smell-review.md +188 -0
- package/agents/token-efficiency-review.md +139 -0
- package/agents/ui-ux-designer.md +54 -0
- package/agents/vue-reactivity-review.md +95 -0
- package/bin/__pycache__/claudecpython-314.pyc +0 -0
- package/bin/claude +258 -0
- package/docs/upstream/.pages +1 -0
- package/docs/upstream/CHANGELOG.md +2586 -0
- package/docs/upstream/README.md +155 -0
- package/docs/upstream/agent-architecture.md +214 -0
- package/docs/upstream/agent_info.md +187 -0
- package/docs/upstream/artifact-migration.md +124 -0
- package/docs/upstream/code-intelligence-nudge.md +149 -0
- package/docs/upstream/code-review-process.md +294 -0
- package/docs/upstream/concurrent-use.md +73 -0
- package/docs/upstream/context-management.md +111 -0
- package/docs/upstream/developer-notes.md +280 -0
- package/docs/upstream/diagrams/architecture-overview.svg +101 -0
- package/docs/upstream/diagrams/review-dispatch.svg +139 -0
- package/docs/upstream/diagrams/team-agents.svg +128 -0
- package/docs/upstream/diagrams/test-improve-flow.svg +166 -0
- package/docs/upstream/diagrams/workflow-linear.svg +66 -0
- package/docs/upstream/diagrams/workflow-three-phase.svg +200 -0
- package/docs/upstream/eval-maintenance.md +95 -0
- package/docs/upstream/eval-running-guide.md +147 -0
- package/docs/upstream/eval-system.md +291 -0
- package/docs/upstream/session-review-oss-complements.md +75 -0
- package/docs/upstream/session-review.md +212 -0
- package/docs/upstream/skills.md +188 -0
- package/docs/upstream/team-structure.md +21 -0
- package/docs/upstream/telemetry-ci-access.md +129 -0
- package/docs/upstream/telemetry-repo-security.md +120 -0
- package/docs/upstream/test-evaluation.md +277 -0
- package/docs/upstream/test-improve.md +154 -0
- package/docs/upstream/triage-workflow.md +282 -0
- package/docs/upstream/workflows.md +289 -0
- package/extensions/dev-team/index.ts +539 -0
- package/extensions/dev-team/lib/agents.ts +272 -0
- package/extensions/dev-team/lib/ai-credits.ts +92 -0
- package/extensions/dev-team/lib/autocompact.ts +81 -0
- package/extensions/dev-team/lib/child-run.ts +102 -0
- package/extensions/dev-team/lib/config.ts +236 -0
- package/extensions/dev-team/lib/gh-command.ts +103 -0
- package/extensions/dev-team/lib/github-style.ts +307 -0
- package/extensions/dev-team/lib/hooks.ts +350 -0
- package/extensions/dev-team/lib/metrics.ts +115 -0
- package/extensions/dev-team/lib/safe-read.ts +49 -0
- package/extensions/dev-team/lib/session-files.ts +57 -0
- package/extensions/dev-team/lib/session-spend.ts +123 -0
- package/extensions/dev-team/lib/shell-scan.ts +205 -0
- package/extensions/dev-team/lib/skills.ts +213 -0
- package/extensions/dev-team/lib/subagent-render.ts +245 -0
- package/extensions/dev-team/lib/subagent-types.ts +164 -0
- package/extensions/dev-team/lib/subagent.ts +596 -0
- package/extensions/dev-team/lib/terminal-text.ts +54 -0
- package/extensions/dev-team/lib/tools-misc.ts +152 -0
- package/extensions/dev-team/lib/transcript.ts +110 -0
- package/extensions/dev-team/lib/trust.ts +52 -0
- package/extensions/dev-team/lib/usage-breakdown.ts +176 -0
- package/extensions/dev-team/lib/usage-chart.ts +153 -0
- package/extensions/dev-team/lib/usage-command.ts +107 -0
- package/extensions/dev-team/lib/usage-history.ts +203 -0
- package/extensions/dev-team/lib/usage-render.ts +225 -0
- package/extensions/dev-team/lib/usage-split-bar.ts +127 -0
- package/extensions/dev-team/lib/usage-state.ts +116 -0
- package/extensions/dev-team/lib/usage-text.ts +159 -0
- package/extensions/dev-team/lib/usage-view.ts +109 -0
- package/hooks/__pycache__/refactor_test_freeze_guard.cpython-314.pyc +0 -0
- package/hooks/agent_dispatch_ledger.py +190 -0
- package/hooks/autocompact_setup_nudge.py +99 -0
- package/hooks/bash_retry_guard.py +228 -0
- package/hooks/boundary_events_write_guard.py +352 -0
- package/hooks/code_intelligence_nudge.py +293 -0
- package/hooks/code_intelligence_turn_mark.py +317 -0
- package/hooks/codegraph_bootstrap.py +139 -0
- package/hooks/contract_version_guard.py +362 -0
- package/hooks/cost_meter.py +106 -0
- package/hooks/destructive-commands.json +62 -0
- package/hooks/destructive_guard.py +477 -0
- package/hooks/eval_compliance_check.py +440 -0
- package/hooks/guards.json +17 -0
- package/hooks/hooks.json +323 -0
- package/hooks/internal_double_gate.py +296 -0
- package/hooks/js_fp_review.py +212 -0
- package/hooks/knowledge_index.py +119 -0
- package/hooks/lib/__pycache__/artifact_paths.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/atomic_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/autocompact_config.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/boundary_events.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/doc_classification.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/gh_pr_create_detect.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/git_safe_diff.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/instrument_log.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/metrics_query.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/plugin_version.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/pre_commit_doc_classifier.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_agent_registry.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_corroboration.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_gate_hash.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/review_verdicts.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stdin_json.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/stryker_invocation.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/telemetry_consent.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/test_file_classify.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/token_efficiency_limits.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/verify_guard_state.cpython-314.pyc +0 -0
- package/hooks/lib/__pycache__/xunit_v3_operator_gate.cpython-314.pyc +0 -0
- package/hooks/lib/agent_skill_hints.py +74 -0
- package/hooks/lib/artifact_paths.py +263 -0
- package/hooks/lib/atomic_state.py +557 -0
- package/hooks/lib/autocompact_config.py +103 -0
- package/hooks/lib/autoship_log.py +106 -0
- package/hooks/lib/banned_scripts_policy.py +51 -0
- package/hooks/lib/boundary_events.py +436 -0
- package/hooks/lib/build_knowledge_index.py +504 -0
- package/hooks/lib/build_skills_index.py +361 -0
- package/hooks/lib/build_state.py +116 -0
- package/hooks/lib/classify_ship_outcome.py +126 -0
- package/hooks/lib/config_changelog_schema.py +115 -0
- package/hooks/lib/cost_meter.py +955 -0
- package/hooks/lib/doc_classification.py +116 -0
- package/hooks/lib/gh_pr_create_detect.py +136 -0
- package/hooks/lib/git_safe_diff.py +123 -0
- package/hooks/lib/instrument_log.py +66 -0
- package/hooks/lib/iteration_journal_gate.py +197 -0
- package/hooks/lib/knowledge_index_paths.py +88 -0
- package/hooks/lib/mcp_json_repowise.py +177 -0
- package/hooks/lib/metrics_query.py +202 -0
- package/hooks/lib/minimal_yaml.py +434 -0
- package/hooks/lib/plugin_version.py +142 -0
- package/hooks/lib/pre_commit_detect.py +537 -0
- package/hooks/lib/pre_commit_doc_classifier.py +126 -0
- package/hooks/lib/pricing.py +118 -0
- package/hooks/lib/report_pdf.py +371 -0
- package/hooks/lib/review_agent_registry.py +142 -0
- package/hooks/lib/review_dispatch_ledger.py +101 -0
- package/hooks/lib/review_gate_corroboration.py +521 -0
- package/hooks/lib/review_gate_hash.py +252 -0
- package/hooks/lib/review_gate_normalized_hash.py +1115 -0
- package/hooks/lib/review_verdicts.py +301 -0
- package/hooks/lib/run_report.py +160 -0
- package/hooks/lib/skill_categories.yaml +125 -0
- package/hooks/lib/stdin_json.py +57 -0
- package/hooks/lib/stryker_invocation.py +102 -0
- package/hooks/lib/telemetry_consent.py +41 -0
- package/hooks/lib/telemetry_report.py +108 -0
- package/hooks/lib/test_file_classify.py +160 -0
- package/hooks/lib/token_efficiency_limits.py +51 -0
- package/hooks/lib/turn_identity.py +77 -0
- package/hooks/lib/verify_guard_state.py +110 -0
- package/hooks/lib/workflow_state.py +206 -0
- package/hooks/lib/xunit_v3_operator_gate.py +596 -0
- package/hooks/mcp_json_repowise_nudge.py +74 -0
- package/hooks/mutation_adapters/__init__.py +7 -0
- package/hooks/mutation_adapters/__pycache__/__init__.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/lib.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/mutmut.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/pitest.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/__pycache__/stryker_net.cpython-314.pyc +0 -0
- package/hooks/mutation_adapters/lib.py +478 -0
- package/hooks/mutation_adapters/mutmut.py +188 -0
- package/hooks/mutation_adapters/pitest.py +266 -0
- package/hooks/mutation_adapters/stryker.py +157 -0
- package/hooks/mutation_adapters/stryker_net.py +264 -0
- package/hooks/mutation_gate.py +193 -0
- package/hooks/mutation_testing_smoke_gate.py +371 -0
- package/hooks/pending_review_notify.py +121 -0
- package/hooks/phase_marker.py +138 -0
- package/hooks/post_compact_state_reinject.py +180 -0
- package/hooks/post_format.py +115 -0
- package/hooks/pre_commit_knowledge_index.py +128 -0
- package/hooks/pre_commit_review.py +66 -0
- package/hooks/pre_pr_review.py +694 -0
- package/hooks/pre_tool_guard.py +405 -0
- package/hooks/py.sh +73 -0
- package/hooks/refactor-bash-write-patterns.json +29 -0
- package/hooks/refactor_test_bash_guard.py +253 -0
- package/hooks/refactor_test_freeze_guard.py +139 -0
- package/hooks/refactor_test_revert_guard.py +186 -0
- package/hooks/repo_review_nudge.py +287 -0
- package/hooks/review_verdict_recorder.py +464 -0
- package/hooks/scan_bash_command_for_banned_scripts.py +428 -0
- package/hooks/scan_worktree_for_banned_scripts.py +238 -0
- package/hooks/session_learning_trigger.py +248 -0
- package/hooks/skills_index.py +126 -0
- package/hooks/stryker_xunit_shim_guard.py +571 -0
- package/hooks/subagent_completion_guard.py +309 -0
- package/hooks/subagent_skill_context.py +139 -0
- package/hooks/task_completion_metrics.py +216 -0
- package/hooks/tdd_guard.py +229 -0
- package/hooks/telemetry.py +341 -0
- package/hooks/token_efficiency_review.py +194 -0
- package/hooks/verify_guard.py +183 -0
- package/hooks/verify_guard_edit_marker.py +73 -0
- package/hooks/version_check.py +173 -0
- package/knowledge/accepted-risks-schema.md +98 -0
- package/knowledge/adr-decision-criteria.md +64 -0
- package/knowledge/adversarial-review-protocol.md +139 -0
- package/knowledge/agent-registry.md +228 -0
- package/knowledge/agent-review-methodology.md +80 -0
- package/knowledge/ai-friendly-repo-guidelines.md +67 -0
- package/knowledge/architecture-assessment.md +96 -0
- package/knowledge/artifact-lifecycle.md +57 -0
- package/knowledge/cd-maturity-model.md +82 -0
- package/knowledge/cd-test-architecture.md +190 -0
- package/knowledge/ci-cd-file-scope.md +24 -0
- package/knowledge/codegraph-vs-graphify.md +192 -0
- package/knowledge/component-test-patterns.md +139 -0
- package/knowledge/database-change-management.md +80 -0
- package/knowledge/database-test-patterns.md +79 -0
- package/knowledge/decision-defaults.md +88 -0
- package/knowledge/dependency-breaking-techniques.md +116 -0
- package/knowledge/deployment-pipeline.md +86 -0
- package/knowledge/design-smells.md +122 -0
- package/knowledge/directory-enumeration.md +38 -0
- package/knowledge/domain-modeling.md +123 -0
- package/knowledge/evidence-bundle.md +90 -0
- package/knowledge/exploratory-testing-field-guide.md +122 -0
- package/knowledge/failure-routing.md +28 -0
- package/knowledge/fixture-construction.md +56 -0
- package/knowledge/frontend-component-architecture.md +139 -0
- package/knowledge/gherkin-quality-review-dispatch.md +135 -0
- package/knowledge/index.json +6766 -0
- package/knowledge/internal-collaborator-doubling.md +101 -0
- package/knowledge/legacy-test-strategy.md +71 -0
- package/knowledge/long-run-waiting.md +66 -0
- package/knowledge/microservice-testing.md +71 -0
- package/knowledge/model-pricing.json +23 -0
- package/knowledge/mutation-score-formulas.md +60 -0
- package/knowledge/object-calisthenics.md +147 -0
- package/knowledge/oracle-provenance.md +94 -0
- package/knowledge/orchestrator-script-implementation.md +185 -0
- package/knowledge/owasp-detection.md +148 -0
- package/knowledge/plan-review-rubric.md +56 -0
- package/knowledge/proxy-connectivity.md +62 -0
- package/knowledge/reactive-effect-patterns.md +73 -0
- package/knowledge/recon-inventory-excludes.txt +32 -0
- package/knowledge/references/bdd-value-guide.md +61 -0
- package/knowledge/references/csharp-http-client-testing.md +264 -0
- package/knowledge/release-strategies.md +74 -0
- package/knowledge/report-output-location.md +117 -0
- package/knowledge/report-pdf-integration.md +63 -0
- package/knowledge/report-print.css +129 -0
- package/knowledge/report-template.md +114 -0
- package/knowledge/report-to-pdf.md +69 -0
- package/knowledge/request-processing-flow.md +63 -0
- package/knowledge/result-verification.md +52 -0
- package/knowledge/review-agent-output-contract.md +121 -0
- package/knowledge/review-lens-classification.md +113 -0
- package/knowledge/review-rubric.md +62 -0
- package/knowledge/review-template.md +104 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/negative.js +1 -0
- package/knowledge/rule-fixtures/A02.insecure-random-js/positive.js +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/negative.py +1 -0
- package/knowledge/rule-fixtures/A02.weak-hashing-md5/positive.py +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.command-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.sql-injection/positive.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/negative.js +1 -0
- package/knowledge/rule-fixtures/A03.xss-innerhtml/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.cors-wildcard/positive.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/negative.js +1 -0
- package/knowledge/rule-fixtures/A05.default-credentials/positive.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/negative.js +1 -0
- package/knowledge/rule-fixtures/A07.jwt-alg-none/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/negative.cs +1 -0
- package/knowledge/rule-fixtures/A08.binary-formatter/positive.cs +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/negative.js +1 -0
- package/knowledge/rule-fixtures/A08.js-eval/positive.js +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/negative.java +1 -0
- package/knowledge/rule-fixtures/A08.object-input-stream/positive.java +1 -0
- package/knowledge/schemas/disposition-register-v1.json +65 -0
- package/knowledge/schemas/recon-envelope-v1.json +198 -0
- package/knowledge/schemas/unified-finding-v1.json +72 -0
- package/knowledge/security-primitives-contract.md +301 -0
- package/knowledge/security-review-rule-map.yaml +107 -0
- package/knowledge/skills-registry.md +72 -0
- package/knowledge/task-size-classifier.md +103 -0
- package/knowledge/telemetry-schema.md +881 -0
- package/knowledge/test-automation-maturity.md +56 -0
- package/knowledge/test-automation-principles.md +71 -0
- package/knowledge/test-cadence-tradeoffs.md +68 -0
- package/knowledge/test-doubles.md +105 -0
- package/knowledge/test-file-indicators.md +22 -0
- package/knowledge/test-layer-gates.md +35 -0
- package/knowledge/test-matrix-examples/django-batch.md +24 -0
- package/knowledge/test-matrix-examples/dotnet-grpc-fronting-api.md +90 -0
- package/knowledge/test-matrix-examples/dotnet-http-consumer.md +131 -0
- package/knowledge/test-matrix-examples/react-node-spa.md +24 -0
- package/knowledge/test-matrix-examples/spring-boot-service.md +25 -0
- package/knowledge/test-matrix-examples/ssr-htmx.md +24 -0
- package/knowledge/test-organization.md +70 -0
- package/knowledge/test-pyramid.md +84 -0
- package/knowledge/test-refactoring.md +67 -0
- package/knowledge/test-review-division-of-labor.md +85 -0
- package/knowledge/test-smells.md +80 -0
- package/knowledge/test-stack-profiles/bdd-frameworks.md +235 -0
- package/knowledge/test-stack-profiles/django.md +13 -0
- package/knowledge/test-stack-profiles/dotnet.md +18 -0
- package/knowledge/test-stack-profiles/go.md +16 -0
- package/knowledge/test-stack-profiles/node.md +16 -0
- package/knowledge/test-stack-profiles/react.md +12 -0
- package/knowledge/test-stack-profiles/spring-boot.md +16 -0
- package/knowledge/test-stack-profiles/ssr-htmx.md +14 -0
- package/knowledge/test-stack-profiles/vue.md +12 -0
- package/knowledge/test-strategy.md +70 -0
- package/knowledge/testability-patterns.md +240 -0
- package/knowledge/testing-quadrants.md +44 -0
- package/knowledge/testing-techniques/approval.md +15 -0
- package/knowledge/testing-techniques/chaos.md +17 -0
- package/knowledge/testing-techniques/fuzz.md +15 -0
- package/knowledge/testing-techniques/property-based.md +15 -0
- package/knowledge/testing-techniques/schema-validation.md +15 -0
- package/knowledge/testing-techniques/screenshot.md +15 -0
- package/knowledge/three-phase-workflow.md +198 -0
- package/knowledge/value-patterns.md +55 -0
- package/knowledge/verification-mode.md +116 -0
- package/knowledge/virtual-service-libraries.md +75 -0
- package/knowledge/wave-consolidation-guidance.md +21 -0
- package/overrides/agents/Explore.md +15 -0
- package/overrides/agents/general-purpose.md +10 -0
- package/overrides/notes/autoship.md +6 -0
- package/overrides/notes/issues-from-assessment.md +3 -0
- package/overrides/notes/issues-from-plan.md +3 -0
- package/overrides/notes/mutation-night-watch.md +3 -0
- package/overrides/notes/mutation-testing.md +3 -0
- package/overrides/notes/pr.md +7 -0
- package/overrides/notes/project-init.md +6 -0
- package/overrides/notes/setup.md +13 -0
- package/overrides/notes/specs.md +3 -0
- package/overrides/skills/headless-run/SKILL.md +45 -0
- package/overrides/skills/upgrade/SKILL.md +30 -0
- package/overrides/skills/version/SKILL.md +25 -0
- package/package.json +36 -0
- package/scripts/authoring_digest.py +93 -0
- package/scripts/autoship_discover.py +121 -0
- package/scripts/autoship_group.py +409 -0
- package/scripts/autoship_proposals.py +494 -0
- package/scripts/autoship_queue.py +291 -0
- package/scripts/autoship_reclaim.py +495 -0
- package/scripts/build_jobs.py +108 -0
- package/scripts/build_rollback_point.py +240 -0
- package/scripts/build_slice_scope.py +157 -0
- package/scripts/build_wave.py +109 -0
- package/scripts/build_wave_reconcile.py +252 -0
- package/scripts/build_worktree_baseref.py +113 -0
- package/scripts/check_agent_scope.py +117 -0
- package/scripts/check_agent_tool_mapping.py +213 -0
- package/scripts/check_review_agent_mcp_tools.py +317 -0
- package/scripts/check_security_assessment_mcp_tools.py +165 -0
- package/scripts/checkpoint_abort.py +502 -0
- package/scripts/claude_setup_review.py +438 -0
- package/scripts/codebase_recon.py +556 -0
- package/scripts/coverage_config.py +623 -0
- package/scripts/coverage_delta_steering.py +330 -0
- package/scripts/coverage_discovery_dotnet.py +315 -0
- package/scripts/coverage_discovery_java.py +742 -0
- package/scripts/coverage_discovery_js.py +546 -0
- package/scripts/coverage_gap_ranking.py +556 -0
- package/scripts/coverage_readiness.py +455 -0
- package/scripts/coverage_report_parse.py +521 -0
- package/scripts/detect_bdd_convention.py +252 -0
- package/scripts/eval_ablation.py +376 -0
- package/scripts/gherkin_analysis_coverage_gate.py +306 -0
- package/scripts/gherkin_cross_feature_duplicate_titles_gate.py +173 -0
- package/scripts/gherkin_effectiveness_rollup.py +238 -0
- package/scripts/gherkin_failure_path_gate.py +206 -0
- package/scripts/gherkin_feature_merge.py +720 -0
- package/scripts/gherkin_stub_gate.py +163 -0
- package/scripts/gherkin_stub_merge.py +479 -0
- package/scripts/git_origin_host.py +88 -0
- package/scripts/install-java-static-analysis.py +110 -0
- package/scripts/issue_deps.py +74 -0
- package/scripts/lib/_bdd_markers.py +28 -0
- package/scripts/lib/_gherkin_text.py +93 -0
- package/scripts/lib/_vendored_tree.py +70 -0
- package/scripts/lib/autoship_state.py +397 -0
- package/scripts/lib/claude_md_guard.py +226 -0
- package/scripts/lib/deterministic_recon.py +446 -0
- package/scripts/lib/mcp_tool_grants.py +211 -0
- package/scripts/lib/plan_parse.py +386 -0
- package/scripts/lib/review_result.py +84 -0
- package/scripts/lib/review_roster.py +86 -0
- package/scripts/lib/session_log/__init__.py +34 -0
- package/scripts/lib/session_log/__pycache__/__init__.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/__pycache__/records.cpython-314.pyc +0 -0
- package/scripts/lib/session_log/classify.py +231 -0
- package/scripts/lib/session_log/corrections.py +194 -0
- package/scripts/lib/session_log/discovery.py +108 -0
- package/scripts/lib/session_log/records.py +218 -0
- package/scripts/lib/session_log/redact.py +76 -0
- package/scripts/lib/session_log/signals.py +373 -0
- package/scripts/lib/session_report_downstream.py +614 -0
- package/scripts/lib/session_report_maintainer.py +1273 -0
- package/scripts/lib/session_report_shared.py +262 -0
- package/scripts/lib/settings_hook_guard.py +157 -0
- package/scripts/lib/slug.py +33 -0
- package/scripts/lib/stub_extractors/__init__.py +82 -0
- package/scripts/lib/stub_extractors/_common.py +328 -0
- package/scripts/lib/stub_extractors/csharp.py +19 -0
- package/scripts/lib/stub_extractors/go.py +173 -0
- package/scripts/lib/stub_extractors/java.py +18 -0
- package/scripts/lib/stub_extractors/jsts.py +126 -0
- package/scripts/mutation_stack_sections.py +149 -0
- package/scripts/mutation_yield_steering.py +345 -0
- package/scripts/orchestrator.py +895 -0
- package/scripts/plan_gherkin_export.py +227 -0
- package/scripts/plan_waves.py +208 -0
- package/scripts/pr_close_keyword_lint.py +108 -0
- package/scripts/progress_guardian.py +888 -0
- package/scripts/recon_inventory.py +273 -0
- package/scripts/review_findings_log.py +93 -0
- package/scripts/run_invariants.py +124 -0
- package/scripts/select_lenses.py +640 -0
- package/scripts/session_report.py +486 -0
- package/scripts/set_autocompact_env.py +221 -0
- package/scripts/ship_resume_guard.py +135 -0
- package/scripts/ship_review_gate.py +63 -0
- package/scripts/specs_convention_marker.py +103 -0
- package/scripts/test_improve_resume.py +277 -0
- package/scripts/test_review_mechanics.py +958 -0
- package/scripts/token_efficiency_review.py +322 -0
- package/scripts/verdict_scope.py +285 -0
- package/scripts/verify_gherkin_quality_critic_isolation.py +296 -0
- package/scripts/verify_tier.py +157 -0
- package/skills/adr-tools/SKILL.md +118 -0
- package/skills/agent-readiness/SKILL.md +105 -0
- package/skills/agent-readiness/ai_friendly_analyzers.py +326 -0
- package/skills/agent-readiness/scanner.py +441 -0
- package/skills/agent-readiness/scorecard.yaml +88 -0
- package/skills/api-design/SKILL.md +115 -0
- package/skills/apply-fixes/SKILL.md +171 -0
- package/skills/apply-test-doubles/SKILL.md +321 -0
- package/skills/artifact-lifecycle/SKILL.md +127 -0
- package/skills/autoship/SKILL.md +1124 -0
- package/skills/benchmark/SKILL.md +105 -0
- package/skills/branch-workflow/SKILL.md +89 -0
- package/skills/browse/SKILL.md +184 -0
- package/skills/browser-testing/SKILL.md +62 -0
- package/skills/browser-testing/references/playwright-patterns.md +216 -0
- package/skills/build/SKILL.md +422 -0
- package/skills/build/references/static-self-heal.md +245 -0
- package/skills/careful/SKILL.md +72 -0
- package/skills/cd-test-architecture/SKILL.md +371 -0
- package/skills/ci-debugging/SKILL.md +105 -0
- package/skills/co-evolution-audit/SKILL.md +269 -0
- package/skills/code-review/SKILL.md +1015 -0
- package/skills/code-review/examples/aggregated-sample.json +56 -0
- package/skills/code-review/examples/sample-report.md +41 -0
- package/skills/code-review/output-format.md +478 -0
- package/skills/code-review/scripts/activation.py +86 -0
- package/skills/code-review/scripts/change_impact.py +357 -0
- package/skills/code-review/scripts/change_shape.py +372 -0
- package/skills/code-review/scripts/change_size.py +212 -0
- package/skills/code-review/scripts/changed_file_list.py +141 -0
- package/skills/code-review/scripts/closing_pass.py +187 -0
- package/skills/code-review/scripts/consolidate.py +277 -0
- package/skills/code-review/scripts/contract_failure_report.py +185 -0
- package/skills/code-review/scripts/dispatch_reconcile.py +66 -0
- package/skills/code-review/scripts/dispatch_waves.py +164 -0
- package/skills/code-review/scripts/finding_signature.py +446 -0
- package/skills/code-review/scripts/ledger.py +283 -0
- package/skills/code-review/scripts/partition.py +169 -0
- package/skills/code-review/scripts/render_tiered_findings.py +274 -0
- package/skills/code-review/scripts/repo_invariants.py +1066 -0
- package/skills/code-review/scripts/review_context_pack.py +306 -0
- package/skills/code-review/scripts/review_round_log.py +345 -0
- package/skills/code-review/scripts/review_value_coverage.py +297 -0
- package/skills/code-review/scripts/validate_review_output.py +467 -0
- package/skills/code-review/sliced-mode.md +205 -0
- package/skills/competitive-analysis/SKILL.md +191 -0
- package/skills/context-loading-protocol/SKILL.md +157 -0
- package/skills/continue/SKILL.md +90 -0
- package/skills/cost-report/SKILL.md +178 -0
- package/skills/coverage-baseline/SKILL.md +335 -0
- package/skills/coverage-baseline/references/multi-project-discovery.md +202 -0
- package/skills/coverage-delta/SKILL.md +181 -0
- package/skills/coverage-delta/references/mutation-gate.md +70 -0
- package/skills/design-doc/SKILL.md +95 -0
- package/skills/design-interrogation/SKILL.md +89 -0
- package/skills/design-it-twice/SKILL.md +91 -0
- package/skills/docker-image-audit/SKILL.md +108 -0
- package/skills/docker-image-audit/references/install-guide.md +64 -0
- package/skills/docker-image-audit/references/report-template.md +73 -0
- package/skills/docker-image-create/SKILL.md +185 -0
- package/skills/domain-analysis/SKILL.md +183 -0
- package/skills/domain-driven-design/SKILL.md +194 -0
- package/skills/exploratory-testing/SKILL.md +108 -0
- package/skills/explore/SKILL.md +51 -0
- package/skills/farley-score/SKILL.md +165 -0
- package/skills/feature-file-validation/SKILL.md +78 -0
- package/skills/feature-file-validation/references/validation-rules.md +115 -0
- package/skills/feedback-learning/SKILL.md +414 -0
- package/skills/fix/SKILL.md +450 -0
- package/skills/freeze/SKILL.md +68 -0
- package/skills/frontend-architecture/SKILL.md +113 -0
- package/skills/gherkin-derive/SKILL.md +630 -0
- package/skills/gherkin-public/SKILL.md +266 -0
- package/skills/governance-compliance/SKILL.md +150 -0
- package/skills/guard/SKILL.md +75 -0
- package/skills/handoff/SKILL.md +139 -0
- package/skills/handoff/references/summary-templates.md +242 -0
- package/skills/harness-audit/SKILL.md +751 -0
- package/skills/harness-audit/scripts/lesson_validate.py +386 -0
- package/skills/harness-audit/scripts/redundancy_criterion.py +188 -0
- package/skills/headless-run/SKILL.md +45 -0
- package/skills/headless-run/scripts/isolated_dispatch.py +381 -0
- package/skills/help/SKILL.md +72 -0
- package/skills/hexagonal-architecture/SKILL.md +85 -0
- package/skills/human-oversight-protocol/SKILL.md +224 -0
- package/skills/issues-from-assessment/SKILL.md +223 -0
- package/skills/issues-from-plan/SKILL.md +133 -0
- package/skills/legacy-code/SKILL.md +132 -0
- package/skills/mermaid-diagramming/SKILL.md +120 -0
- package/skills/mutation-night-watch/SKILL.md +154 -0
- package/skills/mutation-night-watch/references/scheduling.md +135 -0
- package/skills/mutation-testing/SKILL.md +396 -0
- package/skills/mutation-testing/references/languages/csharp-stryker-net.md +676 -0
- package/skills/mutation-testing/references/languages/go-go-mutesting.md +95 -0
- package/skills/mutation-testing/references/languages/java-pitest.md +77 -0
- package/skills/mutation-testing/references/languages/javascript-stryker.md +188 -0
- package/skills/mutation-testing/references/languages/python-mutmut.md +97 -0
- package/skills/mutation-testing/references/time-estimation.md +34 -0
- package/skills/mutation-testing/references/tool-detection.md +15 -0
- package/skills/mutation-testing/references/workflow-callers.md +23 -0
- package/skills/mutation-testing/scripts/__pycache__/xunit_v3_feature_detector.cpython-314.pyc +0 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_slice_runner.py +635 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_status_loop.py +525 -0
- package/skills/mutation-testing/scripts/csharp_stryker_net_wrapper.py +681 -0
- package/skills/mutation-testing/scripts/mutation_baseline_reuse.py +292 -0
- package/skills/mutation-testing/scripts/mutation_exclude_policy.py +268 -0
- package/skills/mutation-testing/scripts/mutation_feasibility_gate.py +463 -0
- package/skills/mutation-testing/scripts/mutation_kill_headless.py +331 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert.py +199 -0
- package/skills/mutation-testing/scripts/mutation_kill_insert_python.py +150 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop.py +869 -0
- package/skills/mutation-testing/scripts/mutation_kill_loop_python.py +949 -0
- package/skills/mutation-testing/scripts/mutation_kill_retry.py +592 -0
- package/skills/mutation-testing/scripts/mutation_kill_shared.py +620 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch.py +462 -0
- package/skills/mutation-testing/scripts/mutation_nightwatch_stacks.py +425 -0
- package/skills/mutation-testing/scripts/mutation_report.py +743 -0
- package/skills/mutation-testing/scripts/mutation_report_cli.py +175 -0
- package/skills/mutation-testing/scripts/mutation_safety_gate.py +69 -0
- package/skills/mutation-testing/scripts/stryker_shard_pipeline.py +847 -0
- package/skills/mutation-testing/scripts/stryker_shard_setup.py +440 -0
- package/skills/mutation-testing/scripts/stryker_timeout_retry.py +142 -0
- package/skills/mutation-testing/scripts/xunit_v3_feature_detector.py +341 -0
- package/skills/performance-benchmark/SKILL.md +174 -0
- package/skills/performance-benchmark/examples/report-format.md +43 -0
- package/skills/performance-benchmark/references/benchmark-script.md +169 -0
- package/skills/performance-metrics/SKILL.md +265 -0
- package/skills/plan/SKILL.md +199 -0
- package/skills/plan/references/gherkin-persistence.md +43 -0
- package/skills/plan/references/plan-template.md +182 -0
- package/skills/pr/SKILL.md +289 -0
- package/skills/pr/scripts/gate_retry_state.py +368 -0
- package/skills/project-init/README.md +141 -0
- package/skills/project-init/SKILL.md +1197 -0
- package/skills/project-init/evals/evals.json +200 -0
- package/skills/project-init/references/capability-tools.md +55 -0
- package/skills/project-init/references/configs.md +221 -0
- package/skills/property-based-testing/SKILL.md +121 -0
- package/skills/property-based-testing/fixtures/invariant_fixture.py +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/README.md +42 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/README.md +263 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/fast-check.js +12147 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/cjs/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/fast-check.js +12011 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/rolldown-runtime-D7D4PA-g.js +13 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/lib/types57/fast-check.d.ts +5165 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/fast-check/package.json +94 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/LICENSE +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/README.md +168 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformBigInt.js +38 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat32.js +18 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformFloat64.js +22 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/distribution/uniformInt.js +134 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/RandomGenerator-DcXj09Ch.d.ts +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformBigInt.js +37 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat32.js +17 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformFloat64.js +21 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.d.ts +15 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/distribution/uniformInt.js +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/congruential32.js +44 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/mersenne.js +90 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xoroshiro128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/generator/xorshift128plus.js +78 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/package.json +3 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/generateN.js +8 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/purify.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/esm/utils/skipN.js +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/congruential32.js +46 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/mersenne.js +92 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xoroshiro128plus.js +82 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.d.ts +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/generator/xorshift128plus.js +80 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.d.ts +16 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/JumpableRandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.d.ts +2 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/types/RandomGenerator.js +0 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/generateN.js +9 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.d.ts +12 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/purify.js +10 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.d.ts +6 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/lib/utils/skipN.js +7 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/node_modules/pure-rand/package.json +133 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package-lock.json +1179 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/package.json +14 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.js +29 -0
- package/skills/property-based-testing/fixtures/js-roundtrip/roundtrip.properties.test.js +16 -0
- package/skills/property-based-testing/fixtures/no_property_fixture.py +10 -0
- package/skills/property-based-testing/fixtures/roundtrip_fixture.py +16 -0
- package/skills/property-based-testing/references/languages/javascript.md +54 -0
- package/skills/property-based-testing/scripts/detect_and_dispatch.py +80 -0
- package/skills/property-based-testing/scripts/hypothesis_scaffold.py +276 -0
- package/skills/proxy-resilience/SKILL.md +84 -0
- package/skills/quality-gate-pipeline/SKILL.md +184 -0
- package/skills/quality-targets-converge/SKILL.md +254 -0
- package/skills/repo-review/SKILL.md +159 -0
- package/skills/report-pdf/SKILL.md +66 -0
- package/skills/review/SKILL.md +47 -0
- package/skills/review-agent/SKILL.md +152 -0
- package/skills/review-summary/SKILL.md +73 -0
- package/skills/run-report/SKILL.md +70 -0
- package/skills/semantic-duplication-scan/SKILL.md +337 -0
- package/skills/semantic-scan/SKILL.md +53 -0
- package/skills/semgrep-analyze/SKILL.md +139 -0
- package/skills/setup/SKILL.md +1122 -0
- package/skills/ship/SKILL.md +240 -0
- package/skills/source-verification/SKILL.md +210 -0
- package/skills/source-verification/scripts/claim_extractor.py +155 -0
- package/skills/specs/.size-baseline.json +4 -0
- package/skills/specs/SKILL.md +243 -0
- package/skills/specs/references/completeness-checklist.md +83 -0
- package/skills/specs/references/extraction.md +58 -0
- package/skills/specs/references/glossary.md +59 -0
- package/skills/specs/references/persistence.md +115 -0
- package/skills/specs/references/predictability-check.md +77 -0
- package/skills/static-analysis-integration/SKILL.md +235 -0
- package/skills/static-analysis-integration/adapters/_envelope.py +26 -0
- package/skills/static-analysis-integration/adapters/jscpd-adapter.py +66 -0
- package/skills/static-analysis-integration/adapters/lizard-adapter.py +81 -0
- package/skills/static-analysis-integration/adapters/mypy-adapter.py +50 -0
- package/skills/static-analysis-integration/adapters/mypy-src-layout.py +93 -0
- package/skills/static-analysis-integration/adapters/security-review-adapter.py +212 -0
- package/skills/static-analysis-integration/maintenance.md +23 -0
- package/skills/static-analysis-integration/references/language-setup.md +228 -0
- package/skills/static-analysis-integration/references/sarif-parser.md +124 -0
- package/skills/static-analysis-integration/references/security-review-adapter.md +118 -0
- package/skills/static-analysis-integration/references/tool-configs.md +617 -0
- package/skills/static-analysis-integration/rulesets/pmd-quickstart.xml +24 -0
- package/skills/stryker-xunit-v2-shim/SKILL.md +274 -0
- package/skills/stryker-xunit-v2-shim/references/shim-howto.md +256 -0
- package/skills/stryker-xunit-v2-shim/scripts/generate_shim.py +143 -0
- package/skills/systematic-debugging/SKILL.md +130 -0
- package/skills/telemetry/SKILL.md +75 -0
- package/skills/test-audit-disable/SKILL.md +129 -0
- package/skills/test-design/SKILL.md +177 -0
- package/skills/test-design/scripts/__pycache__/internal_double_detector.cpython-314.pyc +0 -0
- package/skills/test-design/scripts/internal_double_detector.py +631 -0
- package/skills/test-design-advisor/SKILL.md +166 -0
- package/skills/test-driven-development/SKILL.md +169 -0
- package/skills/test-health/SKILL.md +262 -0
- package/skills/test-improve/SKILL.md +239 -0
- package/skills/test-improve/references/phase-0-approach-contract.md +228 -0
- package/skills/test-improve/references/phase-1-analyze.md +131 -0
- package/skills/test-improve/references/phase-2-baseline.md +121 -0
- package/skills/test-improve/references/phase-3-derive-gherkin.md +53 -0
- package/skills/test-improve/references/phase-4-plan-fixes.md +34 -0
- package/skills/test-improve/references/phase-5-improve.md +215 -0
- package/skills/test-improve/references/phase-6-refactor-decision.md +45 -0
- package/skills/test-improve/references/phase-7-refactor.md +44 -0
- package/skills/test-improve/references/phase-8-validate.md +66 -0
- package/skills/test-improve/references/phase-9-close-out-prompt.md +11 -0
- package/skills/test-improve/references/phase-9-report.md +62 -0
- package/skills/test-improve/references/review-loop.md +92 -0
- package/skills/test-improve/templates/executive-summary.md +123 -0
- package/skills/threat-modeling/SKILL.md +108 -0
- package/skills/triage/SKILL.md +211 -0
- package/skills/ubiquitous-language/SKILL.md +192 -0
- package/skills/ubiquitous-language/scripts/collect_domain_signals.py +300 -0
- package/skills/unfreeze/SKILL.md +37 -0
- package/skills/upgrade/SKILL.md +31 -0
- package/skills/upgrade/scripts/check_version_drift.py +113 -0
- package/skills/upgrade/scripts/enable_autoupdate.py +149 -0
- package/skills/version/SKILL.md +25 -0
- package/sync/__pycache__/sync_upstream.cpython-314.pyc +0 -0
- package/sync/sync_upstream.py +293 -0
- package/templates/ACCEPTED-RISKS.md.tmpl +46 -0
- package/templates/agents/agent-template.md +151 -0
- package/templates/agents/angular-testing.md +66 -0
- package/templates/agents/csharp-quality.md +63 -0
- package/templates/agents/esm-enforcer.md +52 -0
- package/templates/agents/front-end-testing.md +65 -0
- package/templates/agents/go-quality.md +65 -0
- package/templates/agents/python-quality.md +62 -0
- package/templates/agents/react-testing.md +61 -0
- package/templates/agents/ts-enforcer.md +60 -0
- package/templates/agents/twelve-factor-audit.md +49 -0
- package/tools/entropy-check.py +250 -0
- package/tools/model-hash-verify.py +213 -0
|
@@ -0,0 +1,955 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Runtime cost/token meter for dispatched work (issues #102, #134).
|
|
3
|
+
|
|
4
|
+
PostToolUse hooks do NOT carry token usage in Claude Code; the canonical source
|
|
5
|
+
is the session transcript JSONL, where each assistant message records a `usage`
|
|
6
|
+
block (input/output/cache tokens). Every hook payload includes `transcript_path`,
|
|
7
|
+
so a Stop hook can hand this script the transcript to parse. This converts token
|
|
8
|
+
usage to dollars via the named instrument knowledge/model-pricing.json (#102 is
|
|
9
|
+
why that table exists) and writes an append-only metrics log.
|
|
10
|
+
|
|
11
|
+
Attribution dimensions (#102, #170, #1094)
|
|
12
|
+
------------------------------------------
|
|
13
|
+
Attribution is limited to what the harness actually records on transcript
|
|
14
|
+
records — verified empirically (#170, re-verified for #1094). Spend is
|
|
15
|
+
attributed to:
|
|
16
|
+
* the MODEL (`message.model`),
|
|
17
|
+
* the THREAD: main-loop vs subagent, from the native top-level `isSidechain`
|
|
18
|
+
flag (true on subagent/sidechain turns),
|
|
19
|
+
* the AGENT TYPE (#1094): `main` for main-loop turns; for sidechain turns the
|
|
20
|
+
subagent type (e.g. `security-review`, `general-purpose`) via two
|
|
21
|
+
harness-recorded signals, with an honest `unattributed` bucket when neither
|
|
22
|
+
is present:
|
|
23
|
+
1. the native top-level `attributionAgent` field the harness stamps on
|
|
24
|
+
sidechain records (primary — present on every usage-bearing sidechain
|
|
25
|
+
record in real transcripts), or
|
|
26
|
+
2. the Task/Agent dispatch join: a main-thread `tool_use` block named
|
|
27
|
+
`Task`/`Agent` carries `input.subagent_type` and its paired
|
|
28
|
+
`tool_result` record carries top-level `toolUseResult.agentId`; each
|
|
29
|
+
sidechain record carries the matching `agentId` (fallback).
|
|
30
|
+
* plus the session TOTAL.
|
|
31
|
+
|
|
32
|
+
Newer harness versions write sidechain turns to sibling per-subagent transcript
|
|
33
|
+
files (`<dir>/<session-id>/subagents/agent-<agentId>.jsonl`) instead of inline
|
|
34
|
+
`isSidechain` records; the meter scans those siblings so subagent spend stays
|
|
35
|
+
visible either way (#1094).
|
|
36
|
+
|
|
37
|
+
What is deliberately NOT attributed, and why (#170): per-command, per-phase, and
|
|
38
|
+
per-fix-loop-iteration attribution were attempted via `attributionSkill` /
|
|
39
|
+
`orchestrationPhase` / `fixLoopIteration` markers, but **the Claude Code harness
|
|
40
|
+
authors the transcript and exposes none of those fields** (0/312 in a real
|
|
41
|
+
transcript), and a plugin has no write-path into the transcript. Those buckets
|
|
42
|
+
were therefore always inert ("untagged"/"other"/"unattributed") and have been
|
|
43
|
+
removed rather than ship misleading empty dimensions. Re-deriving them would
|
|
44
|
+
require fragile heuristics (correlating Stop-hook timestamps with command
|
|
45
|
+
boundaries) and is out of scope. The agent-type dimension (#1094) is different
|
|
46
|
+
in kind: it reads only fields the harness demonstrably writes.
|
|
47
|
+
|
|
48
|
+
Privacy boundary
|
|
49
|
+
----------------
|
|
50
|
+
This meter persists ONLY token counts, dollar amounts, model identifiers, and
|
|
51
|
+
the thread/agent-type identifiers. It never reads or records prompt text, code,
|
|
52
|
+
file paths, or tool payloads from the transcript — only the `usage`/`model`/
|
|
53
|
+
`isSidechain`/`attributionAgent`/`agentId`/`subagent_type` fields and tool-use
|
|
54
|
+
ids needed to join them. The append-only metrics log is a metrics-only artifact
|
|
55
|
+
by construction.
|
|
56
|
+
|
|
57
|
+
Subcommands
|
|
58
|
+
-----------
|
|
59
|
+
report --transcript T [--json]
|
|
60
|
+
Parse a transcript and print tokens + cost per model and per thread
|
|
61
|
+
(main vs subagent), plus the session total. The acceptance command:
|
|
62
|
+
"after a run, print actual tokens spent."
|
|
63
|
+
|
|
64
|
+
record --transcript T --log .claude/metrics/cost-metering.jsonl
|
|
65
|
+
Append one session-summary line to the append-only metrics log
|
|
66
|
+
(follows the .claude/metrics/config-changelog.jsonl convention). Used by the
|
|
67
|
+
Stop hook. Idempotent on directory creation; never errors out loudly.
|
|
68
|
+
|
|
69
|
+
regression --log .claude/metrics/cost-metering.jsonl [--tolerance 0.5] [--window N]
|
|
70
|
+
Compare the most recent session's total cost against the rolling mean
|
|
71
|
+
of prior sessions; exit 1 if it exceeds mean * (1 + tolerance). With
|
|
72
|
+
--window N the baseline is the mean of only the N most recent prior
|
|
73
|
+
sessions (a windowed rolling baseline) instead of all-time mean.
|
|
74
|
+
|
|
75
|
+
pace --log .claude/metrics/cost-metering.jsonl [--budget B] [--period-days 30]
|
|
76
|
+
[--window-days 7]
|
|
77
|
+
Account-level pace guidance (#142): cumulative spend over a rolling
|
|
78
|
+
window, the implied daily rate, and the projected spend for a billing
|
|
79
|
+
period. With --budget it flags when the current pace would exhaust the
|
|
80
|
+
budget and suggests dropping a model tier for the rest of the window.
|
|
81
|
+
|
|
82
|
+
phase-mark --transcript T --phase LABEL [--log .claude/metrics/phase-markers.jsonl]
|
|
83
|
+
Context-pollution measurement (#1520): append one phase-boundary
|
|
84
|
+
marker capturing the main-loop resident context occupancy and the
|
|
85
|
+
cumulative output spend at a `/handoff` boundary. Fired by the
|
|
86
|
+
`phase_marker.py` PostToolUse hook. Kept in its own log, never folded
|
|
87
|
+
into the incremental `record` state.
|
|
88
|
+
|
|
89
|
+
phase-report --log .claude/metrics/phase-markers.jsonl [--json]
|
|
90
|
+
Report per-phase resident-vs-spent ratios from the phase markers: for
|
|
91
|
+
each phase, the context still resident at its boundary vs the output
|
|
92
|
+
tokens spent during it. A high ratio distinguishes context that
|
|
93
|
+
lingered (pollution) from one-time cost — a distinction the session
|
|
94
|
+
totals cannot make.
|
|
95
|
+
|
|
96
|
+
The transcript schema is read defensively (usage may sit on the record or under
|
|
97
|
+
`message`; model + agent attribution likewise), so it tolerates schema drift.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
from __future__ import annotations
|
|
101
|
+
|
|
102
|
+
import argparse
|
|
103
|
+
import hashlib
|
|
104
|
+
import json
|
|
105
|
+
import os
|
|
106
|
+
import sys
|
|
107
|
+
import time
|
|
108
|
+
from pathlib import Path
|
|
109
|
+
|
|
110
|
+
# hooks/lib/cost_meter.py -> plugin root is three parents up.
|
|
111
|
+
_PLUGIN_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
112
|
+
_DEFAULT_PRICING = _PLUGIN_ROOT / "knowledge/model-pricing.json"
|
|
113
|
+
|
|
114
|
+
_LIB_DIR = Path(__file__).resolve().parent
|
|
115
|
+
if str(_LIB_DIR) not in sys.path:
|
|
116
|
+
sys.path.insert(0, str(_LIB_DIR))
|
|
117
|
+
|
|
118
|
+
import artifact_paths
|
|
119
|
+
from atomic_state import append_line_locked
|
|
120
|
+
from pricing import cost as _cost
|
|
121
|
+
from pricing import load_pricing as _load_pricing
|
|
122
|
+
from pricing import rate as _rate
|
|
123
|
+
|
|
124
|
+
# session_log/ (#2050) -- unlike pricing.py's own placement in hooks/lib/
|
|
125
|
+
# (chosen specifically so scripts/ -> hooks/lib/ stayed the only cross-
|
|
126
|
+
# directory dependency direction, #1461), session_log/ has to live under
|
|
127
|
+
# scripts/lib/session_log/ per the epic's own decision (ADR 0042): it is the
|
|
128
|
+
# ONE sanctioned home for transcript-record/usage-block parsing, and
|
|
129
|
+
# session_report.py -- a scripts/ module -- is its primary consumer. This
|
|
130
|
+
# import is therefore genuinely hooks/lib/ -> scripts/lib/, the reverse of
|
|
131
|
+
# #1461's rule. It is safe to invert here for a reason #1461 itself did not
|
|
132
|
+
# have to consider: session_log/ ships INSIDE this same plugin package,
|
|
133
|
+
# always present wherever cost_meter.py (a real Stop hook) runs -- unlike
|
|
134
|
+
# the monorepo-only scripts/ tooling #1461 was written to keep hooks/lib/
|
|
135
|
+
# independent of. session_log/ itself imports nothing from hooks/lib/ (no
|
|
136
|
+
# cycle). Mirrors pricing.py's own sys.path.insert + bare-package-import
|
|
137
|
+
# MECHANISM, not its directionality.
|
|
138
|
+
sys.path.insert(
|
|
139
|
+
0, str(Path(__file__).resolve().parent.parent.parent / "scripts" / "lib")
|
|
140
|
+
)
|
|
141
|
+
from session_log import records as _records
|
|
142
|
+
|
|
143
|
+
# ---------------------------------------------------------------------------
|
|
144
|
+
# Incremental `record` state — a byte offset + running aggregates, keyed by
|
|
145
|
+
# transcript path, so a Stop/SubagentStop hook fire only tails bytes appended
|
|
146
|
+
# since the last fire instead of re-parsing the whole transcript (#732).
|
|
147
|
+
# Same TTL-purge-on-write pattern as tdd_guard.py / mutation_adapters/lib.py.
|
|
148
|
+
# ---------------------------------------------------------------------------
|
|
149
|
+
|
|
150
|
+
_STATE_TTL_SECONDS = 14400 # 4 hours
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _state_dir() -> Path:
|
|
154
|
+
return Path(os.environ.get("TMPDIR", "/tmp")) / "dev-team-cost-meter"
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _state_file(transcript_path: Path) -> Path:
|
|
158
|
+
digest = hashlib.sha256(
|
|
159
|
+
str(Path(transcript_path).resolve()).encode("utf-8")
|
|
160
|
+
).hexdigest()[:12]
|
|
161
|
+
return _state_dir() / f"session-{digest}.json"
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _purge_stale(state_dir: Path) -> None:
|
|
165
|
+
if not state_dir.is_dir():
|
|
166
|
+
return
|
|
167
|
+
now = time.time()
|
|
168
|
+
for path in state_dir.glob("session-*.json"):
|
|
169
|
+
try:
|
|
170
|
+
if now - path.stat().st_mtime > _STATE_TTL_SECONDS:
|
|
171
|
+
path.unlink()
|
|
172
|
+
except OSError:
|
|
173
|
+
pass
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _read_new_lines(path: Path, offset: int) -> tuple[int, list[str]]:
|
|
177
|
+
"""Tail `path` from `offset`, returning (new_offset, complete_lines).
|
|
178
|
+
|
|
179
|
+
Reads in binary mode and stops at the last full line so a line still
|
|
180
|
+
being written doesn't get parsed half-formed; the trailing partial bytes
|
|
181
|
+
are left unconsumed (picked up on the next fire once complete).
|
|
182
|
+
"""
|
|
183
|
+
try:
|
|
184
|
+
with path.open("rb") as fh:
|
|
185
|
+
fh.seek(offset)
|
|
186
|
+
chunk = fh.read()
|
|
187
|
+
except OSError:
|
|
188
|
+
return offset, []
|
|
189
|
+
if not chunk:
|
|
190
|
+
return offset, []
|
|
191
|
+
if chunk.endswith(b"\n"):
|
|
192
|
+
consumed = chunk
|
|
193
|
+
else:
|
|
194
|
+
last_newline = chunk.rfind(b"\n")
|
|
195
|
+
consumed = chunk[: last_newline + 1] if last_newline != -1 else b""
|
|
196
|
+
if not consumed:
|
|
197
|
+
return offset, []
|
|
198
|
+
new_offset = offset + len(consumed)
|
|
199
|
+
lines = consumed.decode("utf-8", errors="replace").splitlines()
|
|
200
|
+
return new_offset, lines
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _load_state(state_file: Path) -> dict | None:
|
|
204
|
+
if not state_file.is_file():
|
|
205
|
+
return None
|
|
206
|
+
try:
|
|
207
|
+
data = json.loads(state_file.read_text())
|
|
208
|
+
except (OSError, ValueError):
|
|
209
|
+
return None
|
|
210
|
+
return data if isinstance(data, dict) else None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _save_state(state_file: Path, payload: dict) -> None:
|
|
214
|
+
try:
|
|
215
|
+
state_file.write_text(json.dumps(payload))
|
|
216
|
+
except OSError:
|
|
217
|
+
pass
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _first_present_field(rec: dict, *keys: str):
|
|
221
|
+
"""Return the first present key on rec or rec['message']."""
|
|
222
|
+
for src in (
|
|
223
|
+
rec,
|
|
224
|
+
rec.get("message", {}) if isinstance(rec.get("message"), dict) else {},
|
|
225
|
+
):
|
|
226
|
+
for k in keys:
|
|
227
|
+
if k in src and src[k] is not None:
|
|
228
|
+
return src[k]
|
|
229
|
+
return None
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
# The four usage-block fields, and the join-map/sidechain/attribution
|
|
233
|
+
# primitives, now live in session_log.records (#2050) -- imported below as
|
|
234
|
+
# `_records`, aliased to the original private names so every call site in
|
|
235
|
+
# this file is otherwise unchanged.
|
|
236
|
+
_TOKEN_FIELDS = _records.USAGE_FIELDS
|
|
237
|
+
_harvest_agent_dispatch = _records.join_dispatch_agent_ids
|
|
238
|
+
_agent_type_key = _records.agent_type_for
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _new_bucket() -> dict:
|
|
242
|
+
bucket = {f: 0 for f in _TOKEN_FIELDS}
|
|
243
|
+
bucket["cost_usd"] = 0.0
|
|
244
|
+
bucket["messages"] = 0
|
|
245
|
+
return bucket
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _subagent_files(transcript_path: Path) -> list[Path]:
|
|
249
|
+
"""Sibling per-subagent transcript files for a session transcript.
|
|
250
|
+
|
|
251
|
+
Newer harness versions store sidechain turns in
|
|
252
|
+
`<dir>/<session-id>/subagents/agent-<agentId>.jsonl` rather than inline
|
|
253
|
+
`isSidechain` records in the session transcript (#1094). Returns [] when
|
|
254
|
+
the layout is absent (older format, or a subagent transcript itself).
|
|
255
|
+
"""
|
|
256
|
+
subagents_dir = transcript_path.parent / transcript_path.stem / "subagents"
|
|
257
|
+
try:
|
|
258
|
+
if not subagents_dir.is_dir():
|
|
259
|
+
return []
|
|
260
|
+
return sorted(p for p in subagents_dir.glob("agent-*.jsonl") if p.is_file())
|
|
261
|
+
except OSError:
|
|
262
|
+
return []
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _accumulate_lines(
|
|
266
|
+
lines,
|
|
267
|
+
pricing: dict,
|
|
268
|
+
by_model: dict,
|
|
269
|
+
by_thread: dict,
|
|
270
|
+
by_agent_type: dict,
|
|
271
|
+
totals: dict,
|
|
272
|
+
unpriced_models: set,
|
|
273
|
+
dispatch_types: dict,
|
|
274
|
+
agent_types: dict,
|
|
275
|
+
) -> None:
|
|
276
|
+
"""Fold `lines` (raw JSONL transcript records) into the given aggregates.
|
|
277
|
+
|
|
278
|
+
Mutates the bucket dicts, `totals`, `unpriced_models`, and the
|
|
279
|
+
`dispatch_types`/`agent_types` join maps in place so callers can seed them
|
|
280
|
+
from a persisted running state and only pass in the newly-appended lines
|
|
281
|
+
(#732) — or seed them empty and pass the whole transcript for a one-shot
|
|
282
|
+
full parse.
|
|
283
|
+
"""
|
|
284
|
+
for line in lines:
|
|
285
|
+
line = line.strip()
|
|
286
|
+
if not line:
|
|
287
|
+
continue
|
|
288
|
+
try:
|
|
289
|
+
rec = json.loads(line)
|
|
290
|
+
except json.JSONDecodeError:
|
|
291
|
+
continue
|
|
292
|
+
# Dispatch metadata can sit on records with no usage (tool_use /
|
|
293
|
+
# tool_result turns), so harvest before the usage gate.
|
|
294
|
+
_harvest_agent_dispatch(rec, dispatch_types, agent_types)
|
|
295
|
+
usage = _first_present_field(rec, "usage")
|
|
296
|
+
if not isinstance(usage, dict):
|
|
297
|
+
continue
|
|
298
|
+
model = _first_present_field(rec, "model") or "unknown"
|
|
299
|
+
# Main-loop vs subagent: the native top-level `isSidechain` flag is true
|
|
300
|
+
# on sidechain (subagent) turns.
|
|
301
|
+
thread = "subagent" if _records.is_sidechain(rec) else "main"
|
|
302
|
+
agent_type = _agent_type_key(rec, agent_types)
|
|
303
|
+
|
|
304
|
+
rate = _rate(pricing, model)
|
|
305
|
+
cost = _cost(usage, rate, pricing)
|
|
306
|
+
# Excludes zero-token records from `unpriced_models` (#1830 fix
|
|
307
|
+
# review), not just the literal "unknown" placeholder: the harness
|
|
308
|
+
# writes assistant records with `model: "<synthetic>"` and every usage
|
|
309
|
+
# field zero for interrupt/auth-failure notices (observed in local
|
|
310
|
+
# session transcripts inspected during #1830's investigation; not
|
|
311
|
+
# covered by an automated probe). Those records are not billable —
|
|
312
|
+
# zero tokens cost zero regardless of whether the model has a pricing
|
|
313
|
+
# entry — so flagging them as "unpriced" would fire on nearly every
|
|
314
|
+
# session and bury the real signal (a genuinely billable model with no
|
|
315
|
+
# rate) under permanent noise. Gating on token presence rather than
|
|
316
|
+
# naming "<synthetic>" specifically also covers any future
|
|
317
|
+
# harness-internal pseudo-model without another denylist edit.
|
|
318
|
+
if not rate and model != "unknown" and any(usage.get(f) for f in _TOKEN_FIELDS):
|
|
319
|
+
unpriced_models.add(model)
|
|
320
|
+
|
|
321
|
+
for bucket, key in (
|
|
322
|
+
(by_model, model),
|
|
323
|
+
(by_thread, thread),
|
|
324
|
+
(by_agent_type, agent_type),
|
|
325
|
+
):
|
|
326
|
+
b = bucket.setdefault(key, _new_bucket())
|
|
327
|
+
for f in _TOKEN_FIELDS:
|
|
328
|
+
b[f] += usage.get(f, 0) or 0
|
|
329
|
+
b["cost_usd"] = round(b["cost_usd"] + cost, 6)
|
|
330
|
+
b["messages"] += 1
|
|
331
|
+
|
|
332
|
+
for f in _TOKEN_FIELDS:
|
|
333
|
+
totals[f] += usage.get(f, 0) or 0
|
|
334
|
+
totals["cost_usd"] = round(totals["cost_usd"] + cost, 6)
|
|
335
|
+
totals["messages"] += 1
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def parse_transcript(path: Path, pricing: dict) -> dict:
|
|
339
|
+
"""Aggregate transcript usage by model, thread, and agent type.
|
|
340
|
+
|
|
341
|
+
Only dimensions the harness actually records are attributed (#170, #1094):
|
|
342
|
+
the model (`message.model`), the main/subagent split (native top-level
|
|
343
|
+
`isSidechain`), and the agent type (`attributionAgent`, or the Task/Agent
|
|
344
|
+
dispatch join — see module docstring), plus the session total. Sibling
|
|
345
|
+
per-subagent transcript files are folded in when present (#1094).
|
|
346
|
+
|
|
347
|
+
Full one-shot parse — used by `report`/`regression`/`pace`, which run
|
|
348
|
+
on-demand rather than once per hook fire. `record` (the Stop-hook hot
|
|
349
|
+
path) uses the incremental `_read_new_lines` + `_accumulate_lines` path
|
|
350
|
+
in `cmd_record` instead so it doesn't re-parse the whole transcript on
|
|
351
|
+
every turn (#732).
|
|
352
|
+
"""
|
|
353
|
+
by_model: dict[str, dict] = {}
|
|
354
|
+
by_thread: dict[str, dict] = {}
|
|
355
|
+
by_agent_type: dict[str, dict] = {}
|
|
356
|
+
totals = _new_bucket()
|
|
357
|
+
unpriced_models: set[str] = set()
|
|
358
|
+
dispatch_types: dict[str, str] = {}
|
|
359
|
+
agent_types: dict[str, str] = {}
|
|
360
|
+
|
|
361
|
+
# Main transcript first so the dispatch join maps are populated before the
|
|
362
|
+
# sibling subagent files (whose records fall back on them) are folded in.
|
|
363
|
+
sources = [path] + _subagent_files(path)
|
|
364
|
+
for source in sources:
|
|
365
|
+
try:
|
|
366
|
+
source_lines = source.read_text().splitlines()
|
|
367
|
+
except OSError:
|
|
368
|
+
if source is path:
|
|
369
|
+
raise
|
|
370
|
+
continue
|
|
371
|
+
_accumulate_lines(
|
|
372
|
+
source_lines,
|
|
373
|
+
pricing,
|
|
374
|
+
by_model,
|
|
375
|
+
by_thread,
|
|
376
|
+
by_agent_type,
|
|
377
|
+
totals,
|
|
378
|
+
unpriced_models,
|
|
379
|
+
dispatch_types,
|
|
380
|
+
agent_types,
|
|
381
|
+
)
|
|
382
|
+
|
|
383
|
+
return {
|
|
384
|
+
"by_model": by_model,
|
|
385
|
+
"by_thread": by_thread,
|
|
386
|
+
"by_agent_type": by_agent_type,
|
|
387
|
+
"totals": totals,
|
|
388
|
+
"unpriced_models": sorted(unpriced_models),
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def _print_dimension(title: str, bucket: dict) -> None:
|
|
393
|
+
if not bucket:
|
|
394
|
+
return
|
|
395
|
+
print(f"\n{title:<28} {'IN':>10} {'OUT':>10} {'COST $':>10}")
|
|
396
|
+
print("-" * 60)
|
|
397
|
+
for key, b in sorted(bucket.items(), key=lambda kv: -kv[1]["cost_usd"]):
|
|
398
|
+
print(
|
|
399
|
+
f"{key:<28} {b['input_tokens']:>10} {b['output_tokens']:>10} "
|
|
400
|
+
f"{b['cost_usd']:>10.4f}"
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _print_report(summary: dict) -> None:
|
|
405
|
+
t = summary["totals"]
|
|
406
|
+
print(f"# Cost meter — {t['messages']} assistant message(s)")
|
|
407
|
+
_print_dimension("MODEL", summary["by_model"])
|
|
408
|
+
_print_dimension("THREAD (main/subagent)", summary["by_thread"])
|
|
409
|
+
_print_dimension("AGENT TYPE", summary.get("by_agent_type", {}))
|
|
410
|
+
print("-" * 60)
|
|
411
|
+
print(
|
|
412
|
+
f"{'TOTAL':<28} {t['input_tokens']:>10} {t['output_tokens']:>10} "
|
|
413
|
+
f"{t['cost_usd']:>10.4f}"
|
|
414
|
+
)
|
|
415
|
+
if summary["unpriced_models"]:
|
|
416
|
+
print(
|
|
417
|
+
f"\n⚠ no pricing for: {', '.join(summary['unpriced_models'])} "
|
|
418
|
+
f"(add to knowledge/model-pricing.json)"
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def cmd_report(args, pricing) -> int:
|
|
423
|
+
summary = parse_transcript(Path(args.transcript), pricing)
|
|
424
|
+
if args.json:
|
|
425
|
+
print(json.dumps(summary, indent=2))
|
|
426
|
+
else:
|
|
427
|
+
_print_report(summary)
|
|
428
|
+
return 0
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _record_state_is_usable(
|
|
432
|
+
state: dict | None, size: int, subagent_paths: list[Path]
|
|
433
|
+
) -> bool:
|
|
434
|
+
"""Whether a persisted `record` state can be resumed from (#732, #1094).
|
|
435
|
+
|
|
436
|
+
Unusable when: absent/corrupt; the main offset is not an int within the
|
|
437
|
+
current file size (rotation/truncation); the schema predates the
|
|
438
|
+
agent-type dimension (resuming would ship a by_agent_type that no longer
|
|
439
|
+
sums to totals); or a tracked subagent file shrank below its offset.
|
|
440
|
+
"""
|
|
441
|
+
if not state:
|
|
442
|
+
return False
|
|
443
|
+
offset = state.get("offset")
|
|
444
|
+
if not isinstance(offset, int) or not 0 <= offset <= size:
|
|
445
|
+
return False
|
|
446
|
+
if "by_agent_type" not in state:
|
|
447
|
+
return False # pre-#1094 state schema — rebuild from byte 0
|
|
448
|
+
subagent_offsets = state.get("subagent_offsets")
|
|
449
|
+
if subagent_offsets is not None and not isinstance(subagent_offsets, dict):
|
|
450
|
+
return False
|
|
451
|
+
sizes_by_name = {}
|
|
452
|
+
for p in subagent_paths:
|
|
453
|
+
try:
|
|
454
|
+
sizes_by_name[p.name] = p.stat().st_size
|
|
455
|
+
except OSError:
|
|
456
|
+
sizes_by_name[p.name] = 0
|
|
457
|
+
for name, sub_offset in (subagent_offsets or {}).items():
|
|
458
|
+
if not isinstance(sub_offset, int) or sub_offset > sizes_by_name.get(name, 0):
|
|
459
|
+
return False
|
|
460
|
+
return True
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def cmd_record(args, pricing) -> int:
|
|
464
|
+
tpath = Path(args.transcript)
|
|
465
|
+
if not tpath.is_file():
|
|
466
|
+
return 0 # fail-open: hook must never break the session
|
|
467
|
+
|
|
468
|
+
# Incremental read (#732): persist a byte offset + the running aggregates
|
|
469
|
+
# in a tmp-state file keyed by the transcript path, so this Stop-hook fire
|
|
470
|
+
# only tails bytes appended since the last fire instead of re-parsing the
|
|
471
|
+
# whole transcript every time (O(new content), not O(turns) per fire).
|
|
472
|
+
state_dir = _state_dir()
|
|
473
|
+
state_dir.mkdir(parents=True, exist_ok=True)
|
|
474
|
+
_purge_stale(state_dir)
|
|
475
|
+
state_file = _state_file(tpath)
|
|
476
|
+
|
|
477
|
+
state = _load_state(state_file)
|
|
478
|
+
size = tpath.stat().st_size
|
|
479
|
+
subagent_paths = _subagent_files(tpath)
|
|
480
|
+
|
|
481
|
+
if _record_state_is_usable(state, size, subagent_paths):
|
|
482
|
+
offset = state["offset"]
|
|
483
|
+
subagent_offsets = {
|
|
484
|
+
k: v
|
|
485
|
+
for k, v in (state.get("subagent_offsets") or {}).items()
|
|
486
|
+
if isinstance(v, int)
|
|
487
|
+
}
|
|
488
|
+
by_model = state.get("by_model") or {}
|
|
489
|
+
by_thread = state.get("by_thread") or {}
|
|
490
|
+
by_agent_type = state.get("by_agent_type") or {}
|
|
491
|
+
totals = state.get("totals") or _new_bucket()
|
|
492
|
+
for f in _TOKEN_FIELDS:
|
|
493
|
+
totals.setdefault(f, 0)
|
|
494
|
+
totals.setdefault("cost_usd", 0.0)
|
|
495
|
+
totals.setdefault("messages", 0)
|
|
496
|
+
unpriced_models = set(state.get("unpriced_models") or [])
|
|
497
|
+
dispatch_types = state.get("dispatch_types") or {}
|
|
498
|
+
agent_types = state.get("agent_types") or {}
|
|
499
|
+
else:
|
|
500
|
+
# First fire, a pre-#1094 state schema, or a transcript that shrank
|
|
501
|
+
# (rotated/truncated) since the last fire — start fresh rather than
|
|
502
|
+
# seek past a stale offset or ship a partial agent-type dimension.
|
|
503
|
+
offset = 0
|
|
504
|
+
subagent_offsets = {}
|
|
505
|
+
by_model, by_thread, by_agent_type = {}, {}, {}
|
|
506
|
+
totals, unpriced_models = _new_bucket(), set()
|
|
507
|
+
dispatch_types, agent_types = {}, {}
|
|
508
|
+
|
|
509
|
+
def _fold(lines) -> None:
|
|
510
|
+
_accumulate_lines(
|
|
511
|
+
lines,
|
|
512
|
+
pricing,
|
|
513
|
+
by_model,
|
|
514
|
+
by_thread,
|
|
515
|
+
by_agent_type,
|
|
516
|
+
totals,
|
|
517
|
+
unpriced_models,
|
|
518
|
+
dispatch_types,
|
|
519
|
+
agent_types,
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
# Main transcript first so dispatch joins land before subagent turns that
|
|
523
|
+
# may need them, then each sibling per-subagent transcript (#1094) — each
|
|
524
|
+
# with its own persisted byte offset so every source is tailed, not
|
|
525
|
+
# re-parsed, per fire (#732).
|
|
526
|
+
new_offset, new_lines = _read_new_lines(tpath, offset)
|
|
527
|
+
_fold(new_lines)
|
|
528
|
+
for sub_path in subagent_paths:
|
|
529
|
+
sub_offset = subagent_offsets.get(sub_path.name, 0)
|
|
530
|
+
sub_new_offset, sub_lines = _read_new_lines(sub_path, sub_offset)
|
|
531
|
+
_fold(sub_lines)
|
|
532
|
+
subagent_offsets[sub_path.name] = sub_new_offset
|
|
533
|
+
|
|
534
|
+
_save_state(
|
|
535
|
+
state_file,
|
|
536
|
+
{
|
|
537
|
+
"offset": new_offset,
|
|
538
|
+
"subagent_offsets": subagent_offsets,
|
|
539
|
+
"by_model": by_model,
|
|
540
|
+
"by_thread": by_thread,
|
|
541
|
+
"by_agent_type": by_agent_type,
|
|
542
|
+
"totals": totals,
|
|
543
|
+
"unpriced_models": sorted(unpriced_models),
|
|
544
|
+
"dispatch_types": dispatch_types,
|
|
545
|
+
"agent_types": agent_types,
|
|
546
|
+
},
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
summary = {
|
|
550
|
+
"by_model": by_model,
|
|
551
|
+
"by_thread": by_thread,
|
|
552
|
+
"by_agent_type": by_agent_type,
|
|
553
|
+
"totals": totals,
|
|
554
|
+
"unpriced_models": sorted(unpriced_models),
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
from datetime import datetime, timezone
|
|
558
|
+
|
|
559
|
+
def _slim(bucket: dict) -> dict:
|
|
560
|
+
# Persist cost + every token field per bucket (#1513), deriving the
|
|
561
|
+
# token names from _TOKEN_FIELDS (the single source of truth used by
|
|
562
|
+
# _new_bucket/_accumulate_lines) so a future field flows into the
|
|
563
|
+
# durable log automatically. cache_creation is the spawn-floor F
|
|
564
|
+
# component the epic measures — it must live in the log, not only in
|
|
565
|
+
# `report --json`.
|
|
566
|
+
return {
|
|
567
|
+
k: {"cost_usd": b["cost_usd"], **{f: b[f] for f in _TOKEN_FIELDS}}
|
|
568
|
+
for k, b in bucket.items()
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
line = {
|
|
572
|
+
"timestamp": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
573
|
+
"transcript": tpath.name,
|
|
574
|
+
"total": summary["totals"],
|
|
575
|
+
"by_model": _slim(summary["by_model"]),
|
|
576
|
+
"by_thread": _slim(summary["by_thread"]),
|
|
577
|
+
"by_agent_type": _slim(summary["by_agent_type"]),
|
|
578
|
+
# Always present, even when empty (#1830). This field was computed and
|
|
579
|
+
# warned about in `report`, and persisted to the incremental state file,
|
|
580
|
+
# but omitted from the durable line — which is the ONLY thing every
|
|
581
|
+
# downstream consumer reads: /autoship's `--max-cost-usd` gate,
|
|
582
|
+
# `regression`, and `pace`. A model with no pricing entry therefore
|
|
583
|
+
# contributed $0.00 to a budget ceiling and was indistinguishable from
|
|
584
|
+
# one that is genuinely free. Emitting `[]` rather than omitting the key
|
|
585
|
+
# on the clean path matters too: a consumer can then treat absence as
|
|
586
|
+
# "this record predates the check" instead of "nothing was unpriced".
|
|
587
|
+
"unpriced_models": summary["unpriced_models"],
|
|
588
|
+
}
|
|
589
|
+
log = Path(args.log)
|
|
590
|
+
log.parent.mkdir(parents=True, exist_ok=True)
|
|
591
|
+
append_line_locked(log, json.dumps(line) + "\n", fail_open=False)
|
|
592
|
+
return 0
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
def _warn_unpriced(entries: list) -> None:
|
|
596
|
+
"""Print a warning naming every model that contributed $0.00 to the records
|
|
597
|
+
being summarized because it has no pricing entry (#1830).
|
|
598
|
+
|
|
599
|
+
Any verdict computed over such records understates real spend, so say so
|
|
600
|
+
rather than reporting a number that reads as complete. Records written
|
|
601
|
+
before `unpriced_models` was added to the durable line simply lack the key
|
|
602
|
+
and contribute nothing here — silence means "no unpriced model was seen in
|
|
603
|
+
the records that carry the field", not "every record was checked"."""
|
|
604
|
+
unpriced = sorted(
|
|
605
|
+
{
|
|
606
|
+
model
|
|
607
|
+
for entry in entries
|
|
608
|
+
for model in (entry.get("unpriced_models") or [])
|
|
609
|
+
if isinstance(model, str)
|
|
610
|
+
}
|
|
611
|
+
)
|
|
612
|
+
if not unpriced:
|
|
613
|
+
return
|
|
614
|
+
print(
|
|
615
|
+
f"⚠ model(s) that priced at $0.00 in these records: "
|
|
616
|
+
f"{', '.join(unpriced)} — the figures below UNDERSTATE real cost. If "
|
|
617
|
+
"they are absent from knowledge/model-pricing.json, add them; if they "
|
|
618
|
+
"were added recently, re-record the affected sessions."
|
|
619
|
+
)
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def _read_log_entries(log: Path) -> list:
|
|
623
|
+
entries = []
|
|
624
|
+
for line in log.read_text().splitlines():
|
|
625
|
+
line = line.strip()
|
|
626
|
+
if not line:
|
|
627
|
+
continue
|
|
628
|
+
try:
|
|
629
|
+
entries.append(json.loads(line))
|
|
630
|
+
except json.JSONDecodeError:
|
|
631
|
+
continue
|
|
632
|
+
return entries
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def cmd_regression(args, pricing) -> int:
|
|
636
|
+
log = Path(args.log)
|
|
637
|
+
if not log.is_file():
|
|
638
|
+
print("no metrics log yet; nothing to compare")
|
|
639
|
+
return 0
|
|
640
|
+
entries = _read_log_entries(log)
|
|
641
|
+
_warn_unpriced(entries)
|
|
642
|
+
costs = [e.get("total", {}).get("cost_usd", 0.0) for e in entries]
|
|
643
|
+
if len(costs) < 2:
|
|
644
|
+
print(f"only {len(costs)} session(s) logged; need >=2 to compare")
|
|
645
|
+
return 0
|
|
646
|
+
latest = costs[-1]
|
|
647
|
+
prior = costs[:-1]
|
|
648
|
+
# Windowed rolling baseline (#134): mean of only the N most recent priors.
|
|
649
|
+
window = getattr(args, "window", 0) or 0
|
|
650
|
+
if window > 0:
|
|
651
|
+
prior = prior[-window:]
|
|
652
|
+
mean = sum(prior) / len(prior)
|
|
653
|
+
limit = mean * (1 + args.tolerance)
|
|
654
|
+
win_label = f"window {len(prior)}" if window > 0 else f"prior {len(prior)}"
|
|
655
|
+
print(
|
|
656
|
+
f"latest=${latest:.4f} rolling-mean({win_label})=${mean:.4f} "
|
|
657
|
+
f"limit(+{int(args.tolerance * 100)}%)=${limit:.4f}"
|
|
658
|
+
)
|
|
659
|
+
if mean > 0 and latest > limit:
|
|
660
|
+
print(f"COST REGRESSION: latest ${latest:.4f} exceeds limit ${limit:.4f}")
|
|
661
|
+
return 1
|
|
662
|
+
print("no cost regression")
|
|
663
|
+
return 0
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def _parse_ts(s: str):
|
|
667
|
+
from datetime import datetime, timezone
|
|
668
|
+
|
|
669
|
+
try:
|
|
670
|
+
return datetime.strptime(s, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc)
|
|
671
|
+
except (ValueError, TypeError):
|
|
672
|
+
return None
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
def cmd_pace(args, pricing) -> int:
|
|
676
|
+
"""Account-level pace: cumulative spend over a rolling window, projected
|
|
677
|
+
against a budget for a billing period; flags when pace would exhaust it."""
|
|
678
|
+
from datetime import datetime, timedelta, timezone
|
|
679
|
+
|
|
680
|
+
log = Path(args.log)
|
|
681
|
+
if not log.is_file():
|
|
682
|
+
print("no metrics log yet; nothing to pace")
|
|
683
|
+
return 0
|
|
684
|
+
now = datetime.now(timezone.utc)
|
|
685
|
+
window_start = now - timedelta(days=args.window_days)
|
|
686
|
+
in_window = []
|
|
687
|
+
windowed_entries = []
|
|
688
|
+
for e in _read_log_entries(log):
|
|
689
|
+
ts = _parse_ts(e.get("timestamp", ""))
|
|
690
|
+
cost = e.get("total", {}).get("cost_usd", 0.0)
|
|
691
|
+
if ts is not None and ts >= window_start:
|
|
692
|
+
in_window.append((ts, cost))
|
|
693
|
+
windowed_entries.append(e)
|
|
694
|
+
|
|
695
|
+
# Scoped to the window, not the whole log: a budget projection understates
|
|
696
|
+
# only when an unpriced model appears in the records it actually sums (#1830).
|
|
697
|
+
_warn_unpriced(windowed_entries)
|
|
698
|
+
|
|
699
|
+
spend = round(sum(c for _, c in in_window), 6)
|
|
700
|
+
print(f"# Account pace — last {args.window_days} day(s)")
|
|
701
|
+
print(f" sessions in window: {len(in_window)}")
|
|
702
|
+
print(f" cumulative spend: ${spend:.4f}")
|
|
703
|
+
if not in_window:
|
|
704
|
+
print(" (no sessions in window; nothing to project)")
|
|
705
|
+
return 0
|
|
706
|
+
|
|
707
|
+
earliest = min(ts for ts, _ in in_window)
|
|
708
|
+
elapsed_days = max((now - earliest).total_seconds() / 86400.0, 1e-9)
|
|
709
|
+
daily_rate = spend / elapsed_days
|
|
710
|
+
projected = daily_rate * args.period_days
|
|
711
|
+
print(
|
|
712
|
+
f" daily rate: ${daily_rate:.4f}/day "
|
|
713
|
+
f"(over {elapsed_days:.2f} active day(s))"
|
|
714
|
+
)
|
|
715
|
+
print(f" projected / {args.period_days}d: ${projected:.4f}")
|
|
716
|
+
|
|
717
|
+
if args.budget is not None and args.budget > 0:
|
|
718
|
+
pct = projected / args.budget * 100
|
|
719
|
+
print(
|
|
720
|
+
f" budget / {args.period_days}d: ${args.budget:.2f} "
|
|
721
|
+
f"({pct:.0f}% of budget at current pace)"
|
|
722
|
+
)
|
|
723
|
+
if projected > args.budget:
|
|
724
|
+
print(
|
|
725
|
+
f"\n⚠ PACE EXCEEDS BUDGET: at ${daily_rate:.4f}/day you would "
|
|
726
|
+
f"spend ${projected:.2f} over {args.period_days} days, past the "
|
|
727
|
+
f"${args.budget:.2f} budget."
|
|
728
|
+
)
|
|
729
|
+
print(
|
|
730
|
+
" Consider dropping Opus→Sonnet for the remainder of the "
|
|
731
|
+
"window (see .claude/model-overrides.json / /harness-audit)."
|
|
732
|
+
)
|
|
733
|
+
return 0
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def _resident_and_spent(path: Path) -> tuple[int, int]:
|
|
737
|
+
"""(resident_tokens, spent_output_cumulative) for the MAIN-LOOP context (#1520).
|
|
738
|
+
|
|
739
|
+
Context-pollution measurement is about the *live orchestrating* context —
|
|
740
|
+
the one that "charges rent" every subsequent turn per Martin Fowler's "The
|
|
741
|
+
Orchestrator's Tax" — so both figures are computed over main-loop
|
|
742
|
+
(non-sidechain) turns only; subagent contexts are separate and are
|
|
743
|
+
discarded at their own SubagentStop.
|
|
744
|
+
|
|
745
|
+
* resident_tokens = the MOST-RECENT main-loop usage record's context
|
|
746
|
+
occupancy (input + cache_read + cache_creation) — what still occupies
|
|
747
|
+
the window at this point, the same numerator the retired context ceiling
|
|
748
|
+
guard used (#2177). Overwritten each turn so it ends as the last turn's occupancy.
|
|
749
|
+
* spent_output_cumulative = the sum of output_tokens across all main-loop
|
|
750
|
+
turns so far — the one-time generation bill accrued to this point.
|
|
751
|
+
|
|
752
|
+
Read defensively (schema drift tolerated the same way parse_transcript does);
|
|
753
|
+
a missing/unreadable transcript returns (0, 0) rather than raising, so the
|
|
754
|
+
fail-open phase-mark hook never breaks a turn.
|
|
755
|
+
"""
|
|
756
|
+
resident = 0
|
|
757
|
+
spent_output = 0
|
|
758
|
+
try:
|
|
759
|
+
lines = path.read_text().splitlines()
|
|
760
|
+
except OSError:
|
|
761
|
+
return 0, 0
|
|
762
|
+
for line in lines:
|
|
763
|
+
line = line.strip()
|
|
764
|
+
if not line:
|
|
765
|
+
continue
|
|
766
|
+
try:
|
|
767
|
+
rec = json.loads(line)
|
|
768
|
+
except json.JSONDecodeError:
|
|
769
|
+
continue
|
|
770
|
+
if _records.is_sidechain(rec):
|
|
771
|
+
continue # main-loop context only
|
|
772
|
+
usage = _first_present_field(rec, "usage")
|
|
773
|
+
if not isinstance(usage, dict):
|
|
774
|
+
continue
|
|
775
|
+
resident = (
|
|
776
|
+
_records.usage_field(usage, "input_tokens")
|
|
777
|
+
+ _records.usage_field(usage, "cache_read_input_tokens")
|
|
778
|
+
+ _records.usage_field(usage, "cache_creation_input_tokens")
|
|
779
|
+
)
|
|
780
|
+
spent_output += _records.usage_field(usage, "output_tokens")
|
|
781
|
+
return resident, spent_output
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
def cmd_phase_mark(args, pricing) -> int:
|
|
785
|
+
"""Append one phase-boundary marker to the phase-markers log (#1520).
|
|
786
|
+
|
|
787
|
+
Fired by the `phase_marker.py` PostToolUse hook when `/handoff` runs (a
|
|
788
|
+
phase boundary). Captures the resident/spent snapshot at the boundary so
|
|
789
|
+
`phase-report` can later derive per-phase context-pollution ratios the
|
|
790
|
+
session totals alone cannot (the harness records no phase marker itself —
|
|
791
|
+
the reason per-phase cost attribution was removed in #170). Kept in its OWN
|
|
792
|
+
log, deliberately NOT folded into the incremental `record` state, so this
|
|
793
|
+
additive dimension never touches that security-sensitive hot path.
|
|
794
|
+
|
|
795
|
+
Fail-open: a missing transcript is a no-op success (the hook must never
|
|
796
|
+
break a turn)."""
|
|
797
|
+
tpath = Path(args.transcript)
|
|
798
|
+
if not tpath.is_file():
|
|
799
|
+
return 0
|
|
800
|
+
resident, spent = _resident_and_spent(tpath)
|
|
801
|
+
from datetime import datetime, timezone
|
|
802
|
+
|
|
803
|
+
line = {
|
|
804
|
+
"timestamp": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
805
|
+
"transcript": tpath.name,
|
|
806
|
+
"phase": args.phase or "unlabeled",
|
|
807
|
+
"resident_tokens": resident,
|
|
808
|
+
"spent_output_cumulative": spent,
|
|
809
|
+
}
|
|
810
|
+
log = Path(args.log)
|
|
811
|
+
log.parent.mkdir(parents=True, exist_ok=True)
|
|
812
|
+
append_line_locked(log, json.dumps(line) + "\n", fail_open=False)
|
|
813
|
+
return 0
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
def _phase_rows(entries: list) -> list:
|
|
817
|
+
"""Per-phase resident/spent rows from ordered phase markers (#1520).
|
|
818
|
+
|
|
819
|
+
`spent_output_cumulative` is monotonic across a session, so a phase's own
|
|
820
|
+
spend is the delta from the prior marker; `resident_tokens` is a snapshot,
|
|
821
|
+
used as-is. `resident_to_spent_ratio = resident / spent_phase` is the
|
|
822
|
+
context-pollution proxy — high means a large resident footprint relative to
|
|
823
|
+
the fresh generation the phase did (context that lingered rather than being
|
|
824
|
+
compacted). `None` when the phase spent nothing new (ratio undefined), never
|
|
825
|
+
a divide-by-zero.
|
|
826
|
+
"""
|
|
827
|
+
rows = []
|
|
828
|
+
prev_spent = 0
|
|
829
|
+
for e in entries:
|
|
830
|
+
resident = e.get("resident_tokens", 0) or 0
|
|
831
|
+
spent_cum = e.get("spent_output_cumulative", 0) or 0
|
|
832
|
+
# A cumulative counter can only go up within a session; clamp at 0 so a
|
|
833
|
+
# transcript rotation (counter reset) never yields a negative phase spend.
|
|
834
|
+
spent_phase = max(spent_cum - prev_spent, 0)
|
|
835
|
+
prev_spent = spent_cum
|
|
836
|
+
ratio = round(resident / spent_phase, 2) if spent_phase > 0 else None
|
|
837
|
+
rows.append(
|
|
838
|
+
{
|
|
839
|
+
"phase": e.get("phase", "unlabeled"),
|
|
840
|
+
"resident_tokens": resident,
|
|
841
|
+
"spent_tokens": spent_phase,
|
|
842
|
+
"resident_to_spent_ratio": ratio,
|
|
843
|
+
}
|
|
844
|
+
)
|
|
845
|
+
return rows
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def cmd_phase_report(args, pricing) -> int:
|
|
849
|
+
"""Print per-phase resident/spent ratios from the phase-markers log (#1520)."""
|
|
850
|
+
log = Path(args.log)
|
|
851
|
+
if not log.is_file():
|
|
852
|
+
print("no phase markers recorded yet; nothing to report")
|
|
853
|
+
return 0
|
|
854
|
+
entries = []
|
|
855
|
+
for line in log.read_text().splitlines():
|
|
856
|
+
line = line.strip()
|
|
857
|
+
if not line:
|
|
858
|
+
continue
|
|
859
|
+
try:
|
|
860
|
+
entries.append(json.loads(line))
|
|
861
|
+
except json.JSONDecodeError:
|
|
862
|
+
continue
|
|
863
|
+
rows = _phase_rows(entries)
|
|
864
|
+
if getattr(args, "json", False):
|
|
865
|
+
print(json.dumps({"by_phase": rows}, indent=2))
|
|
866
|
+
return 0
|
|
867
|
+
print("# Context pollution — resident vs one-time spend, per phase (#1520)")
|
|
868
|
+
print(f"{'PHASE':<24} {'RESIDENT':>10} {'SPENT':>10} {'RESIDENT/SPENT':>15}")
|
|
869
|
+
print("-" * 62)
|
|
870
|
+
for r in rows:
|
|
871
|
+
ratio = "-" if r["resident_to_spent_ratio"] is None else f"{r['resident_to_spent_ratio']:.2f}"
|
|
872
|
+
print(
|
|
873
|
+
f"{r['phase']:<24} {r['resident_tokens']:>10} {r['spent_tokens']:>10} "
|
|
874
|
+
f"{ratio:>15}"
|
|
875
|
+
)
|
|
876
|
+
print(
|
|
877
|
+
"\nresident = main-loop context occupancy at the phase's /handoff boundary; "
|
|
878
|
+
"spent = output tokens generated during the phase. A high ratio flags a "
|
|
879
|
+
"phase whose context lingered (pollution) rather than being one-time cost. "
|
|
880
|
+
"Session-scoped proxy: resident is sampled at the /handoff marker, the "
|
|
881
|
+
"closest phase boundary the harness exposes."
|
|
882
|
+
)
|
|
883
|
+
return 0
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
def main(argv: list[str]) -> int:
|
|
887
|
+
ap = argparse.ArgumentParser(description=__doc__)
|
|
888
|
+
ap.add_argument("--pricing", default=str(_DEFAULT_PRICING))
|
|
889
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
890
|
+
|
|
891
|
+
p = sub.add_parser("report")
|
|
892
|
+
p.add_argument("--transcript", required=True)
|
|
893
|
+
p.add_argument("--json", action="store_true")
|
|
894
|
+
p = sub.add_parser("record")
|
|
895
|
+
p.add_argument("--transcript", required=True)
|
|
896
|
+
p.add_argument("--log", default=str(artifact_paths.metrics_dir() / "cost-metering.jsonl"))
|
|
897
|
+
p = sub.add_parser("regression")
|
|
898
|
+
p.add_argument("--log", default=str(artifact_paths.metrics_dir() / "cost-metering.jsonl"))
|
|
899
|
+
p.add_argument("--tolerance", type=float, default=0.5)
|
|
900
|
+
p.add_argument(
|
|
901
|
+
"--window",
|
|
902
|
+
type=int,
|
|
903
|
+
default=0,
|
|
904
|
+
help="baseline = mean of the N most recent prior sessions (0 = all-time mean)",
|
|
905
|
+
)
|
|
906
|
+
p = sub.add_parser("pace")
|
|
907
|
+
p.add_argument("--log", default=str(artifact_paths.metrics_dir() / "cost-metering.jsonl"))
|
|
908
|
+
p.add_argument(
|
|
909
|
+
"--budget",
|
|
910
|
+
type=float,
|
|
911
|
+
default=None,
|
|
912
|
+
help="account budget for one billing period (dollars)",
|
|
913
|
+
)
|
|
914
|
+
p.add_argument(
|
|
915
|
+
"--period-days",
|
|
916
|
+
type=int,
|
|
917
|
+
default=30,
|
|
918
|
+
help="length of the billing period to project against",
|
|
919
|
+
)
|
|
920
|
+
p.add_argument(
|
|
921
|
+
"--window-days",
|
|
922
|
+
type=int,
|
|
923
|
+
default=7,
|
|
924
|
+
help="rolling lookback used to estimate the current daily rate",
|
|
925
|
+
)
|
|
926
|
+
p = sub.add_parser("phase-mark")
|
|
927
|
+
p.add_argument("--transcript", required=True)
|
|
928
|
+
p.add_argument(
|
|
929
|
+
"--phase",
|
|
930
|
+
default=None,
|
|
931
|
+
help="phase label for this boundary marker (e.g. research/plan/implement)",
|
|
932
|
+
)
|
|
933
|
+
p.add_argument(
|
|
934
|
+
"--log", default=str(artifact_paths.metrics_dir() / "phase-markers.jsonl")
|
|
935
|
+
)
|
|
936
|
+
p = sub.add_parser("phase-report")
|
|
937
|
+
p.add_argument(
|
|
938
|
+
"--log", default=str(artifact_paths.metrics_dir() / "phase-markers.jsonl")
|
|
939
|
+
)
|
|
940
|
+
p.add_argument("--json", action="store_true")
|
|
941
|
+
|
|
942
|
+
args = ap.parse_args(argv)
|
|
943
|
+
pricing = _load_pricing(Path(args.pricing))
|
|
944
|
+
return {
|
|
945
|
+
"report": cmd_report,
|
|
946
|
+
"record": cmd_record,
|
|
947
|
+
"regression": cmd_regression,
|
|
948
|
+
"pace": cmd_pace,
|
|
949
|
+
"phase-mark": cmd_phase_mark,
|
|
950
|
+
"phase-report": cmd_phase_report,
|
|
951
|
+
}[args.cmd](args, pricing)
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
if __name__ == "__main__":
|
|
955
|
+
raise SystemExit(main(sys.argv[1:]))
|