@ccoalm/ccl-skills 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +49 -0
- package/dist/assets/marketplace/.agents/plugins/marketplace.json +12 -0
- package/dist/assets/marketplace/.claude-plugin/marketplace.json +13 -0
- package/dist/assets/marketplace/marketplace-manifest.json +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/.claude-plugin/marketplace.json +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/.claude-plugin/plugin.json +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/.codex-plugin/plugin.json +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/.worktree-only +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/subagent-start.md +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/AGENTS.md +19 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-delegation-owner.sh +125 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-edit-isolation.sh +102 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/guard-merge-authorization.sh +1156 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/hooks.json +131 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/merge-authorization-prompt.sh +142 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/owner-dispatch-guard.sh +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/owner-dispatch-stop.sh +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/remind-post-merge-cleanup.sh +144 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/session-context.sh +87 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/session-start.sh +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/skill-extraction-gate-stop.sh +69 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/subagent-start.sh +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_guard_delegation_owner.sh +329 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_guard_edit_isolation.sh +322 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_guard_merge_authorization.sh +902 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_merge_authorization_prompt.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_remind_post_merge_cleanup.sh +121 -0
- package/dist/assets/marketplace/plugins/ccl-skills/hooks/test_session_start.sh +170 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/AGENTS.md +17 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/ccl-skills.ts +564 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-install-skills.md +14 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-update-skills.md +44 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-verify-skills.md +109 -0
- package/dist/assets/marketplace/plugins/ccl-skills/packages/opencode-plugin/commands/ccl-worktree-check.md +36 -0
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/AGENTS.md +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/README.md +276 -0
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.example.json +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/owner-dispatch.sh +1307 -0
- package/dist/assets/marketplace/plugins/ccl-skills/scripts/owner-dispatch/test.sh +941 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/agents-file-coverage-gate/SKILL.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/agents-file-coverage-gate/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +188 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/android-dev.md +92 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/flutter-dev.md +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/ios-dev.md +72 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/kotlin-multiplatform.md +93 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-platform-boundaries.md +77 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/mobile-quality-release.md +77 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/references/source-evidence-map.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +353 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +419 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +197 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/timeout-auth-and-capabilities.md +179 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +98 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/classify_envelope.py +93 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/classify_timeout_exit.sh +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +1438 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +324 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/concern_excerpt.py +295 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/egress_schema.py +214 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +642 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +181 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_review.sh +1165 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/opencode_review.sh +1190 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +946 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_opencode_review.py +474 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +1899 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_review_json.py +200 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +2845 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.sh +6 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/run_claude_capture.py +71 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_classify_envelope.sh +68 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +2311 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +1832 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_code_review_identity.sh +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_concern_excerpt.sh +245 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_egress_schema.sh +177 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +272 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +195 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_opencode_review_concurrency.sh +120 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_opencode_review_retry.sh +1005 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_opencode_review.sh +258 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +574 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_review_json.sh +349 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +434 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_order.sh +264 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh +2412 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/verify_native_skill_binding.py +123 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +153 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/references/diagnosis-playbook.md +54 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/references/prevention-routing.md +36 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/SKILL.md +69 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/feature-risk-router/references/security-review-gate.md +41 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/SKILL.md +165 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/api-security-boundaries.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/architecture-playbook.md +160 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/artifact-generation-architecture.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/audit-history-architecture.md +29 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/bulk-workflow-architecture.md +33 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/config-rule-routing-architecture.md +34 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/cross-cutting-concerns.md +72 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/data-modeling-and-migrations.md +79 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/data-platform-architecture.md +210 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/dependency-platform.md +105 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/developer-tooling-architecture.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/error-contract-architecture.md +36 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/event-driven-architecture.md +260 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/http-gateway-architecture.md +74 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/mq-consumer-architecture.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/multi-tenant-isolation.md +275 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/notification-architecture.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/ops-checklist.md +57 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/performance-capacity-architecture.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/protobuf-contract-architecture.md +119 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/redis-cache-coordination.md +93 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/release-runtime-readiness.md +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/replay-comparison-architecture.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/runtime-observability.md +94 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/service-scaffold.md +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/source-evidence-map.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-architecture/references/workflow-state-architecture.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +159 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/artifact-generation-patterns.md +37 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/audit-history-patterns.md +28 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/bulk-import-export-patterns.md +56 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/config-rule-routing-patterns.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/data-access-patterns.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/db-schema-and-dal-patterns.md +109 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/dependency-client-patterns.md +130 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/developer-tooling-patterns.md +70 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/domain-feature-patterns.md +78 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/engineering-patterns.md +119 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/error-contract-patterns.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/feature-playbook.md +61 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/http-gateway-client-patterns.md +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/mq-consumer-patterns.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/notification-patterns.md +42 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/observability-implementation-patterns.md +101 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/performance-capacity-patterns.md +44 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/protobuf-contract-patterns.md +72 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/public-api-integration-patterns.md +56 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/quality-and-testing-patterns.md +91 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/redis-cache-lock-patterns.md +123 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/release-ops-patterns.md +112 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/reliability-patterns.md +83 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/replay-comparison-patterns.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/scaffold-and-codegen.md +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/source-evidence-map.md +54 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/references/state-machine-task-patterns.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +80 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +117 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-approval-auto-reviewer.md +106 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-command-sandbox.md +441 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-context-freshness.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-credentials-auth.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-extensions-skills.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-file-edit-protocol.md +129 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-ide-integration.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-input-ingestion.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-instruction-composition.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-lifecycle-hooks.md +92 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-messaging.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-runtime-bootstrap.md +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-session-persistence.md +448 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-task-orchestration.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-tool-dispatch.md +123 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/agent-turn-lifecycle.md +131 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/inference-capacity-operations.md +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/llm-client-gateway.md +156 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/model-prompt-evaluation.md +146 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/references/retrieval-agent-safety.md +273 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +202 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/contracts-and-state.md +62 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/cross-stack-alignment.md +94 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/framework-choice.md +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/online-practice-uptake.md +56 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/platform-capabilities.md +91 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/product-page-checklist.md +40 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/qa-release.md +72 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/references/source-evidence-map.md +82 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +103 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/agents/openai.yaml +5 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/references/multi-agent-delegation-playbook.md +100 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/SKILL.md +70 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/references/public-data-acquisition.md +549 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/references/public-disclosure-channels.md +97 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/references/research-prompts.md +66 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/scripts/AGENTS.md +32 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-perspective-research/scripts/test-public-data-acquisition-recipes.sh +379 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +244 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/alerting-and-on-call.md +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/framework-middleware-checklist.md +142 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/infra-component-deployment.md +268 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/log-correlation-recipe.md +124 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/log-schema-canonical.md +208 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/metrics-conventions.md +105 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/obs-stack-architecture.md +107 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/sli-slo-design.md +95 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/references/source-register.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +303 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/canary-and-rollout-strategy.md +163 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/config-center-via-etcd.md +245 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/custom-control-plane-boundary.md +298 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/deploy-cli-concrete-recipe.md +312 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/deploy-pipeline.md +165 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/env-and-lane-matrix.md +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/lane-orchestration-control-plane.md +383 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/multi-region-and-cluster.md +135 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/promotion-gate-and-review.md +149 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/python-package-registry-release.md +462 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/rollback-playbook.md +123 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/secret-and-config-management.md +231 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/references/version-authority-and-deprecation.md +21 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +276 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/dual-sidecar-and-traffic-config-center.md +127 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/framework-middleware.md +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/grpc-authority-workaround.md +90 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/http-response-envelope-contract.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/mesh-architecture.md +127 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/multi-env-routing.md +192 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/protobuf-http-contract-signals.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/retry-timeout-circuit-breaker.md +124 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/rpc-framework-recipe.md +494 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-choice.md +113 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-migration-playbook.md +231 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/references/service-discovery-recipe.md +131 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +235 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/adr-convention.md +146 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/algorithm-launch-checklist.md +30 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/algorithm-launch-evaluation-report-template.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/algorithm-launch-execution-spec.md +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/algorithm-launch-sop.md +457 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/algorithm-launch-templates.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/artifact-egress-confidentiality.md +58 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/code-review-checklist.md +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/cross-repo-coordination.md +46 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/delivery-lifecycle.md +192 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +62 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-routing-and-readiness.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/diagnostic-spec-match-gate.md +36 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/dispatch-owner-skills.md +35 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/dormant-code-activation.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/existing-project-assessment-report.md +223 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/external-skill-augmentation.md +46 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/feature-deprecation-cascade.md +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/high-risk-resilience-gates.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/implementation-completeness-and-minimality.md +120 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/implementation-entry-reentry-gate.md +122 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/modular-monolith-heuristic.md +105 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +115 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/problem-resolution-and-learning.md +62 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/quality-attributes.md +112 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/quality-remediation-program.md +88 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +27 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +52 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/review-reception.md +34 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/shared-gate-artifact-classification.md +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/source-evidence-map.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/status-tracker-sync.md +77 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/sync-spec-repo-contract.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/verify-developer-experience.md +34 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/worktree-mechanics.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/scripts/AGENTS.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/scripts/check-agent-contract-coverage.sh +213 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/SKILL.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/agents/openai.yaml +9 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/analytics-visualization-interactions.md +206 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/behavioral-aesthetic-logic.md +108 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/complex-creation-interactions.md +194 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-execution-checklist.md +214 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-impl-naming-and-versioning.md +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-intake-and-acceptance.md +129 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/design-system-source-of-truth.md +97 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/external-ui-ux-quality-benchmarks.md +79 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/frontend-code-evidence-map.md +63 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/interaction-design-patterns.md +146 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/layout-recipes-and-screenshot-acceptance.md +250 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-project-token-consistency.md +237 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/multi-stack-strategy.md +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/operational-processing-workflows.md +237 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-mobile-patterns.md +324 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/platform-web-desktop-patterns.md +456 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-lifecycle-acceptance-and-iteration.md +114 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/product-surface-patterns.md +79 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/resource-management-interactions.md +113 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/scenario-community-patterns.md +133 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/source-map.md +130 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/tokens-and-components.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/trust-sensitive-ai-and-data-patterns.md +96 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-audit.md +106 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/ui-ux-design-development.md +176 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-ui-ux-design/references/visual-craft.md +111 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/SKILL.md +157 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/ai-service-integration-boundaries.md +57 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/api-contract-and-schema.md +62 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/api-security-boundaries.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/architecture-playbook.md +46 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/async-execution-model.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/background-jobs-and-scheduling.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/batch-and-pipeline-architecture.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/config-secrets-runtime.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/data-modeling-and-migrations.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/data-platform-architecture.md +211 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/event-driven-architecture.md +263 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/multi-tenant-isolation.md +281 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/observability-and-ops.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/packaging-runtime-readiness.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/redis-cache-coordination.md +41 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/reliability-and-error-contract.md +17 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/source-evidence-map.md +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-architecture/references/web-framework-boundaries.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/ai-service-wiring-patterns.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/async-and-worker-patterns.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/background-job-patterns.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/batch-and-artifact-patterns.md +13 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/dependency-client-patterns.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/error-handling-patterns.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/feature-playbook.md +43 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/observability-implementation-patterns.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/project-structure-and-tooling.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/public-api-security-patterns.md +52 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/redis-cache-lock-patterns.md +78 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/schema-and-validation-patterns.md +23 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/source-evidence-map.md +56 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/sqlalchemy-and-migrations-patterns.md +99 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/testing-and-quality-patterns.md +61 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/references/web-framework-patterns.md +35 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/SKILL.md +91 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/config-runtime-readback.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/mr-merge-authorization.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/post-release-env-reset.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/release-closeout-evidence.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/release-scope-confirmation.md +21 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/tag-and-prod-pipeline-gate.md +20 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/test-scope-prompt.md +24 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-coordination/references/watcher-discipline.md +14 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-doc-writer/SKILL.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-doc-writer/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-doc-writer/references/comment-safe-release-doc.md +19 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-doc-writer/references/release-evidence-workflow.md +23 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/release-doc-writer/references/release-testing-scope-section.md +15 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +87 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +130 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/references/prd-composition-contract.md +35 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/references/requirement-closure-contract.md +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/references/security-four-questions.md +38 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-intent/SKILL.md +91 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-intent/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +88 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +337 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/analysis-parse-fix-test-challenge-replay.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attribution-verification.md +69 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/bootstrap-slim-c3-obligation-table.md +112 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +45 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +162 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +507 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/evidence-card-template.md +51 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/example-domain-preselect.md +79 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +57 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-lifecycle-handoff.md +65 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +194 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/firing-point-placement.md +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +286 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +190 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/l0-l1-l2-routing.md +114 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/online-skill-review.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/parallel-stack-references-pattern.md +164 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/r0-leakage-audit.md +90 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/recurring-anti-patterns-checklist.md +320 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/review-feedback-mining.md +33 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/review-finding-standards.md +57 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/review-rubric.md +40 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +118 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/skill-listing-budget.md +19 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +254 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-to-skill-extraction.md +658 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/two-source-extraction-pattern.md +167 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +179 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-routing-map.md +51 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +180 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/AGENTS.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-ccl-skills.sh +1452 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-evidence-card-leak.sh +491 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-mr-target-freshness.sh +173 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-size-budget.sh +488 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/check-sync-pointers.sh +419 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-golden-trace.rb +197 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-health.rb +327 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +401 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing.rb +248 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/generic-r0-leak-scan.sh +282 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/governing-chain-diff.py +321 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +964 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/register-firing-path-resolution.rb +708 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/skill-behavior-eval.py +540 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/source-register-lifecycle.rb +51 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/source-register-pending-status.rb +55 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +829 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +1203 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_r0_status.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_register_pending_exclusion.sh +137 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +173 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_route_drift.sh +377 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_size_budget.sh +833 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_skill_catalog.sh +491 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_source_register_lifecycle.sh +114 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_mr_target_freshness.sh +261 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_sync_pointers.sh +538 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +154 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_grader_diagnostics.sh +190 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_surface_binding.sh +178 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_prose_target.sh +86 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_generic_r0_leak_scan.sh +131 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_git_identity_predicate_gate.sh +243 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_governing_chain_diff.sh +419 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_dateless_host.sh +120 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_resolution.sh +724 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_register_firing_path_wiring.sh +414 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_regression_runner_registration.sh +34 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_bank_integrity.sh +205 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_routing_pointer_integrity.sh +194 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_credential_cwd.sh +61 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_cross_refs.sh +111 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_skill_root_depth.sh +53 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate-skill.sh +257 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +98 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/references/input-state-machines.md +36 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/references/streaming-rich-output.md +130 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/references/terminal-side-channels.md +96 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/SKILL.md +408 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/AGENTS.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/bitable-setup.md +573 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/ci_templates/README.md +120 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/ci_templates/github-actions.yml +119 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/ci_templates/gitlab-ci.yml +76 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/ci_templates/jenkins.Jenkinsfile +106 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/classical-test-design-techniques.md +279 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/gen_report.py +2807 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/makefile-template.md +200 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/report-config-schema.md +272 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/run_pytestless.py +475 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/source-to-case-workflows.md +258 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-marker-conventions.md +316 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc-review-and-prioritization.md +145 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc_helpers/AGENTS.md +16 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc_helpers/tc.dart +129 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc_helpers/tc.go +197 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc_helpers/tc.py +135 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/tc_helpers/tc.ts +285 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/test_gen_report.py +2144 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/test-artifact-management/references/update-lifecycle.md +62 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +212 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/ci-fixtures-and-flake-control.md +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/client-runtime-test-matrices.md +50 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/data-and-workflow-testing.md +34 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/design-closed-contract-oracles.md +31 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/e2e-real-flow-testing.md +71 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/fitness-functions.md +240 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/integration-contract-testing.md +235 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/non-functional-specialized-scenarios.md +296 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/rd-testing-standard-template.md +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/run-killing-mutation-walk.md +43 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/scenario-testing.md +136 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/source-evidence-map.md +59 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/structured-tc-input-translation.md +67 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-code-authoring-patterns.md +392 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-data-and-determinism.md +39 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/test-topology-and-commands.md +92 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/unit-testing.md +46 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/vendored-contract-drift-checklist.md +64 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/references/verify-enforcement-mechanisms.md +18 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/AGENTS.md +17 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/client-terminal-ansi-check.py +140 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/client-terminal-ansi-check.test.sh +75 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/lang-basics-ast-check.py +170 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/lang-basics-ast-check.test.sh +87 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/lang-basics-go-check.go +198 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/lang-basics-go-check.test.sh +109 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/scripts/test_mutation_backup_recipe.sh +237 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +184 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/comment-safe-feishu.md +93 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/cross-model-co-review.md +3 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/delivery-face-closeout.md +60 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/doc-charter-first.md +17 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/session-vantage-leakage.md +58 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +126 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/complex-workspace-patterns.md +47 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/embedded-h5-in-host.md +87 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/react-architecture.md +194 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/source-evidence-map.md +60 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-quality-release.md +190 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/references/web-ui-quality.md +83 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/SKILL.md +179 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/agents/openai.yaml +4 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/references/shared-branch-rebase.md +25 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/AGENTS.md +23 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/test_worktree_status.sh +207 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/test_worktree_sweep.sh +481 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/worktree-status.sh +325 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/worktree-isolation/scripts/worktree-sweep.sh +245 -0
- package/dist/assets/release.json +2797 -0
- package/dist/claude-adapter.d.ts +9 -0
- package/dist/claude-adapter.js +240 -0
- package/dist/cli-worker.d.ts +1 -0
- package/dist/cli-worker.js +32 -0
- package/dist/cli.d.ts +22 -0
- package/dist/cli.js +214 -0
- package/dist/codex-host.d.ts +30 -0
- package/dist/codex-host.js +162 -0
- package/dist/fs-safe.d.ts +21 -0
- package/dist/fs-safe.js +241 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +1 -0
- package/dist/manifest.d.ts +8 -0
- package/dist/manifest.js +135 -0
- package/dist/opencode-adapter.d.ts +10 -0
- package/dist/opencode-adapter.js +416 -0
- package/dist/operations.d.ts +3 -0
- package/dist/operations.js +956 -0
- package/dist/paths.d.ts +20 -0
- package/dist/paths.js +4 -0
- package/dist/types.d.ts +58 -0
- package/dist/types.js +1 -0
- package/dist/unified.d.ts +4 -0
- package/dist/unified.js +64 -0
- package/dist/version.d.ts +2 -0
- package/dist/version.js +5 -0
- package/package.json +35 -0
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_gate.sh
ADDED
|
@@ -0,0 +1,2412 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Deterministic safety tests for packet freezing and cross-client gate behavior.
|
|
3
|
+
# Fake wrappers ensure no live reviewer CLI is invoked.
|
|
4
|
+
set -uo pipefail
|
|
5
|
+
|
|
6
|
+
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)"
|
|
7
|
+
WORK="$(mktemp -d "${TMPDIR:-/tmp}/review-gate-test.XXXXXX")"
|
|
8
|
+
WORK_REAL="$(cd "$WORK" && pwd -P)"
|
|
9
|
+
# The escaped-descendant case below starts a TERM-ignoring process that outlives
|
|
10
|
+
# its wrapper by design. It kills it inline, but an abort (outer timeout, CI
|
|
11
|
+
# cancellation) can land while that case is mid-flight and the in-memory handle is
|
|
12
|
+
# not assigned yet, so fall back to the pid the fixture recorded on disk.
|
|
13
|
+
cleanup_escaped_descendant() {
|
|
14
|
+
escaped_cleanup_pid="${escaped_child_trap_pid:-}"
|
|
15
|
+
[ -n "$escaped_cleanup_pid" ] ||
|
|
16
|
+
escaped_cleanup_pid="$(cat "$WORK/state/escaped_hang_child_pid" 2>/dev/null || true)"
|
|
17
|
+
case "$escaped_cleanup_pid" in
|
|
18
|
+
''|*[!0-9]*) return 0 ;;
|
|
19
|
+
esac
|
|
20
|
+
kill -KILL "$escaped_cleanup_pid" 2>/dev/null || true
|
|
21
|
+
}
|
|
22
|
+
trap 'cleanup_escaped_descendant; rm -rf "$WORK"' EXIT
|
|
23
|
+
fails=0
|
|
24
|
+
|
|
25
|
+
mkdir -p "$WORK/harness/scripts" "$WORK/state" "$WORK/repo" "$WORK/empty-registry"
|
|
26
|
+
cp "$DIR/review_gate.sh" "$DIR/review_gate.py" "$WORK/harness/scripts/"
|
|
27
|
+
cp "$DIR/../SKILL.md" "$WORK/harness/SKILL.md"
|
|
28
|
+
mkdir -p "$WORK/testing-strategy/references" "$WORK/python-service-dev" "$WORK/terminal-cli-dev" \
|
|
29
|
+
"$WORK/app-cross-platform-dev" "$WORK/go-microservice-dev" "$WORK/web-react-dev"
|
|
30
|
+
printf '%s\n' '# Testing Strategy' >"$WORK/testing-strategy/SKILL.md"
|
|
31
|
+
printf '%s\n' '# Focused Tests' >"$WORK/testing-strategy/references/focused-tests.md"
|
|
32
|
+
printf '%s\n' '# Python Service Dev' >"$WORK/python-service-dev/SKILL.md"
|
|
33
|
+
printf '%s\n' '# Terminal CLI Dev' >"$WORK/terminal-cli-dev/SKILL.md"
|
|
34
|
+
printf '%s\n' '# App Cross Platform Dev' >"$WORK/app-cross-platform-dev/SKILL.md"
|
|
35
|
+
printf '%s\n' '# Go Microservice Dev' >"$WORK/go-microservice-dev/SKILL.md"
|
|
36
|
+
printf '%s\n' '# Web React Dev' >"$WORK/web-react-dev/SKILL.md"
|
|
37
|
+
|
|
38
|
+
PYTHONPATH="$WORK/harness/scripts" python3 - "$WORK" <<'PY'
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
import sys
|
|
41
|
+
|
|
42
|
+
from review_gate import candidate_paths_from_packet, derive_owner_selection
|
|
43
|
+
|
|
44
|
+
registry = Path(sys.argv[1])
|
|
45
|
+
packet = (
|
|
46
|
+
b'diff --git "a/services/my file.py" "b/services/my file.py"\n'
|
|
47
|
+
b'--- "a/services/my file.py"\n'
|
|
48
|
+
b'+++ "b/services/my file.py"\n'
|
|
49
|
+
b'@@ -1 +1 @@\n'
|
|
50
|
+
b'--- a/skills/fake-owner/SKILL.md\n'
|
|
51
|
+
b'+++ b/skills/fake-owner/SKILL.md\n'
|
|
52
|
+
b'diff --git "a/services/caf\\303\\251.py" "b/services/caf\\303\\251.py"\n'
|
|
53
|
+
)
|
|
54
|
+
paths = candidate_paths_from_packet(packet)
|
|
55
|
+
assert "services/my file.py" in paths, paths
|
|
56
|
+
assert "services/caf\N{LATIN SMALL LETTER E WITH ACUTE}.py" in paths, paths
|
|
57
|
+
assert "skills/fake-owner/SKILL.md" not in paths, paths
|
|
58
|
+
assert not any(
|
|
59
|
+
row["skill"] == "README.md"
|
|
60
|
+
for row in derive_owner_selection(["skills/README.md"], registry)
|
|
61
|
+
)
|
|
62
|
+
assert not derive_owner_selection(["service.py", "tests/test_service.py"], registry / "empty-registry")
|
|
63
|
+
assert any(
|
|
64
|
+
row["skill"] == "testing-strategy"
|
|
65
|
+
for row in derive_owner_selection(
|
|
66
|
+
["skills/testing-strategy/references/focused-tests.md"], registry
|
|
67
|
+
)
|
|
68
|
+
)
|
|
69
|
+
for suffix in (".cjs", ".js", ".jsx", ".mjs", ".ts", ".tsx", ".vue"):
|
|
70
|
+
assert any(
|
|
71
|
+
row["skill"] == "web-react-dev"
|
|
72
|
+
for row in derive_owner_selection([f"src/component{suffix}"], registry)
|
|
73
|
+
), suffix
|
|
74
|
+
|
|
75
|
+
# A single-concern profile (challenge mode) has one slot, so the concern id
|
|
76
|
+
# carries no coverage information. A reviewer that names that slot after the
|
|
77
|
+
# supplied focus instead of echoing the profile id is still a complete answer;
|
|
78
|
+
# rejecting it loses a valid review round to model-output jitter.
|
|
79
|
+
from review_gate import normalize_concern_results
|
|
80
|
+
|
|
81
|
+
single = [{"id": "challenge_focus", "title": "focus"}]
|
|
82
|
+
conclusion = "The release path drops an outcome when the caller is cancelled mid-hook."
|
|
83
|
+
renamed = normalize_concern_results(
|
|
84
|
+
{"concern_results": [{"concern": "any_remaining_defect", "conclusion": conclusion}]},
|
|
85
|
+
single,
|
|
86
|
+
synthetic_slot=True,
|
|
87
|
+
)
|
|
88
|
+
assert renamed == [{"concern": "challenge_focus", "conclusion": conclusion}], renamed
|
|
89
|
+
echoed = normalize_concern_results(
|
|
90
|
+
{"concern_results": [{"concern": "challenge_focus", "conclusion": conclusion}]},
|
|
91
|
+
single,
|
|
92
|
+
synthetic_slot=True,
|
|
93
|
+
)
|
|
94
|
+
assert echoed == [{"concern": "challenge_focus", "conclusion": conclusion}], echoed
|
|
95
|
+
|
|
96
|
+
# The relaxation is scoped to one slot in and one slot out: it must not let a
|
|
97
|
+
# multi-concern profile pass with partial coverage, and must not accept two
|
|
98
|
+
# results against a single-concern profile.
|
|
99
|
+
multi = [{"id": "correctness", "title": "c"}, {"id": "safety", "title": "s"}]
|
|
100
|
+
# synthetic_slot=True on purpose. With the flag off this exercises only the
|
|
101
|
+
# pre-existing set comparison, so it would stay green if the branch ever stopped
|
|
102
|
+
# requiring exactly one concern — which is the escape hatch it is named for. The
|
|
103
|
+
# flag is what makes the assertion reach the new code.
|
|
104
|
+
assert (
|
|
105
|
+
normalize_concern_results(
|
|
106
|
+
{"concern_results": [{"concern": "correctness", "conclusion": conclusion}]},
|
|
107
|
+
multi,
|
|
108
|
+
synthetic_slot=True,
|
|
109
|
+
)
|
|
110
|
+
is None
|
|
111
|
+
)
|
|
112
|
+
assert (
|
|
113
|
+
normalize_concern_results(
|
|
114
|
+
{
|
|
115
|
+
"concern_results": [
|
|
116
|
+
{"concern": "a", "conclusion": conclusion},
|
|
117
|
+
{"concern": "b", "conclusion": conclusion},
|
|
118
|
+
]
|
|
119
|
+
},
|
|
120
|
+
single,
|
|
121
|
+
synthetic_slot=True,
|
|
122
|
+
)
|
|
123
|
+
is None
|
|
124
|
+
)
|
|
125
|
+
# Every substantive check still applies to the renamed single result, not just the
|
|
126
|
+
# length floor: the branch runs after the per-item loop, so a renamed result must
|
|
127
|
+
# not become a way past placeholder, over-length, malformed-value, or duplicate-id
|
|
128
|
+
# rejection.
|
|
129
|
+
assert (
|
|
130
|
+
normalize_concern_results(
|
|
131
|
+
{"concern_results": [{"concern": "whatever", "conclusion": "too short"}]},
|
|
132
|
+
single,
|
|
133
|
+
synthetic_slot=True,
|
|
134
|
+
)
|
|
135
|
+
is None
|
|
136
|
+
), "a renamed result must still fail the length floor"
|
|
137
|
+
assert (
|
|
138
|
+
normalize_concern_results(
|
|
139
|
+
{"concern_results": [{"concern": "whatever", "conclusion": "No issues were found here."}]},
|
|
140
|
+
single,
|
|
141
|
+
synthetic_slot=True,
|
|
142
|
+
)
|
|
143
|
+
is None
|
|
144
|
+
), "a renamed result must still fail placeholder rejection"
|
|
145
|
+
assert (
|
|
146
|
+
normalize_concern_results(
|
|
147
|
+
{"concern_results": [{"concern": "whatever", "conclusion": "x" * 2001}]},
|
|
148
|
+
single,
|
|
149
|
+
synthetic_slot=True,
|
|
150
|
+
)
|
|
151
|
+
is None
|
|
152
|
+
), "a renamed result must still fail the length ceiling"
|
|
153
|
+
assert (
|
|
154
|
+
normalize_concern_results(
|
|
155
|
+
{"concern_results": [{"concern": 17, "conclusion": conclusion}]},
|
|
156
|
+
single,
|
|
157
|
+
synthetic_slot=True,
|
|
158
|
+
)
|
|
159
|
+
is None
|
|
160
|
+
), "a non-string concern value must still be rejected"
|
|
161
|
+
assert (
|
|
162
|
+
normalize_concern_results(
|
|
163
|
+
{"concern_results": [{"concern": "a", "conclusion": conclusion}, {"concern": "a", "conclusion": conclusion}]},
|
|
164
|
+
single,
|
|
165
|
+
synthetic_slot=True,
|
|
166
|
+
)
|
|
167
|
+
is None
|
|
168
|
+
), "duplicate ids must be rejected in the loop, never collapsed into one slot"
|
|
169
|
+
for blank in ("", " "):
|
|
170
|
+
assert (
|
|
171
|
+
normalize_concern_results(
|
|
172
|
+
{"concern_results": [{"concern": blank, "conclusion": conclusion}]},
|
|
173
|
+
single,
|
|
174
|
+
synthetic_slot=True,
|
|
175
|
+
)
|
|
176
|
+
is None
|
|
177
|
+
), "an empty or whitespace-only concern id is structurally broken output, not a rename"
|
|
178
|
+
|
|
179
|
+
# Accepting any id made it an unbounded reviewer-controlled string that reaches
|
|
180
|
+
# the result verbatim through the attempt record. Bounded by shape, not by
|
|
181
|
+
# spelling: length and control characters, nothing about which words are allowed.
|
|
182
|
+
from review_gate import MAX_CONCERN_ID_LENGTH
|
|
183
|
+
|
|
184
|
+
for oversized in ("x" * (MAX_CONCERN_ID_LENGTH + 1), "y" * 100000):
|
|
185
|
+
assert (
|
|
186
|
+
normalize_concern_results(
|
|
187
|
+
{"concern_results": [{"concern": oversized, "conclusion": conclusion}]},
|
|
188
|
+
single,
|
|
189
|
+
synthetic_slot=True,
|
|
190
|
+
)
|
|
191
|
+
is None
|
|
192
|
+
), f"an id of {len(oversized)} characters must not reach the record"
|
|
193
|
+
assert (
|
|
194
|
+
normalize_concern_results(
|
|
195
|
+
{"concern_results": [{"concern": "x" * MAX_CONCERN_ID_LENGTH, "conclusion": conclusion}]},
|
|
196
|
+
single,
|
|
197
|
+
synthetic_slot=True,
|
|
198
|
+
)
|
|
199
|
+
== [{"concern": "challenge_focus", "conclusion": conclusion}]
|
|
200
|
+
), "the bound is a ceiling, not an off-by-one rejection at the limit"
|
|
201
|
+
# One per rejected category, not a list of remembered codepoints: C0 and DEL
|
|
202
|
+
# (Cc), NEL (Cc above C0 — the one that walked through the first attempt),
|
|
203
|
+
# zero-width space and BOM (Cf), a lone surrogate (Cs), private use (Co),
|
|
204
|
+
# unassigned (Cn), and the line and paragraph separators (Zl, Zp).
|
|
205
|
+
for control in (
|
|
206
|
+
"focus\x00slug", "focus\nslug", "focus\rslug", "focus\x1bslug", "focus\x7fslug",
|
|
207
|
+
"focus\x85slug", "focus\u200bslug", "focus\ufeffslug", "focus\ud800slug",
|
|
208
|
+
"focus\ue000slug", "focus\u0378slug", "focus\u2028slug", "focus\u2029slug",
|
|
209
|
+
):
|
|
210
|
+
assert (
|
|
211
|
+
normalize_concern_results(
|
|
212
|
+
{"concern_results": [{"concern": control, "conclusion": conclusion}]},
|
|
213
|
+
single,
|
|
214
|
+
synthetic_slot=True,
|
|
215
|
+
)
|
|
216
|
+
is None
|
|
217
|
+
), f"a control character in an id must be rejected: {control!r}"
|
|
218
|
+
assert (
|
|
219
|
+
normalize_concern_results(
|
|
220
|
+
{"concern_results": [{"concern": "a_focus_shaped_slug", "conclusion": conclusion}]},
|
|
221
|
+
single,
|
|
222
|
+
synthetic_slot=True,
|
|
223
|
+
)
|
|
224
|
+
== [{"concern": "challenge_focus", "conclusion": conclusion}]
|
|
225
|
+
), "the bound must not re-reject the ordinary renamed slug this change exists to accept"
|
|
226
|
+
# The id that broke the ceiling in the field, verbatim. A focus is a sentence,
|
|
227
|
+
# so its slug is that sentence: this one is 133 characters and the first
|
|
228
|
+
# ceiling was 128, which lost a complete verdict to invalid_model_output — the
|
|
229
|
+
# failure the relaxation exists to prevent, reintroduced by its own guard.
|
|
230
|
+
field_slug = (
|
|
231
|
+
"ways_a_registered_cleanup_could_be_skipped_run_twice_or_leak_a_resource_"
|
|
232
|
+
"when_the_request_is_cancelled_times_out_or_the_handler_raises"
|
|
233
|
+
)
|
|
234
|
+
assert len(field_slug) > 128, len(field_slug)
|
|
235
|
+
assert (
|
|
236
|
+
normalize_concern_results(
|
|
237
|
+
{"concern_results": [{"concern": field_slug, "conclusion": conclusion}]},
|
|
238
|
+
single,
|
|
239
|
+
synthetic_slot=True,
|
|
240
|
+
)
|
|
241
|
+
== [{"concern": "challenge_focus", "conclusion": conclusion}]
|
|
242
|
+
), "a focus-sentence slug must fit under the ceiling"
|
|
243
|
+
assert (
|
|
244
|
+
normalize_concern_results(
|
|
245
|
+
{"concern_results": [{"concern": "\u53d6\u6d88\u8def\u5f84", "conclusion": conclusion}]},
|
|
246
|
+
single,
|
|
247
|
+
synthetic_slot=True,
|
|
248
|
+
)
|
|
249
|
+
== [{"concern": "challenge_focus", "conclusion": conclusion}]
|
|
250
|
+
), "the rule bounds shape, not vocabulary: letters in any script are text"
|
|
251
|
+
# The positive half. A rejected-category list is still a denylist: a round
|
|
252
|
+
# reached past it with an id of nothing but combining marks — visually empty,
|
|
253
|
+
# every character structurally legal. An id must carry a letter or a digit.
|
|
254
|
+
for markup_only in ("\ufe0f", "\u0301", "\u0301\u0302", "---", "___", "..."):
|
|
255
|
+
assert (
|
|
256
|
+
normalize_concern_results(
|
|
257
|
+
{"concern_results": [{"concern": markup_only, "conclusion": conclusion}]},
|
|
258
|
+
single,
|
|
259
|
+
synthetic_slot=True,
|
|
260
|
+
)
|
|
261
|
+
is None
|
|
262
|
+
), f"an id with no letter or digit must be rejected: {markup_only!r}"
|
|
263
|
+
assert (
|
|
264
|
+
normalize_concern_results(
|
|
265
|
+
{"concern_results": [{"concern": "r\u0301esum\u0301e_slug", "conclusion": conclusion}]},
|
|
266
|
+
single,
|
|
267
|
+
synthetic_slot=True,
|
|
268
|
+
)
|
|
269
|
+
== [{"concern": "challenge_focus", "conclusion": conclusion}]
|
|
270
|
+
), "combining marks alongside letters are ordinary text, not a rejection"
|
|
271
|
+
# The bound applies to reviewer-chosen ids, not to ids the controller itself
|
|
272
|
+
# put in the profile. A profile carrying an unusual required id must still
|
|
273
|
+
# match a reply that echoes it exactly, or the gate rejects its own contract.
|
|
274
|
+
for controller_id in ("z" * (MAX_CONCERN_ID_LENGTH + 1), "---", "\u0301"):
|
|
275
|
+
assert normalize_concern_results(
|
|
276
|
+
{"concern_results": [{"concern": controller_id, "conclusion": conclusion}]},
|
|
277
|
+
[{"id": controller_id, "description": "a controller-owned concern"}],
|
|
278
|
+
synthetic_slot=False,
|
|
279
|
+
) == [{"concern": controller_id, "conclusion": conclusion}], (
|
|
280
|
+
f"an exact match on a controller-owned id is exempt from the shape bound: {controller_id!r}"
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
# The relaxation rests on challenge mode's slot naming no review dimension, so it
|
|
284
|
+
# applies only there. Inside that slot the id is accepted as-is, including forms
|
|
285
|
+
# that resemble a dimension name: distinguishing those would be a denylist over an
|
|
286
|
+
# open set of spellings, and it cannot detect the thing it appears to protect
|
|
287
|
+
# against, since a model answering the wrong question can still label it correctly.
|
|
288
|
+
for alias in ("any_remaining_defect", "challenge_focus", " safety ", "tests-evidence", "safety."):
|
|
289
|
+
assert normalize_concern_results(
|
|
290
|
+
{"concern_results": [{"concern": alias, "conclusion": conclusion}]}, single, synthetic_slot=True
|
|
291
|
+
) == [{"concern": "challenge_focus", "conclusion": conclusion}], alias
|
|
292
|
+
|
|
293
|
+
# The closed guard is the scoping: a one-item profile whose id IS a review
|
|
294
|
+
# dimension gets no relaxation, so a result answering a different dimension can
|
|
295
|
+
# never be recorded as coverage of it.
|
|
296
|
+
semantic_slot = [{"id": "safety", "title": "safety"}]
|
|
297
|
+
assert (
|
|
298
|
+
normalize_concern_results(
|
|
299
|
+
{"concern_results": [{"concern": "tests_evidence", "conclusion": conclusion}]},
|
|
300
|
+
semantic_slot,
|
|
301
|
+
synthetic_slot=False,
|
|
302
|
+
)
|
|
303
|
+
is None
|
|
304
|
+
), "a semantic one-item profile must not accept another dimension's result"
|
|
305
|
+
|
|
306
|
+
# The relaxation is told by the construction site, so the same one-item profile
|
|
307
|
+
# gets no relaxation when that fact is absent.
|
|
308
|
+
assert (
|
|
309
|
+
normalize_concern_results(
|
|
310
|
+
{"concern_results": [{"concern": "invented", "conclusion": conclusion}]},
|
|
311
|
+
single,
|
|
312
|
+
synthetic_slot=False,
|
|
313
|
+
)
|
|
314
|
+
is None
|
|
315
|
+
), "without the synthetic-slot fact a mismatched id is not an alias"
|
|
316
|
+
assert (
|
|
317
|
+
normalize_concern_results(
|
|
318
|
+
{"concern_results": [{"concern": "invented", "conclusion": conclusion}]},
|
|
319
|
+
single,
|
|
320
|
+
)
|
|
321
|
+
is None
|
|
322
|
+
), "the default must not enable the relaxation"
|
|
323
|
+
|
|
324
|
+
# The flag is a statement by the construction site, not something this function
|
|
325
|
+
# re-derives, so it is trusted here. What keeps it honest lives in
|
|
326
|
+
# freeze_review_profile: it is set only on the branch that builds the challenge
|
|
327
|
+
# slot, and cleared again when a high-risk run appends a second concern. The
|
|
328
|
+
# assertions below therefore pin the untold case, which is the one this function
|
|
329
|
+
# owns.
|
|
330
|
+
assert (
|
|
331
|
+
normalize_concern_results(
|
|
332
|
+
{"concern_results": [{"concern": "tests_evidence", "conclusion": conclusion}]},
|
|
333
|
+
semantic_slot,
|
|
334
|
+
synthetic_slot=False,
|
|
335
|
+
)
|
|
336
|
+
is None
|
|
337
|
+
), "a profile not declared synthetic gets strict matching"
|
|
338
|
+
assert normalize_concern_results(
|
|
339
|
+
{"concern_results": [{"concern": "safety", "conclusion": conclusion}]},
|
|
340
|
+
semantic_slot,
|
|
341
|
+
synthetic_slot=False,
|
|
342
|
+
) == [{"concern": "safety", "conclusion": conclusion}], "exact matching is unaffected"
|
|
343
|
+
|
|
344
|
+
# The told case: the rule that decides the flag, exercised directly so the
|
|
345
|
+
# construction site and this contract cannot drift apart.
|
|
346
|
+
from review_gate import builds_synthetic_slot
|
|
347
|
+
|
|
348
|
+
assert builds_synthetic_slot("challenge", False) is True
|
|
349
|
+
assert builds_synthetic_slot("challenge", True) is False, "a high-risk run appends a second concern"
|
|
350
|
+
for other in ("review", "complete", "consult"):
|
|
351
|
+
assert builds_synthetic_slot(other, False) is False, other
|
|
352
|
+
assert builds_synthetic_slot(other, True) is False, other
|
|
353
|
+
assert normalize_concern_results(
|
|
354
|
+
{"concern_results": [{"concern": "safety", "conclusion": conclusion}]},
|
|
355
|
+
semantic_slot,
|
|
356
|
+
synthetic_slot=False,
|
|
357
|
+
) == [{"concern": "safety", "conclusion": conclusion}], "an exact match still passes"
|
|
358
|
+
|
|
359
|
+
# What the id check never did: a challenge round objected that the relaxation
|
|
360
|
+
# lets an answer to some other question be recorded as satisfying the focus.
|
|
361
|
+
# It does, and so did strict matching — the required ids are stated in the
|
|
362
|
+
# packet, so echoing one costs a reviewer nothing and buys no topical
|
|
363
|
+
# guarantee. Pinned from the strict side, with synthetic_slot=False, so the
|
|
364
|
+
# demonstration cannot be waved off as a property of the new branch: an
|
|
365
|
+
# unrelated conclusion is accepted whenever the id matches. Nothing here reads
|
|
366
|
+
# the conclusion's subject, and no id rule could. The gate records the focus
|
|
367
|
+
# string beside the conclusion so the pairing stays auditable; judging whether
|
|
368
|
+
# the answer fits the question is the reader's, and it is unchanged by this
|
|
369
|
+
# candidate.
|
|
370
|
+
off_focus = "The retry loop reuses one idempotency key across attempts."
|
|
371
|
+
assert normalize_concern_results(
|
|
372
|
+
{"concern_results": [{"concern": "challenge_focus", "conclusion": off_focus}]},
|
|
373
|
+
[{"id": "challenge_focus", "description": "does cancellation drop an outcome?"}],
|
|
374
|
+
synthetic_slot=False,
|
|
375
|
+
) == [{"concern": "challenge_focus", "conclusion": off_focus}], (
|
|
376
|
+
"strict matching accepts an off-focus conclusion under the literal id"
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
assert any(
|
|
380
|
+
row["skill"] == "go-microservice-dev"
|
|
381
|
+
for row in derive_owner_selection(["cmd/service.go"], registry)
|
|
382
|
+
)
|
|
383
|
+
assert any(
|
|
384
|
+
row["skill"] == "app-cross-platform-dev"
|
|
385
|
+
for row in derive_owner_selection(["lib/screen.dart"], registry)
|
|
386
|
+
)
|
|
387
|
+
PY
|
|
388
|
+
if [ "$?" -ne 0 ]; then
|
|
389
|
+
printf 'FAIL - owner path extraction handles quoted paths and non-skill registry files\n' >&2
|
|
390
|
+
exit 1
|
|
391
|
+
fi
|
|
392
|
+
|
|
393
|
+
cat >"$WORK/harness/scripts/claude_review.sh" <<'CLAUDE_STUB'
|
|
394
|
+
#!/usr/bin/env bash
|
|
395
|
+
set -u
|
|
396
|
+
state="$REVIEW_GATE_TEST_STATE"
|
|
397
|
+
mode="$1"
|
|
398
|
+
shift
|
|
399
|
+
diff_file=""
|
|
400
|
+
profile_file=""
|
|
401
|
+
skill_registry_root=""
|
|
402
|
+
review_skills=""
|
|
403
|
+
host_attempted=0
|
|
404
|
+
while [ "$#" -gt 0 ]; do
|
|
405
|
+
case "$1" in
|
|
406
|
+
--diff-file) diff_file="$2"; shift 2 ;;
|
|
407
|
+
--review-profile-file) profile_file="$2"; shift 2 ;;
|
|
408
|
+
--skill-registry-root) skill_registry_root="$2"; shift 2 ;;
|
|
409
|
+
--review-skill) review_skills="${review_skills}${review_skills:+ }$2"; shift 2 ;;
|
|
410
|
+
--timeout) printf '%s\n' "$2" >"$state/claude_timeout"; shift 2 ;;
|
|
411
|
+
--host-remediation-attempted) host_attempted=1; shift ;;
|
|
412
|
+
*) shift ;;
|
|
413
|
+
esac
|
|
414
|
+
done
|
|
415
|
+
printf '%s\n' "$skill_registry_root" >"$state/claude_skill_registry_root"
|
|
416
|
+
printf '%s\n' "$review_skills" >"$state/claude_review_skills"
|
|
417
|
+
printf '%s\n' claude >>"$state/client_sequence"
|
|
418
|
+
printf '%s\n' "$mode" >"$state/claude_mode"
|
|
419
|
+
shasum -a 256 "$diff_file" | awk '{print $1}' >"$state/claude_hash"
|
|
420
|
+
cp "$diff_file" "$state/claude_packet"
|
|
421
|
+
shasum -a 256 "$profile_file" | awk '{print $1}' >"$state/claude_profile_hash"
|
|
422
|
+
cp "$profile_file" "$state/claude_profile"
|
|
423
|
+
behavior="$(cat "$state/claude_behavior")"
|
|
424
|
+
concern_results="$(python3 - "$profile_file" "$behavior" <<'PY'
|
|
425
|
+
import json
|
|
426
|
+
import sys
|
|
427
|
+
|
|
428
|
+
profile = json.load(open(sys.argv[1], encoding="utf-8"))
|
|
429
|
+
print(json.dumps([
|
|
430
|
+
{
|
|
431
|
+
"concern": item["id"],
|
|
432
|
+
"conclusion": (
|
|
433
|
+
"no issues were found here"
|
|
434
|
+
if sys.argv[2] == "placeholder_coverage"
|
|
435
|
+
else f"Checked {item['id']} against the frozen candidate."
|
|
436
|
+
),
|
|
437
|
+
}
|
|
438
|
+
for item in profile["required_concerns"]
|
|
439
|
+
], separators=(",", ":")))
|
|
440
|
+
PY
|
|
441
|
+
)"
|
|
442
|
+
native_skill_binding="not_requested"
|
|
443
|
+
[ -z "$review_skills" ] || native_skill_binding="established"
|
|
444
|
+
case "$behavior" in
|
|
445
|
+
passed) printf '{"mode":"%s","native_skill_binding":"%s","concern_results":%s,"findings":[]}\n' "$mode" "$native_skill_binding" "$concern_results"; exit 0 ;;
|
|
446
|
+
missing_binding) printf '{"mode":"%s","concern_results":%s,"findings":[]}\n' "$mode" "$concern_results"; exit 0 ;;
|
|
447
|
+
passed_slow)
|
|
448
|
+
sleep 5
|
|
449
|
+
printf '{"mode":"%s","native_skill_binding":"%s","concern_results":%s,"findings":[]}\n' "$mode" "$native_skill_binding" "$concern_results"
|
|
450
|
+
exit 0
|
|
451
|
+
;;
|
|
452
|
+
placeholder_coverage) printf '{"mode":"%s","native_skill_binding":"%s","concern_results":%s,"findings":[]}\n' "$mode" "$native_skill_binding" "$concern_results"; exit 0 ;;
|
|
453
|
+
missing_coverage) printf '{"mode":"%s","findings":[]}\n' "$mode"; exit 0 ;;
|
|
454
|
+
partial_coverage) printf '{"mode":"%s","concern_results":[{"concern":"correctness","conclusion":"Only one concern was checked."}],"findings":[]}\n' "$mode"; exit 0 ;;
|
|
455
|
+
renamed_concern) printf '{"mode":"%s","native_skill_binding":"%s","concern_results":[{"concern":"a_focus_shaped_slug","conclusion":"The release path drops an outcome when the caller is cancelled."}],"findings":[]}\n' "$mode" "$native_skill_binding"; exit 0 ;;
|
|
456
|
+
oversized_concern_id)
|
|
457
|
+
long_id="$(python3 -c 'print("x" * 600)')"
|
|
458
|
+
printf '{"mode":"%s","native_skill_binding":"%s","concern_results":[{"concern":"%s","conclusion":"The release path drops an outcome when the caller is cancelled."}],"findings":[]}\n' "$mode" "$native_skill_binding" "$long_id"
|
|
459
|
+
exit 0 ;;
|
|
460
|
+
extra_coverage)
|
|
461
|
+
extra_results="${concern_results%]}, {\"concern\":\"invented\",\"conclusion\":\"An unrequested concern was injected.\"}]"
|
|
462
|
+
printf '{"mode":"%s","concern_results":%s,"findings":[]}\n' "$mode" "$extra_results"
|
|
463
|
+
exit 0 ;;
|
|
464
|
+
findings) printf '{"mode":"%s","native_skill_binding":"%s","concern_results":%s,"findings":[{"severity":"P1","file":"x","line":1,"failure_path":"breaks","smallest_fix":"fix"}]}\n' "$mode" "$native_skill_binding" "$concern_results"; exit 0 ;;
|
|
465
|
+
quota) printf '{"mode":"%s","status":"inconclusive","reason":"quota","reason_code":"quota","fallback_eligible":true,"next_action":"fallback"}\n' "$mode"; exit 2 ;;
|
|
466
|
+
quota_slow)
|
|
467
|
+
sleep 2
|
|
468
|
+
printf '{"mode":"%s","status":"inconclusive","reason":"quota","reason_code":"quota","fallback_eligible":true,"next_action":"fallback"}\n' "$mode"
|
|
469
|
+
exit 2
|
|
470
|
+
;;
|
|
471
|
+
hang)
|
|
472
|
+
trap '' TERM
|
|
473
|
+
(
|
|
474
|
+
trap '' TERM
|
|
475
|
+
printf '%s\n' "$BASHPID" >"$state/hang_child_pid"
|
|
476
|
+
while :; do sleep 1; done
|
|
477
|
+
) &
|
|
478
|
+
wait "$!"
|
|
479
|
+
;;
|
|
480
|
+
escaped_hang)
|
|
481
|
+
trap '' TERM
|
|
482
|
+
printf '%s\n' "$$" >"$state/escaped_hang_wrapper_pid"
|
|
483
|
+
# Sleeps far longer than any plausible run of this case: the test proves the
|
|
484
|
+
# envelope did not wait for these pipes by observing that this descendant is
|
|
485
|
+
# STILL ALIVE when the gate returns, so its lifetime must not be a deadline
|
|
486
|
+
# the runner can race. The caller kills it as soon as it has read that.
|
|
487
|
+
# Heartbeat rather than one opaque sleep: when this descendant turns up missing,
|
|
488
|
+
# "gone" alone cannot separate a controller that killed it from one that never
|
|
489
|
+
# let it detach. The beat file records the post-setsid identity once and then a
|
|
490
|
+
# liveness stamp, so the failure diagnostic can say whether it ever escaped its
|
|
491
|
+
# wrapper's session and how long it survived after that.
|
|
492
|
+
python3 - "$state/escaped_hang_child_pid" "$state/escaped_hang_child_beat" "$state/escaped_gate_returned" <<'PY' &
|
|
493
|
+
import os
|
|
494
|
+
from pathlib import Path
|
|
495
|
+
import signal
|
|
496
|
+
import sys
|
|
497
|
+
import time
|
|
498
|
+
|
|
499
|
+
os.setsid()
|
|
500
|
+
signal.signal(signal.SIGTERM, signal.SIG_IGN)
|
|
501
|
+
pid_path = Path(sys.argv[1])
|
|
502
|
+
beat_path = Path(sys.argv[2])
|
|
503
|
+
returned_path = Path(sys.argv[3])
|
|
504
|
+
started = time.monotonic()
|
|
505
|
+
witnessed = 0
|
|
506
|
+
# Deliberately not `ppid=`: it carries `pid=` as a substring, and the shell-side
|
|
507
|
+
# suffix-strip parse would then read the PARENT's pid and compare the wrong number.
|
|
508
|
+
identity = "pid=%d sid=%d pgid=%d parent=%d" % (
|
|
509
|
+
os.getpid(),
|
|
510
|
+
os.getsid(0),
|
|
511
|
+
os.getpgrp(),
|
|
512
|
+
os.getppid(),
|
|
513
|
+
)
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
def beat():
|
|
517
|
+
# witnessed= is the load-bearing field, and it is deliberately an observation
|
|
518
|
+
# made BY this descendant rather than a clock comparison made about it: the
|
|
519
|
+
# case wants "the pipes were still held when the envelope came back", and the
|
|
520
|
+
# only party that can testify to that without a shared clock is the holder
|
|
521
|
+
# itself, while it is still running. The caller drops the marker the instant
|
|
522
|
+
# the gate returns; seeing it means this process — which still holds the
|
|
523
|
+
# inherited reviewer pipes, since it never closes them — outlived that return.
|
|
524
|
+
beat_path.write_text(
|
|
525
|
+
"detached %s at=%d alive=%.1f witnessed=%d\n"
|
|
526
|
+
% (identity, int(time.time()), time.monotonic() - started, witnessed),
|
|
527
|
+
encoding="utf-8",
|
|
528
|
+
)
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
beat()
|
|
532
|
+
pid_path.write_text(str(os.getpid()), encoding="utf-8")
|
|
533
|
+
while time.monotonic() - started < 300:
|
|
534
|
+
if not witnessed and returned_path.exists():
|
|
535
|
+
witnessed = 1
|
|
536
|
+
beat()
|
|
537
|
+
# Short enough that the window between the gate returning and this descendant
|
|
538
|
+
# noticing cannot be mistaken for it having died first.
|
|
539
|
+
time.sleep(0.05)
|
|
540
|
+
PY
|
|
541
|
+
wait "$!"
|
|
542
|
+
;;
|
|
543
|
+
quota_mutate)
|
|
544
|
+
printf '\nmutated-by-primary\n' >>"$diff_file"
|
|
545
|
+
printf '{"mode":"%s","status":"inconclusive","reason":"quota","reason_code":"quota","fallback_eligible":true,"next_action":"fallback"}\n' "$mode"
|
|
546
|
+
exit 2 ;;
|
|
547
|
+
auth)
|
|
548
|
+
if [ "$host_attempted" = 1 ]; then
|
|
549
|
+
printf '{"mode":"%s","status":"inconclusive","reason":"auth after host retry","reason_code":"auth_unavailable_after_host_retry","fallback_eligible":true,"next_action":"fallback"}\n' "$mode"
|
|
550
|
+
else
|
|
551
|
+
printf '{"mode":"%s","status":"inconclusive","reason":"auth path","reason_code":"auth_path_unavailable","fallback_eligible":false,"next_action":"host_retry"}\n' "$mode"
|
|
552
|
+
fi
|
|
553
|
+
exit 2 ;;
|
|
554
|
+
legacy) printf '{"mode":"%s","status":"inconclusive","reason":"legacy result"}\n' "$mode"; exit 2 ;;
|
|
555
|
+
esac
|
|
556
|
+
exit 2
|
|
557
|
+
CLAUDE_STUB
|
|
558
|
+
chmod +x "$WORK/harness/scripts/claude_review.sh"
|
|
559
|
+
|
|
560
|
+
cat >"$WORK/harness/scripts/candidate_stub.sh" <<'CLIENT_STUB'
|
|
561
|
+
#!/usr/bin/env bash
|
|
562
|
+
set -u
|
|
563
|
+
state="$REVIEW_GATE_TEST_STATE"
|
|
564
|
+
client="$(basename "$0" _review.sh)"
|
|
565
|
+
mode=review
|
|
566
|
+
diff_file=""
|
|
567
|
+
profile_file=""
|
|
568
|
+
skill_registry_root=""
|
|
569
|
+
review_skills=""
|
|
570
|
+
host_attempted=0
|
|
571
|
+
while [ "$#" -gt 0 ]; do
|
|
572
|
+
case "$1" in
|
|
573
|
+
--mode) mode="$2"; shift 2 ;;
|
|
574
|
+
--diff-file) diff_file="$2"; shift 2 ;;
|
|
575
|
+
--review-profile-file) profile_file="$2"; shift 2 ;;
|
|
576
|
+
--skill-registry-root) skill_registry_root="$2"; shift 2 ;;
|
|
577
|
+
--review-skill) review_skills="${review_skills}${review_skills:+ }$2"; shift 2 ;;
|
|
578
|
+
--timeout) printf '%s\n' "$2" >"$state/${client}_timeout"; shift 2 ;;
|
|
579
|
+
--host-remediation-attempted) host_attempted=1; shift ;;
|
|
580
|
+
*) shift ;;
|
|
581
|
+
esac
|
|
582
|
+
done
|
|
583
|
+
printf '%s\n' "$skill_registry_root" >"$state/${client}_skill_registry_root"
|
|
584
|
+
printf '%s\n' "$review_skills" >"$state/${client}_review_skills"
|
|
585
|
+
printf '%s\n' "$client" >>"$state/client_sequence"
|
|
586
|
+
printf '%s\n' "$mode" >"$state/${client}_mode"
|
|
587
|
+
shasum -a 256 "$diff_file" | awk '{print $1}' >"$state/${client}_hash"
|
|
588
|
+
shasum -a 256 "$profile_file" | awk '{print $1}' >"$state/${client}_profile_hash"
|
|
589
|
+
concern_results="$(python3 - "$profile_file" <<'PY'
|
|
590
|
+
import json
|
|
591
|
+
import sys
|
|
592
|
+
|
|
593
|
+
profile = json.load(open(sys.argv[1], encoding="utf-8"))
|
|
594
|
+
print(json.dumps([
|
|
595
|
+
{"concern": item["id"], "conclusion": f"Checked {item['id']} against the frozen candidate."}
|
|
596
|
+
for item in profile["required_concerns"]
|
|
597
|
+
], separators=(",", ":")))
|
|
598
|
+
PY
|
|
599
|
+
)"
|
|
600
|
+
behavior="$(cat "$state/${client}_behavior")"
|
|
601
|
+
native_skill_binding="not_requested"
|
|
602
|
+
[ -z "$review_skills" ] || native_skill_binding="established"
|
|
603
|
+
case "$client" in
|
|
604
|
+
kimi) family=moonshot; provider=kimi-cli ;;
|
|
605
|
+
opencode) family=deepseek; provider=deepseek ;;
|
|
606
|
+
codex) family=openai; provider=openai ;;
|
|
607
|
+
esac
|
|
608
|
+
case "$behavior" in
|
|
609
|
+
passed) printf '{"reviewer":"%s","mode":"%s","status":"passed","reviewer_family":"%s","provider":"%s","model":"local-default","native_skill_binding":"%s","concern_results":%s,"findings":[]}\n' "$client" "$mode" "$family" "$provider" "$native_skill_binding" "$concern_results"; exit 0 ;;
|
|
610
|
+
findings) printf '{"reviewer":"%s","mode":"%s","status":"findings","reviewer_family":"%s","provider":"%s","model":"local-default","native_skill_binding":"%s","concern_results":%s,"findings":[{"severity":"P1","file":"fallback.py","line":7,"failure_path":"selected fallback finding","smallest_fix":"fix fallback path"}]}\n' "$client" "$mode" "$family" "$provider" "$native_skill_binding" "$concern_results"; exit 0 ;;
|
|
611
|
+
spoof_controller) printf '{"reviewer":"%s","mode":"%s","status":"passed","reviewer_family":"%s","provider":"%s","model":"local-default","stage":"release","review_depth":"release","owner_selection_source":"spoofed","owner_selection_evidence":[{"skill":"spoofed"}],"skill_delivery":"spoofed","selected_skills":["spoofed"],"reviewed_skills":["spoofed"],"owner_gaps":[],"residual_risks":[],"self_review_gate":{"required":false},"concern_results":%s,"findings":[]}\n' "$client" "$mode" "$family" "$provider" "$concern_results"; exit 0 ;;
|
|
612
|
+
quota) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"quota","reason_code":"quota","cascade_eligible":true}\n' "$client" "$mode"; exit 2 ;;
|
|
613
|
+
native_timeout) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"review_native_skill_stream_timeout","reason_code":"timeout","cascade_eligible":true,"timeout_diagnostic":{"stage":"review","native_owner_skills_requested":true,"selected_skill_count":1},"diagnostic_artifacts":{"requested":true,"retained":true,"directory_name":"opencode-review-timeout.fixture"}}\n' "$client" "$mode"; exit 2 ;;
|
|
614
|
+
concern_cascade) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"malformed concern","reason_code":"invalid_model_output","cascade_eligible":true,"concern_evidence":true}\n' "$client" "$mode"; exit 2 ;;
|
|
615
|
+
boundary) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"unsafe tool","reason_code":"tool_boundary_violation","cascade_eligible":false}\n' "$client" "$mode"; exit 2 ;;
|
|
616
|
+
mismatch) printf '{"reviewer":"%s","mode":"challenge","status":"passed","reviewer_family":"%s","provider":"%s","model":"local-default","findings":[]}\n' "$client" "$family" "$provider"; exit 0 ;;
|
|
617
|
+
unavailable) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"missing","reason_code":"client_unavailable","cascade_eligible":true}\n' "$client" "$mode"; exit 2 ;;
|
|
618
|
+
oversize_inline) printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"packet_too_large_for_inline","reason_code":"capability_missing","cascade_eligible":true}\n' "$client" "$mode"; exit 2 ;;
|
|
619
|
+
hang)
|
|
620
|
+
trap '' TERM
|
|
621
|
+
while :; do sleep 1; done
|
|
622
|
+
;;
|
|
623
|
+
auth)
|
|
624
|
+
if [ "$host_attempted" = 1 ]; then
|
|
625
|
+
printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"auth after host retry","reason_code":"auth_unavailable_after_host_retry","cascade_eligible":true,"next_action":"fallback"}\n' "$client" "$mode"
|
|
626
|
+
else
|
|
627
|
+
printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"auth path","reason_code":"auth_path_unavailable","cascade_eligible":false,"next_action":"host_retry"}\n' "$client" "$mode"
|
|
628
|
+
fi
|
|
629
|
+
exit 2 ;;
|
|
630
|
+
host_path)
|
|
631
|
+
if [ "$host_attempted" = 1 ]; then
|
|
632
|
+
printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"host path after retry","reason_code":"host_path_unavailable_after_host_retry","cascade_eligible":true}\n' "$client" "$mode"
|
|
633
|
+
else
|
|
634
|
+
printf '{"reviewer":"%s","mode":"%s","status":"inconclusive","reason":"host path","reason_code":"host_path_unavailable","cascade_eligible":false}\n' "$client" "$mode"
|
|
635
|
+
fi
|
|
636
|
+
exit 2 ;;
|
|
637
|
+
esac
|
|
638
|
+
exit 2
|
|
639
|
+
CLIENT_STUB
|
|
640
|
+
chmod +x "$WORK/harness/scripts/candidate_stub.sh"
|
|
641
|
+
for client in kimi opencode codex; do
|
|
642
|
+
cp "$WORK/harness/scripts/candidate_stub.sh" "$WORK/harness/scripts/${client}_review.sh"
|
|
643
|
+
done
|
|
644
|
+
|
|
645
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
|
|
646
|
+
printf 'diff --git a/c b/c\n--- a/c\n+++ b/c\n@@ -1 +1 @@\n-x\n+aws_key = "AKIAIOSFODNN7EXAMPLE"\n' >"$WORK/secret-diff.patch"
|
|
647
|
+
awk 'BEGIN { for (i = 0; i < 180000; i++) printf "x" }' >"$WORK/large-diff.patch"
|
|
648
|
+
printf 'diff --git a/x b/x\0binary-tail' >"$WORK/nul-diff.patch"
|
|
649
|
+
cat >"$WORK/review-plan.json" <<'JSON'
|
|
650
|
+
{
|
|
651
|
+
"intent": "Preserve the review gate safety contract while adding staged review.",
|
|
652
|
+
"acceptance": ["The selected stage controls the common review focus."],
|
|
653
|
+
"self_review": [
|
|
654
|
+
{"concern": "correctness", "conclusion": "The candidate preserves current correctness invariants.", "evidence_refs": ["e1"]},
|
|
655
|
+
{"concern": "safety", "conclusion": "The candidate preserves packet, tool, and egress boundaries.", "evidence_refs": ["e1"]},
|
|
656
|
+
{"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain fail closed.", "evidence_refs": ["e1"]},
|
|
657
|
+
{"concern": "tests_evidence", "conclusion": "Focused deterministic contract tests cover the change.", "evidence_refs": ["e1"]},
|
|
658
|
+
{"concern": "compatibility", "conclusion": "Existing provider routing remains backward compatible.", "evidence_refs": ["e1"]},
|
|
659
|
+
{"concern": "rollout_rollback", "conclusion": "The local CLI change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
660
|
+
{"concern": "observability_operations", "conclusion": "The JSON envelope exposes stage and depth for diagnosis.", "evidence_refs": ["e1"]}
|
|
661
|
+
],
|
|
662
|
+
"evidence": [
|
|
663
|
+
{"id": "e1", "result": "Deterministic fake-wrapper contract fixture."}
|
|
664
|
+
]
|
|
665
|
+
}
|
|
666
|
+
JSON
|
|
667
|
+
python3 - "$WORK/review-plan.json" "$WORK/owner-review-plan.json" <<'PY'
|
|
668
|
+
import json
|
|
669
|
+
from pathlib import Path
|
|
670
|
+
import sys
|
|
671
|
+
|
|
672
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
673
|
+
for row in plan["self_review"]:
|
|
674
|
+
row["skill"] = (
|
|
675
|
+
"testing-strategy" if row["concern"] == "tests_evidence" else "code-review"
|
|
676
|
+
)
|
|
677
|
+
Path(sys.argv[2]).write_text(json.dumps(plan, separators=(",", ":")))
|
|
678
|
+
PY
|
|
679
|
+
sed 's/testing-strategy/python-service-dev/' \
|
|
680
|
+
"$WORK/owner-review-plan.json" >"$WORK/python-owner-review-plan.json"
|
|
681
|
+
sed 's/testing-strategy/terminal-cli-dev/' \
|
|
682
|
+
"$WORK/owner-review-plan.json" >"$WORK/shell-owner-review-plan.json"
|
|
683
|
+
python3 - "$WORK/owner-review-plan.json" "$WORK/language-owner-review-plan.json" <<'PY'
|
|
684
|
+
import json
|
|
685
|
+
from pathlib import Path
|
|
686
|
+
import sys
|
|
687
|
+
|
|
688
|
+
plan = json.loads(Path(sys.argv[1]).read_text())
|
|
689
|
+
for row in plan["self_review"]:
|
|
690
|
+
row["skill"] = "code-review"
|
|
691
|
+
for row, skill in zip(
|
|
692
|
+
plan["self_review"],
|
|
693
|
+
("app-cross-platform-dev", "go-microservice-dev", "web-react-dev"),
|
|
694
|
+
):
|
|
695
|
+
row["skill"] = skill
|
|
696
|
+
Path(sys.argv[2]).write_text(json.dumps(plan, separators=(",", ":")))
|
|
697
|
+
PY
|
|
698
|
+
sed 's/testing-strategy/missing-owner/' \
|
|
699
|
+
"$WORK/owner-review-plan.json" >"$WORK/missing-owner-review-plan.json"
|
|
700
|
+
mkdir -p "$WORK/missing-entrypoint-owner"
|
|
701
|
+
sed 's/testing-strategy/missing-entrypoint-owner/' \
|
|
702
|
+
"$WORK/owner-review-plan.json" >"$WORK/missing-entrypoint-owner-review-plan.json"
|
|
703
|
+
printf '%s\n' 'not a skill package' >"$WORK/file-owner"
|
|
704
|
+
sed 's/testing-strategy/file-owner/' \
|
|
705
|
+
"$WORK/owner-review-plan.json" >"$WORK/file-owner-review-plan.json"
|
|
706
|
+
sed 's/testing-strategy/testing_strategy/' \
|
|
707
|
+
"$WORK/owner-review-plan.json" >"$WORK/invalid-owner-review-plan.json"
|
|
708
|
+
python3 - "$WORK/owner-review-plan.json" \
|
|
709
|
+
"$WORK/non-string-owner-review-plan.json" \
|
|
710
|
+
"$WORK/empty-owner-review-plan.json" <<'PY'
|
|
711
|
+
import json
|
|
712
|
+
from pathlib import Path
|
|
713
|
+
import sys
|
|
714
|
+
|
|
715
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
716
|
+
for value, target in ((None, sys.argv[2]), ("", sys.argv[3])):
|
|
717
|
+
plan = json.loads(json.dumps(source))
|
|
718
|
+
plan["self_review"][0]["skill"] = value
|
|
719
|
+
Path(target).write_text(json.dumps(plan, separators=(",", ":")))
|
|
720
|
+
PY
|
|
721
|
+
mkdir -p "$WORK/linked-owner-real"
|
|
722
|
+
printf '%s\n' '# Linked Owner Target' >"$WORK/linked-owner-real/SKILL.md"
|
|
723
|
+
ln -s "$WORK/linked-owner-real" "$WORK/linked-owner"
|
|
724
|
+
sed 's/testing-strategy/linked-owner/' \
|
|
725
|
+
"$WORK/owner-review-plan.json" >"$WORK/linked-owner-review-plan.json"
|
|
726
|
+
sed 's/Preserve the review gate safety contract while adding staged review\./Preserve the review gate safety contract while changing staged review evidence./' \
|
|
727
|
+
"$WORK/review-plan.json" >"$WORK/changed-review-plan.json"
|
|
728
|
+
sed 's/The selected stage controls the common review focus\./The selected stage controls a changed completion contract./' \
|
|
729
|
+
"$WORK/review-plan.json" >"$WORK/changed-acceptance-plan.json"
|
|
730
|
+
cat >"$WORK/incomplete-plan.json" <<'JSON'
|
|
731
|
+
{"intent":"Add stages","acceptance":["Stage is recorded"],"self_review":[],"evidence":[]}
|
|
732
|
+
JSON
|
|
733
|
+
cat >"$WORK/placeholder-plan.json" <<'JSON'
|
|
734
|
+
{
|
|
735
|
+
"intent": "Reject self-review filler before invoking a provider.",
|
|
736
|
+
"acceptance": ["Every self-review conclusion carries concern-specific reasoning."],
|
|
737
|
+
"self_review": [
|
|
738
|
+
{"concern": "correctness", "conclusion": "Verified.", "evidence_refs": ["e1"]},
|
|
739
|
+
{"concern": "safety", "conclusion": "Packet and tool boundaries remain fail closed.", "evidence_refs": ["e1"]},
|
|
740
|
+
{"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain terminal.", "evidence_refs": ["e1"]},
|
|
741
|
+
{"concern": "tests_evidence", "conclusion": "A focused regression proves filler is rejected.", "evidence_refs": ["e1"]},
|
|
742
|
+
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]}
|
|
743
|
+
],
|
|
744
|
+
"evidence": [{"id": "e1", "result": "Deterministic placeholder-validation fixture."}]
|
|
745
|
+
}
|
|
746
|
+
JSON
|
|
747
|
+
sed 's/Deterministic fake-wrapper contract fixture\./Verified./' \
|
|
748
|
+
"$WORK/review-plan.json" >"$WORK/placeholder-evidence-plan.json"
|
|
749
|
+
sed 's/The candidate preserves current correctness invariants\./No problems found./' \
|
|
750
|
+
"$WORK/review-plan.json" >"$WORK/low-information-self-review-plan.json"
|
|
751
|
+
sed 's/Deterministic fake-wrapper contract fixture\./Checks all passed./' \
|
|
752
|
+
"$WORK/review-plan.json" >"$WORK/low-information-evidence-plan.json"
|
|
753
|
+
sed 's/Deterministic fake-wrapper contract fixture\./Live token AKIAIOSFODNN7EXAMPLE captured during setup./' \
|
|
754
|
+
"$WORK/review-plan.json" >"$WORK/secret-plan.json"
|
|
755
|
+
cat >"$WORK/high-risk-plan.json" <<'JSON'
|
|
756
|
+
{
|
|
757
|
+
"intent": "Preserve a shared gate while changing its staged review contract.",
|
|
758
|
+
"acceptance": ["High-risk routing raises review depth and checks bypass paths."],
|
|
759
|
+
"self_review": [
|
|
760
|
+
{"concern": "correctness", "conclusion": "The gate still selects a valid independent result.", "evidence_refs": ["e1"]},
|
|
761
|
+
{"concern": "safety", "conclusion": "Packet, tool, and egress boundaries remain fail closed.", "evidence_refs": ["e1"]},
|
|
762
|
+
{"concern": "failure_paths", "conclusion": "Invalid and inconclusive paths remain terminal.", "evidence_refs": ["e1"]},
|
|
763
|
+
{"concern": "tests_evidence", "conclusion": "Deterministic contract tests cover the risky path.", "evidence_refs": ["e1"]},
|
|
764
|
+
{"concern": "compatibility", "conclusion": "Existing provider routing remains compatible.", "evidence_refs": ["e1"]},
|
|
765
|
+
{"concern": "rollout_rollback", "conclusion": "The local contract change has a direct revert path.", "evidence_refs": ["e1"]},
|
|
766
|
+
{"concern": "observability_operations", "conclusion": "The result exposes depth and risk tags for diagnosis.", "evidence_refs": ["e1"]},
|
|
767
|
+
{"concern": "high_risk_boundary", "conclusion": "Bypass attempts cannot remove controller-required concerns.", "evidence_refs": ["e1"]}
|
|
768
|
+
],
|
|
769
|
+
"evidence": [{"id": "e1", "result": "Deterministic high-risk gate fixture."}]
|
|
770
|
+
}
|
|
771
|
+
JSON
|
|
772
|
+
large_intent="$(awk 'BEGIN { for (i = 0; i < 3900; i++) printf "i" }')"
|
|
773
|
+
large_acceptance="$(awk 'BEGIN { for (i = 0; i < 990; i++) printf "a" }')"
|
|
774
|
+
large_conclusion="$(awk 'BEGIN { for (i = 0; i < 1900; i++) printf "c" }')"
|
|
775
|
+
large_evidence="$(awk 'BEGIN { for (i = 0; i < 1900; i++) printf "e" }')"
|
|
776
|
+
cat >"$WORK/near-limit-plan.json" <<JSON
|
|
777
|
+
{
|
|
778
|
+
"intent": "$large_intent",
|
|
779
|
+
"acceptance": [
|
|
780
|
+
"$large_acceptance", "$large_acceptance", "$large_acceptance", "$large_acceptance",
|
|
781
|
+
"$large_acceptance", "$large_acceptance", "$large_acceptance", "$large_acceptance",
|
|
782
|
+
"$large_acceptance", "$large_acceptance", "$large_acceptance", "$large_acceptance",
|
|
783
|
+
"$large_acceptance", "$large_acceptance", "$large_acceptance"
|
|
784
|
+
],
|
|
785
|
+
"self_review": [
|
|
786
|
+
{"concern": "correctness", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
787
|
+
{"concern": "safety", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
788
|
+
{"concern": "failure_paths", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
789
|
+
{"concern": "tests_evidence", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]},
|
|
790
|
+
{"concern": "compatibility", "conclusion": "$large_conclusion", "evidence_refs": ["e1"]}
|
|
791
|
+
],
|
|
792
|
+
"evidence": [{"id": "e1", "result": "$large_evidence"}]
|
|
793
|
+
}
|
|
794
|
+
JSON
|
|
795
|
+
|
|
796
|
+
reset_case() {
|
|
797
|
+
rm -f "$WORK/state"/*
|
|
798
|
+
printf '%s' "$1" >"$WORK/state/claude_behavior"
|
|
799
|
+
printf '%s' "$2" >"$WORK/state/kimi_behavior"
|
|
800
|
+
printf '%s' "$3" >"$WORK/state/opencode_behavior"
|
|
801
|
+
printf '%s' "${4:-unavailable}" >"$WORK/state/codex_behavior"
|
|
802
|
+
}
|
|
803
|
+
|
|
804
|
+
run_gate_from() {
|
|
805
|
+
entrypoint="$1"
|
|
806
|
+
shift
|
|
807
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$entrypoint" \
|
|
808
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
809
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json" "$@"
|
|
810
|
+
}
|
|
811
|
+
|
|
812
|
+
run_gate() {
|
|
813
|
+
run_gate_from "$WORK/harness/scripts/review_gate.sh" "$@"
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
run_challenge_gate() {
|
|
817
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
818
|
+
--mode challenge --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
819
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
820
|
+
--challenge-budget 1 --challenge-index 1 "$@"
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
run_completion_gate() {
|
|
824
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
825
|
+
--mode complete --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
826
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json" "$@"
|
|
827
|
+
}
|
|
828
|
+
|
|
829
|
+
run_base_gate() {
|
|
830
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
831
|
+
--mode review --cwd "$WORK/repo" --base HEAD --implementer-family openai \
|
|
832
|
+
--review-plan-file "$WORK/review-plan.json" "$@"
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
check() {
|
|
836
|
+
if eval "$2"; then echo "ok - $1"; else echo "FAIL - $1"; fails=$((fails+1)); fi
|
|
837
|
+
}
|
|
838
|
+
|
|
839
|
+
json_fields() {
|
|
840
|
+
local payload="$1"
|
|
841
|
+
shift
|
|
842
|
+
JSON_PAYLOAD="$payload" python3 - "$@" <<'PY' 2>/dev/null
|
|
843
|
+
import json, os, sys
|
|
844
|
+
payload = json.loads(os.environ["JSON_PAYLOAD"])
|
|
845
|
+
for expected in sys.argv[1:]:
|
|
846
|
+
path, wanted = expected.split("=", 1)
|
|
847
|
+
value = payload
|
|
848
|
+
for part in path.split("."):
|
|
849
|
+
value = value[int(part)] if isinstance(value, list) else value[part]
|
|
850
|
+
assert str(value).lower() == wanted.lower(), (path, value, wanted)
|
|
851
|
+
PY
|
|
852
|
+
}
|
|
853
|
+
|
|
854
|
+
json_lacks() {
|
|
855
|
+
local payload="$1" field="$2"
|
|
856
|
+
JSON_PAYLOAD="$payload" python3 - "$field" <<'PY' 2>/dev/null
|
|
857
|
+
import json, os, sys
|
|
858
|
+
payload = json.loads(os.environ["JSON_PAYLOAD"])
|
|
859
|
+
assert sys.argv[1] not in payload, (sys.argv[1], payload.get(sys.argv[1]))
|
|
860
|
+
PY
|
|
861
|
+
}
|
|
862
|
+
|
|
863
|
+
# Direct unit coverage for the egress secret-scan tripwire: precision (no false
|
|
864
|
+
# positive on ordinary code or placeholders) and recall (catches representative
|
|
865
|
+
# credential shapes). Locks in the security-sensitive behavior independently of
|
|
866
|
+
# the end-to-end egress tests below.
|
|
867
|
+
scan_out="$(REVIEW_GATE_DIR="$DIR" python3 - <<'PY' 2>&1
|
|
868
|
+
import os, sys
|
|
869
|
+
sys.path.insert(0, os.environ["REVIEW_GATE_DIR"])
|
|
870
|
+
from review_gate import scan_egress_secrets as s
|
|
871
|
+
cases = [
|
|
872
|
+
(b"def handler():\n return get_password(user)\n", []),
|
|
873
|
+
(b"-a\n+b\n", []),
|
|
874
|
+
(b'password = "changeme"\n', []),
|
|
875
|
+
(b'token = "${VAULT_TOKEN}"\n', []),
|
|
876
|
+
(b'api_key: "<your-api-key-here>"\n', []),
|
|
877
|
+
(b'+aws_key = "AKIAIOSFODNN7EXAMPLE"\n', ["aws_access_key_id"]),
|
|
878
|
+
(b"-----BEGIN RSA PRIVATE KEY-----\n", ["private_key"]),
|
|
879
|
+
(b"gh_token = ghp_" + b"a" * 36 + b"\n", ["github_token"]),
|
|
880
|
+
(b"pat = github_pat_" + b"a" * 22 + b"\n", ["github_token"]),
|
|
881
|
+
(b"OPENAI_API_KEY=sk-proj-" + b"a" * 24 + b"\n", ["openai_api_key"]),
|
|
882
|
+
(b"legacy = sk-" + b"a" * 24 + b"\n", ["openai_api_key"]),
|
|
883
|
+
(b"slug = sk-" + b"this-is-an-ordinary-config-name\n", []),
|
|
884
|
+
(b'password = "s3cr3t-value-here"\n', ["secret_assignment"]),
|
|
885
|
+
(b'url = "https://user:hunter2@db.' + b'internal/x"\n', ["credentialed_url"]),
|
|
886
|
+
]
|
|
887
|
+
bad = [(p, want, s(p)) for p, want in cases if s(p) != want]
|
|
888
|
+
if bad:
|
|
889
|
+
for entry in bad:
|
|
890
|
+
print("MISMATCH", entry)
|
|
891
|
+
sys.exit(1)
|
|
892
|
+
print("scanner_ok")
|
|
893
|
+
PY
|
|
894
|
+
)"
|
|
895
|
+
check "egress secret scanner: precision and recall on representative inputs" \
|
|
896
|
+
'[ "$scan_out" = "scanner_ok" ]'
|
|
897
|
+
|
|
898
|
+
reset_case passed unavailable unavailable
|
|
899
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
900
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
901
|
+
check "Claude pass stops before other clients" \
|
|
902
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$profile" owner_selection_source=implementer-declared selected_skills.0.name=code-review && [ "$(printf "%s" "$profile" | python3 -c "import json,sys; print(len(json.load(sys.stdin).get(\"selected_skills\", [])))")" = 1 ] && json_fields "$out" selected_client=claude owner_selection_source=implementer-declared selected_skills.0=code-review fallback_attempt_count=0 completion_gated=true next_action=deep_self_review_before_completion autonomous_review_budget=1 autonomous_review_index=1 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=false review_state=reviewed self_review_gate.required=true self_review_gate.required_triggers.0=before_completion_claim self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=completion_claim && [ "$(printf "%s" "$out" | python3 -c "import json,sys; print(len(json.load(sys.stdin).get(\"selected_skills\", [])))")" = 1 ] && json_lacks "$out" delivery && json_lacks "$out" owner_gaps && json_lacks "$out" residual_risks && json_lacks "$out" decision && json_lacks "$out" skill_gap_candidates'
|
|
903
|
+
|
|
904
|
+
printf '%s\n' "$out" >"$WORK/completion-review.json"
|
|
905
|
+
reset_case passed unavailable unavailable
|
|
906
|
+
out="$(run_completion_gate --completion-review-result-file "$WORK/completion-review.json")"; rc=$?
|
|
907
|
+
check "completion claim requires a separate exact-candidate deep-self-review checkpoint" \
|
|
908
|
+
'[ "$rc" = 0 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" mode=complete status=passed completion_gated=false next_action=complete review_state=self_reviewed self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_completion_claim completion_review_result_sha256='"$(shasum -a 256 "$WORK/completion-review.json" | awk '{print $1}')"''
|
|
909
|
+
|
|
910
|
+
reset_case passed unavailable unavailable
|
|
911
|
+
out="$(run_completion_gate --review-plan-file "$WORK/changed-review-plan.json" --completion-review-result-file "$WORK/completion-review.json")"; rc=$?
|
|
912
|
+
check "an untracked completion checkpoint rejects changed intent" \
|
|
913
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
914
|
+
|
|
915
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-b\n+c\n' >"$WORK/diff.patch"
|
|
916
|
+
reset_case passed unavailable unavailable
|
|
917
|
+
out="$(run_completion_gate --completion-review-result-file "$WORK/completion-review.json")"; rc=$?
|
|
918
|
+
check "a completion checkpoint rejects a review result for an older candidate" \
|
|
919
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true next_action=run_external_review_for_current_candidate self_review_gate.required=true self_review_gate.required_triggers.0=material_candidate_change self_review_gate.blocks.0=completion_claim'
|
|
920
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
|
|
921
|
+
|
|
922
|
+
reset_case passed unavailable unavailable
|
|
923
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
924
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
925
|
+
check "owner-guided self-review passes installed skill names without copying owner bodies" \
|
|
926
|
+
'[ "$rc" = 0 ] && json_fields "$profile" owner_selection_source=implementer-declared selected_skills.0.name=code-review selected_skills.1.name=testing-strategy self_review.3.skill=testing-strategy skill_delivery=native-installed && json_lacks "$profile" owner_context && ! grep -q "# Testing Strategy" <<<"$profile" && [ "$(cat "$WORK/state/claude_skill_registry_root")" = "$WORK_REAL" ] && [ "$(cat "$WORK/state/claude_review_skills")" = testing-strategy ] && json_fields "$out" owner_selection_source=implementer-declared selected_skills.0=code-review selected_skills.1=testing-strategy reviewed_skills.0=testing-strategy skill_usage_evidence.mode=native-explicit-invocation skill_usage_evidence.observed=false skill_usage_evidence.source=claude-wrapper && [ "$(printf "%s" "$out" | python3 -c "import json,sys; p=json.load(sys.stdin); print(len(p.get(\"selected_skills\", [])), len(p.get(\"reviewed_skills\", [])), len(p.get(\"observed_skill_usage\", [])))")" = "2 1 0" ]'
|
|
927
|
+
|
|
928
|
+
reset_case missing_binding unavailable unavailable
|
|
929
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
930
|
+
check "owner-aware success without a wrapper binding receipt fails closed" \
|
|
931
|
+
'[ "$rc" = 2 ] && json_fields "$out" reason_code=binding_mismatch next_action=stop_reviewer_lane && [ "$(printf "%s" "$out" | python3 -c "import json,sys; print(len(json.load(sys.stdin).get(\"reviewed_skills\", [])))")" = 0 ]'
|
|
932
|
+
|
|
933
|
+
printf 'diff --git a/skills/testing-strategy/SKILL.md b/skills/testing-strategy/SKILL.md\n--- a/skills/testing-strategy/SKILL.md\n+++ b/skills/testing-strategy/SKILL.md\n@@ -1 +1 @@\n-old\n+new\n' >"$WORK/diff.patch"
|
|
934
|
+
reset_case passed unavailable unavailable
|
|
935
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
936
|
+
check "a path-derived owner missing from self-review fails before provider execution" \
|
|
937
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete next_action=deep_self_review'
|
|
938
|
+
|
|
939
|
+
reset_case passed unavailable unavailable
|
|
940
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
941
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
942
|
+
check "a changed skill path routes its owner into the frozen reviewer profile" \
|
|
943
|
+
'[ "$rc" = 0 ] && json_fields "$profile" owner_selection_source=controller-derived+implementer-declared owner_selection_evidence.0.skill=testing-strategy owner_selection_evidence.0.source=changed-skill-path skill_delivery=native-installed && [ "$(cat "$WORK/state/claude_review_skills")" = testing-strategy ] && json_fields "$out" owner_selection_source=controller-derived+implementer-declared reviewed_skills.0=testing-strategy'
|
|
944
|
+
|
|
945
|
+
printf 'diff --git a/src/service.py b/src/service.py\n--- a/src/service.py\n+++ b/src/service.py\n@@ -1 +1 @@\n-old\n+new\n' >"$WORK/diff.patch"
|
|
946
|
+
reset_case passed unavailable unavailable
|
|
947
|
+
out="$(run_gate --review-plan-file "$WORK/python-owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
948
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
949
|
+
check "a Python candidate routes the registered Python implementation owner" \
|
|
950
|
+
'[ "$rc" = 0 ] && json_fields "$profile" owner_selection_evidence.0.skill=python-service-dev owner_selection_evidence.0.source=file-type:.py skill_delivery=native-installed && [ "$(cat "$WORK/state/claude_review_skills")" = python-service-dev ] && json_fields "$out" reviewed_skills.0=python-service-dev'
|
|
951
|
+
|
|
952
|
+
printf 'diff --git a/scripts/review.sh b/scripts/review.sh\n--- a/scripts/review.sh\n+++ b/scripts/review.sh\n@@ -1 +1 @@\n-old\n+new\n' >"$WORK/diff.patch"
|
|
953
|
+
reset_case passed unavailable unavailable
|
|
954
|
+
out="$(run_gate --review-plan-file "$WORK/shell-owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
955
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
956
|
+
check "a shell candidate routes the terminal CLI owner" \
|
|
957
|
+
'[ "$rc" = 0 ] && json_fields "$profile" owner_selection_evidence.0.skill=terminal-cli-dev owner_selection_evidence.0.source=file-type:.sh skill_delivery=native-installed && [ "$(cat "$WORK/state/claude_review_skills")" = terminal-cli-dev ] && json_fields "$out" reviewed_skills.0=terminal-cli-dev'
|
|
958
|
+
|
|
959
|
+
printf 'diff --git a/cmd/service.go b/cmd/service.go\n--- a/cmd/service.go\n+++ b/cmd/service.go\n@@ -1 +1 @@\n-old\n+new\ndiff --git a/web/view.ts b/web/view.ts\n--- a/web/view.ts\n+++ b/web/view.ts\n@@ -1 +1 @@\n-old\n+new\ndiff --git a/lib/screen.dart b/lib/screen.dart\n--- a/lib/screen.dart\n+++ b/lib/screen.dart\n@@ -1 +1 @@\n-old\n+new\n' >"$WORK/diff.patch"
|
|
960
|
+
reset_case passed unavailable unavailable
|
|
961
|
+
out="$(run_gate --review-plan-file "$WORK/language-owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
962
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
963
|
+
check "Go, web, and Dart candidates route all registered implementation owners" \
|
|
964
|
+
'[ "$rc" = 0 ] && json_fields "$profile" owner_selection_evidence.0.skill=app-cross-platform-dev owner_selection_evidence.1.skill=go-microservice-dev owner_selection_evidence.2.skill=web-react-dev && [ "$(cat "$WORK/state/claude_review_skills")" = "app-cross-platform-dev go-microservice-dev web-react-dev" ] && json_fields "$out" reviewed_skills.0=app-cross-platform-dev reviewed_skills.1=go-microservice-dev reviewed_skills.2=web-react-dev'
|
|
965
|
+
|
|
966
|
+
printf 'diff --git a/src/service.py b/src/service.py\n--- a/src/service.py\n+++ b/src/service.py\n@@ -1 +1 @@\n-old\n+new\n' >"$WORK/diff.patch"
|
|
967
|
+
reset_case quota passed unavailable
|
|
968
|
+
out="$(run_gate --review-plan-file "$WORK/python-owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
969
|
+
check "fallback reviewers receive the same native owner-skill binding" \
|
|
970
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/claude_review_skills")" = python-service-dev ] && [ "$(cat "$WORK/state/kimi_review_skills")" = python-service-dev ] && [ "$(cat "$WORK/state/claude_skill_registry_root")" = "$(cat "$WORK/state/kimi_skill_registry_root")" ] && json_fields "$out" selected_client=kimi reviewed_skills.0=python-service-dev'
|
|
971
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
|
|
972
|
+
|
|
973
|
+
mkdir -p "$WORK/alternate-registry"
|
|
974
|
+
ln -s "$WORK/harness" "$WORK/alternate-registry/linked-harness"
|
|
975
|
+
reset_case passed unavailable unavailable
|
|
976
|
+
out="$(run_gate_from "$WORK/alternate-registry/linked-harness/scripts/review_gate.sh" \
|
|
977
|
+
--review-plan-file "$WORK/owner-review-plan.json" --allow-fallback-egress)"; rc=$?
|
|
978
|
+
profile="$(cat "$WORK/state/claude_profile" 2>/dev/null || true)"
|
|
979
|
+
check "a symlinked baseline resolves owners from the canonical source registry" \
|
|
980
|
+
'[ "$rc" = 0 ] && json_fields "$profile" selected_skills.0.name=code-review selected_skills.1.name=testing-strategy && json_fields "$out" selected_skills.0=code-review selected_skills.1=testing-strategy'
|
|
981
|
+
|
|
982
|
+
reset_case passed unavailable unavailable
|
|
983
|
+
out="$(run_gate --review-plan-file "$WORK/missing-owner-review-plan.json")"; rc=$?
|
|
984
|
+
check "an explicit owner outside the complete sibling registry fails before provider execution" \
|
|
985
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true self_review_gate.required_triggers.0=before_external_review'
|
|
986
|
+
|
|
987
|
+
reset_case passed unavailable unavailable
|
|
988
|
+
out="$(run_gate --review-plan-file "$WORK/missing-entrypoint-owner-review-plan.json")"; rc=$?
|
|
989
|
+
check "an explicit owner directory without SKILL.md is incomplete and terminal" \
|
|
990
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
|
|
991
|
+
|
|
992
|
+
reset_case passed unavailable unavailable
|
|
993
|
+
out="$(run_gate --review-plan-file "$WORK/file-owner-review-plan.json")"; rc=$?
|
|
994
|
+
check "a regular file cannot occupy an explicit owner package name" \
|
|
995
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=local_tool_failure fallback_eligible=false next_action=stop_reviewer_lane'
|
|
996
|
+
|
|
997
|
+
reset_case passed unavailable unavailable
|
|
998
|
+
out="$(run_gate --review-plan-file "$WORK/invalid-owner-review-plan.json")"; rc=$?
|
|
999
|
+
check "an out-of-contract owner name fails before provider execution" \
|
|
1000
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
|
|
1001
|
+
|
|
1002
|
+
for malformed_plan in non-string-owner-review-plan empty-owner-review-plan; do
|
|
1003
|
+
reset_case passed unavailable unavailable
|
|
1004
|
+
out="$(run_gate --review-plan-file "$WORK/$malformed_plan.json")"; rc=$?
|
|
1005
|
+
check "a malformed owner value is incomplete self-review and remains terminal ($malformed_plan)" \
|
|
1006
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete fallback_eligible=false next_action=deep_self_review self_review_gate.required=true'
|
|
1007
|
+
done
|
|
1008
|
+
|
|
1009
|
+
owner_lstat_classification="$(python3 - "$DIR/review_gate.py" <<'PY'
|
|
1010
|
+
import errno
|
|
1011
|
+
import importlib.util
|
|
1012
|
+
from pathlib import Path
|
|
1013
|
+
import sys
|
|
1014
|
+
import tempfile
|
|
1015
|
+
from unittest.mock import patch
|
|
1016
|
+
|
|
1017
|
+
spec = importlib.util.spec_from_file_location("review_gate_under_test", sys.argv[1])
|
|
1018
|
+
module = importlib.util.module_from_spec(spec)
|
|
1019
|
+
spec.loader.exec_module(module)
|
|
1020
|
+
results = []
|
|
1021
|
+
for error_number in (errno.ENOENT, errno.EACCES, errno.ENOTDIR):
|
|
1022
|
+
with patch.object(Path, "lstat", side_effect=OSError(error_number, "test")):
|
|
1023
|
+
try:
|
|
1024
|
+
module._hash_skill_package(Path("/registry/testing-strategy"), "testing-strategy")
|
|
1025
|
+
except module.GateError as exc:
|
|
1026
|
+
results.append(exc.reason_code)
|
|
1027
|
+
with tempfile.TemporaryDirectory() as directory:
|
|
1028
|
+
skill_root = Path(directory) / "testing-strategy"
|
|
1029
|
+
skill_root.mkdir()
|
|
1030
|
+
(skill_root / "SKILL.md").write_text("# Testing Strategy\n")
|
|
1031
|
+
(skill_root / "references").mkdir()
|
|
1032
|
+
original_lstat = Path.lstat
|
|
1033
|
+
for target in (skill_root / "SKILL.md", skill_root / "references"):
|
|
1034
|
+
def selective_lstat(self, target=target):
|
|
1035
|
+
if self == target:
|
|
1036
|
+
raise OSError(errno.EACCES, "test")
|
|
1037
|
+
return original_lstat(self)
|
|
1038
|
+
|
|
1039
|
+
with patch.object(Path, "lstat", selective_lstat):
|
|
1040
|
+
try:
|
|
1041
|
+
module._hash_skill_package(skill_root, "testing-strategy")
|
|
1042
|
+
except module.GateError as exc:
|
|
1043
|
+
results.append(exc.reason_code)
|
|
1044
|
+
print(" ".join(results))
|
|
1045
|
+
PY
|
|
1046
|
+
)"
|
|
1047
|
+
check "owner root inspection distinguishes absence from local access failure" \
|
|
1048
|
+
'[ "$owner_lstat_classification" = "self_review_incomplete local_tool_failure local_tool_failure local_tool_failure local_tool_failure" ]'
|
|
1049
|
+
|
|
1050
|
+
reset_case passed unavailable unavailable
|
|
1051
|
+
out="$(run_gate --review-plan-file "$WORK/linked-owner-review-plan.json")"; rc=$?
|
|
1052
|
+
check "a linked owner package is a terminal local registry-integrity failure" \
|
|
1053
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=local_tool_failure fallback_eligible=false next_action=stop_reviewer_lane'
|
|
1054
|
+
|
|
1055
|
+
printf '%s\n' '# Linked owner reference target' >"$WORK/owner-reference-target.md"
|
|
1056
|
+
ln -s "$WORK/owner-reference-target.md" \
|
|
1057
|
+
"$WORK/testing-strategy/references/linked.md"
|
|
1058
|
+
reset_case passed unavailable unavailable
|
|
1059
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json")"; rc=$?
|
|
1060
|
+
check "a linked Markdown reference in an explicit owner package is terminal" \
|
|
1061
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=local_tool_failure fallback_eligible=false next_action=stop_reviewer_lane'
|
|
1062
|
+
unlink "$WORK/testing-strategy/references/linked.md"
|
|
1063
|
+
|
|
1064
|
+
mv "$WORK/testing-strategy/references" "$WORK/testing-strategy/references-real"
|
|
1065
|
+
ln -s "$WORK/testing-strategy/references-real" \
|
|
1066
|
+
"$WORK/testing-strategy/references"
|
|
1067
|
+
reset_case passed unavailable unavailable
|
|
1068
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json")"; rc=$?
|
|
1069
|
+
check "a linked references root in an explicit owner package is terminal" \
|
|
1070
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=local_tool_failure fallback_eligible=false next_action=stop_reviewer_lane'
|
|
1071
|
+
unlink "$WORK/testing-strategy/references"
|
|
1072
|
+
mv "$WORK/testing-strategy/references-real" "$WORK/testing-strategy/references"
|
|
1073
|
+
|
|
1074
|
+
mv "$WORK/testing-strategy/SKILL.md" "$WORK/testing-strategy/SKILL.md.real"
|
|
1075
|
+
mkdir "$WORK/testing-strategy/SKILL.md"
|
|
1076
|
+
reset_case passed unavailable unavailable
|
|
1077
|
+
out="$(run_gate --review-plan-file "$WORK/owner-review-plan.json")"; rc=$?
|
|
1078
|
+
check "a non-regular entrypoint in an explicit owner package is terminal" \
|
|
1079
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=local_tool_failure fallback_eligible=false next_action=stop_reviewer_lane'
|
|
1080
|
+
rmdir "$WORK/testing-strategy/SKILL.md"
|
|
1081
|
+
mv "$WORK/testing-strategy/SKILL.md.real" "$WORK/testing-strategy/SKILL.md"
|
|
1082
|
+
|
|
1083
|
+
reset_case passed unavailable unavailable
|
|
1084
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1085
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/large-diff.patch" \
|
|
1086
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json")"; rc=$?
|
|
1087
|
+
check "a bounded 180 KB candidate remains reviewable as one frozen packet" \
|
|
1088
|
+
'[ "$rc" = 0 ] && [ "$(wc -c <"$WORK/state/claude_packet" | tr -d "[:space:]")" = 180000 ] && json_fields "$out" selected_client=claude'
|
|
1089
|
+
|
|
1090
|
+
reset_case passed unavailable unavailable
|
|
1091
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1092
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/nul-diff.patch" \
|
|
1093
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json")"; rc=$?
|
|
1094
|
+
check "a NUL-bearing candidate fails before provider execution" \
|
|
1095
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input completion_gated=true'
|
|
1096
|
+
|
|
1097
|
+
reset_case passed unavailable unavailable
|
|
1098
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1099
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
1100
|
+
--implementer-family openai --review-plan-file "$WORK/near-limit-plan.json")"; rc=$?
|
|
1101
|
+
check "a valid plan near the documented 32 KB input limit can render its profile" \
|
|
1102
|
+
'[ "$(wc -c <"$WORK/near-limit-plan.json")" -le 32000 ] && [ "$rc" = 0 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ]'
|
|
1103
|
+
|
|
1104
|
+
reset_case missing_coverage passed unavailable
|
|
1105
|
+
out="$(run_gate --diff-file "$WORK/secret-diff.patch")"; rc=$?
|
|
1106
|
+
check "missing coverage cannot widen egress for a secret-bearing diff without approval" \
|
|
1107
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && [ ! -e "$WORK/state/kimi_profile_hash" ] && json_fields "$out" reason_code=egress_denied egress.secret_scan.0=aws_access_key_id'
|
|
1108
|
+
|
|
1109
|
+
reset_case missing_coverage passed unavailable
|
|
1110
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1111
|
+
check "a clean reply without per-concern conclusions cannot satisfy the gate" \
|
|
1112
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" selected_client=kimi skipped_clients.0.reason_code=invalid_model_output'
|
|
1113
|
+
|
|
1114
|
+
# The wire, not its ends: builds_synthetic_slot and normalize_concern_results are
|
|
1115
|
+
# each asserted directly, but only a run through the gate proves the flag is
|
|
1116
|
+
# returned by freeze_review_profile and forwarded by main. Drop the forwarding and
|
|
1117
|
+
# the challenge case below stops accepting.
|
|
1118
|
+
#
|
|
1119
|
+
# The review case does NOT cover the opposite mistake. Every review profile has
|
|
1120
|
+
# several concerns, so a flag forwarded unconditionally would still be refused for
|
|
1121
|
+
# failing the single-slot count rather than the flag — no end-to-end review case
|
|
1122
|
+
# can distinguish the two. That direction is covered directly instead, by
|
|
1123
|
+
# builds_synthetic_slot returning False for non-challenge modes and by
|
|
1124
|
+
# normalize_concern_results refusing a renamed result when the flag is absent.
|
|
1125
|
+
# What the review case does prove is that a renamed id is not accepted there.
|
|
1126
|
+
reset_case renamed_concern passed unavailable
|
|
1127
|
+
out="$(run_challenge_gate --focus "the release path under cancellation")"; rc=$?
|
|
1128
|
+
check "challenge accepts a renamed concern id through the whole gate" \
|
|
1129
|
+
'[ "$rc" = 0 ] && json_fields "$out" selected_client=claude status=passed reviewed_concerns.0=challenge_focus attempts.0.concern_results.0.concern=a_focus_shaped_slug'
|
|
1130
|
+
|
|
1131
|
+
reset_case renamed_concern passed unavailable
|
|
1132
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1133
|
+
check "review refuses the same renamed concern id" \
|
|
1134
|
+
'[ "$rc" = 2 ] && json_fields "$out" status=inconclusive attempts.0.status=inconclusive'
|
|
1135
|
+
|
|
1136
|
+
# What the id bound does and does not do. It decides what may be ACCEPTED: an
|
|
1137
|
+
# oversized id cannot become a recorded conclusion. It does not keep the raw
|
|
1138
|
+
# string out of the attempt record, because record_attempt runs before
|
|
1139
|
+
# normalize_concern_results and copies the payload verbatim — an earlier comment
|
|
1140
|
+
# here claimed the opposite ordering and a review round caught it. That is the
|
|
1141
|
+
# design, not a leak to plug: attempt records are evidence of what a reviewer
|
|
1142
|
+
# actually returned, and every field in them is unbounded the same way, so
|
|
1143
|
+
# truncating one would forge the evidence while fixing nothing. Both halves are
|
|
1144
|
+
# pinned so neither can be quietly changed into the other.
|
|
1145
|
+
reset_case oversized_concern_id passed unavailable
|
|
1146
|
+
out="$(run_challenge_gate --focus "the release path under cancellation")"; rc=$?
|
|
1147
|
+
check "an oversized concern id is refused while the raw reply stays readable" \
|
|
1148
|
+
'[ "$rc" = 2 ] && json_fields "$out" status=inconclusive reason_code=invalid_model_output && [ "$(printf %s "$out" | python3 -c "import json,sys; print(len(json.load(sys.stdin)[\"attempts\"][0][\"concern_results\"][0][\"concern\"]))")" = 600 ]'
|
|
1149
|
+
|
|
1150
|
+
reset_case partial_coverage passed unavailable
|
|
1151
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1152
|
+
check "partial concern conclusions are terminal and cannot be laundered by fallback" \
|
|
1153
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=invalid_model_output attempts.0.concern_evidence=true'
|
|
1154
|
+
|
|
1155
|
+
reset_case extra_coverage passed unavailable
|
|
1156
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1157
|
+
check "extra concern conclusions are terminal and cannot expand controller scope" \
|
|
1158
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=invalid_model_output attempts.0.concern_evidence=true'
|
|
1159
|
+
|
|
1160
|
+
reset_case placeholder_coverage passed unavailable
|
|
1161
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1162
|
+
check "complete but placeholder concern conclusions cannot false-green" \
|
|
1163
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=invalid_model_output attempts.0.concern_evidence=true'
|
|
1164
|
+
|
|
1165
|
+
reset_case findings unavailable unavailable
|
|
1166
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1167
|
+
check "Claude findings remain findings" \
|
|
1168
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=claude next_action=triage_findings_and_continue_independent_work autonomous_review_budget=1 autonomous_review_index=1 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
1169
|
+
|
|
1170
|
+
reset_case quota passed unavailable
|
|
1171
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1172
|
+
check "candidate-local Claude failure uses Kimi next" \
|
|
1173
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" selected_client=kimi fallback_attempt_count=1'
|
|
1174
|
+
|
|
1175
|
+
reset_case quota findings unavailable
|
|
1176
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1177
|
+
check "selected fallback findings are promoted to the stable result contract" \
|
|
1178
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings selected_client=kimi selected_attempt_index=1 findings.0.severity=P1 findings.0.file=fallback.py primary.status=inconclusive fallbacks.0.status=findings'
|
|
1179
|
+
|
|
1180
|
+
reset_case quota quota passed
|
|
1181
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1182
|
+
check "candidate-local Kimi failure continues to OpenCode" \
|
|
1183
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi opencode " ] && json_fields "$out" selected_client=opencode fallback_attempt_count=2'
|
|
1184
|
+
claude_hash="$(cat "$WORK/state/claude_hash")"
|
|
1185
|
+
check "all attempted clients receive the same frozen packet" \
|
|
1186
|
+
'[ "$claude_hash" = "$(cat "$WORK/state/kimi_hash")" ] && [ "$claude_hash" = "$(cat "$WORK/state/opencode_hash")" ] && json_fields "$out" packet_sha256="$claude_hash"'
|
|
1187
|
+
claude_profile_hash="$(cat "$WORK/state/claude_profile_hash")"
|
|
1188
|
+
check "all attempted clients receive the same staged review profile" \
|
|
1189
|
+
'[ "$claude_profile_hash" = "$(cat "$WORK/state/kimi_profile_hash")" ] && [ "$claude_profile_hash" = "$(cat "$WORK/state/opencode_profile_hash")" ] && json_fields "$out" review_profile_sha256="$claude_profile_hash" stage=build review_depth=build challenge_budget=0 reviewed_concerns.0=correctness'
|
|
1190
|
+
|
|
1191
|
+
reset_case quota quota native_timeout passed
|
|
1192
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1193
|
+
check "native owner-skill timeout receipt remains fail-closed and cascade-eligible through the gate" \
|
|
1194
|
+
'[ "$rc" = 2 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi opencode " ] && json_fields "$out" status=inconclusive reason_code=timeout attempts.2.status=inconclusive attempts.2.reason=review_native_skill_stream_timeout attempts.2.reason_code=timeout attempts.2.cascade_eligible=true attempts.2.timeout_diagnostic.native_owner_skills_requested=true attempts.2.diagnostic_artifacts.retained=true'
|
|
1195
|
+
|
|
1196
|
+
# A reviewer-local input ceiling is a client capability failure, not a candidate
|
|
1197
|
+
# defect. It must cascade; dropping capability_missing from the allowed set
|
|
1198
|
+
# would turn a routine transport limit into a lane-fatal result.
|
|
1199
|
+
reset_case quota oversize_inline passed
|
|
1200
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1201
|
+
check "a client-local input ceiling cascades instead of stopping the lane" \
|
|
1202
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi opencode " ] && json_fields "$out" selected_client=opencode attempts.1.reason_code=capability_missing'
|
|
1203
|
+
|
|
1204
|
+
reset_case quota concern_cascade passed
|
|
1205
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1206
|
+
check "gate never cascades past explicit concern evidence" \
|
|
1207
|
+
'[ "$rc" = 2 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" reason_code=invalid_model_output attempts.1.concern_evidence=true'
|
|
1208
|
+
|
|
1209
|
+
reset_case quota_mutate passed passed
|
|
1210
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1211
|
+
check "packet mutation fails before another client" \
|
|
1212
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=binding_mismatch client_order.0=claude attempts.0.client=claude attempts.0.reason_code=binding_mismatch'
|
|
1213
|
+
|
|
1214
|
+
reset_case quota boundary passed
|
|
1215
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1216
|
+
check "tool-boundary failure is terminal" \
|
|
1217
|
+
'[ "$rc" = 2 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" reason_code=tool_boundary_violation'
|
|
1218
|
+
|
|
1219
|
+
reset_case quota unavailable mismatch
|
|
1220
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1221
|
+
check "mode attribution mismatch is terminal" \
|
|
1222
|
+
'[ "$rc" = 2 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi opencode " ] && json_fields "$out" reason_code=binding_mismatch'
|
|
1223
|
+
|
|
1224
|
+
reset_case quota passed passed
|
|
1225
|
+
out="$(run_gate --diff-file "$WORK/secret-diff.patch")"; rc=$?
|
|
1226
|
+
check "secret-bearing non-Claude egress is denied unless approved" \
|
|
1227
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=egress_denied egress.allowed=false egress.secret_scan.0=aws_access_key_id'
|
|
1228
|
+
|
|
1229
|
+
reset_case quota passed passed
|
|
1230
|
+
out="$(run_gate)"; rc=$?
|
|
1231
|
+
check "a clean (secret-free) diff egresses to a non-Claude reviewer without the approval flag" \
|
|
1232
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" selected_client=kimi egress.allowed=true egress.approval_flag=false'
|
|
1233
|
+
|
|
1234
|
+
reset_case quota passed passed
|
|
1235
|
+
out="$(run_gate --review-plan-file "$WORK/secret-plan.json")"; rc=$?
|
|
1236
|
+
check "a secret in the review plan (which egresses in the profile) blocks non-Claude egress without approval" \
|
|
1237
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" reason_code=egress_denied egress.secret_scan.0=aws_access_key_id'
|
|
1238
|
+
|
|
1239
|
+
reset_case quota passed passed
|
|
1240
|
+
out="$(run_gate --diff-file "$WORK/secret-diff.patch" --allow-fallback-egress)"; rc=$?
|
|
1241
|
+
check "a secret-bearing diff egresses when the approval flag is passed" \
|
|
1242
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" selected_client=kimi egress.allowed=true egress.approval_flag=true egress.secret_scan.0=aws_access_key_id'
|
|
1243
|
+
|
|
1244
|
+
reset_case passed unavailable unavailable
|
|
1245
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1246
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
1247
|
+
--implementer-family openai)"; rc=$?
|
|
1248
|
+
check "review runs without a --review-plan-file and marks the plan derived-default" \
|
|
1249
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" review_plan_source=derived-default'
|
|
1250
|
+
|
|
1251
|
+
reset_case passed unavailable unavailable
|
|
1252
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1253
|
+
check "a supplied review plan is marked implementer-supplied" \
|
|
1254
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_plan_source=implementer-supplied'
|
|
1255
|
+
|
|
1256
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1257
|
+
--mode complete --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
1258
|
+
--implementer-family openai --completion-review-result-file "$WORK/completion-review.json")"; rc=$?
|
|
1259
|
+
check "complete mode still requires an explicit --review-plan-file" \
|
|
1260
|
+
'[ "$rc" = 2 ] && json_fields "$out" reason_code=completion_checkpoint_invalid'
|
|
1261
|
+
|
|
1262
|
+
reset_case auth passed passed
|
|
1263
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1264
|
+
check "Claude auth-path failure requests host retry first" \
|
|
1265
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ] && json_fields "$out" next_action=host_retry'
|
|
1266
|
+
|
|
1267
|
+
reset_case quota auth passed
|
|
1268
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1269
|
+
check "Kimi auth-path failure requests the same bounded host retry" \
|
|
1270
|
+
'[ "$rc" = 2 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ] && json_fields "$out" reason_code=auth_path_unavailable next_action=host_retry'
|
|
1271
|
+
|
|
1272
|
+
reset_case quota auth passed
|
|
1273
|
+
out="$(run_gate --host-remediation-attempted --allow-fallback-egress)"; rc=$?
|
|
1274
|
+
check "Kimi auth failure after host retry may use the next client" \
|
|
1275
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi opencode " ]'
|
|
1276
|
+
|
|
1277
|
+
reset_case auth passed unavailable
|
|
1278
|
+
out="$(run_gate --host-remediation-attempted --allow-fallback-egress)"; rc=$?
|
|
1279
|
+
check "auth failure after host retry may use the next client" \
|
|
1280
|
+
'[ "$rc" = 0 ] && [ "$(tr "\n" " " < "$WORK/state/client_sequence")" = "claude kimi " ]'
|
|
1281
|
+
|
|
1282
|
+
reset_case unavailable unavailable unavailable host_path
|
|
1283
|
+
out="$(CODE_REVIEW_CLIENT_ORDER=codex run_gate --implementer-family anthropic --allow-fallback-egress)"; rc=$?
|
|
1284
|
+
check "Codex sandbox host-path failure requests one host retry" \
|
|
1285
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = codex ] && json_fields "$out" reason_code=host_path_unavailable next_action=host_retry'
|
|
1286
|
+
|
|
1287
|
+
reset_case unavailable unavailable unavailable host_path
|
|
1288
|
+
out="$(CODE_REVIEW_CLIENT_ORDER=codex run_gate --implementer-family anthropic --host-remediation-attempted --allow-fallback-egress)"; rc=$?
|
|
1289
|
+
check "Codex repeated host-path failure becomes fallback-eligible only after retry" \
|
|
1290
|
+
'[ "$rc" = 2 ] && json_fields "$out" reason_code=host_path_unavailable_after_host_retry'
|
|
1291
|
+
|
|
1292
|
+
# The gate's own fail-closed floor: when the configured order contains no
|
|
1293
|
+
# cross-family client, every candidate is skipped in preflight and the loop
|
|
1294
|
+
# exhausts without ever setting a per-client reason. The initial
|
|
1295
|
+
# `no_independent_reviewer_available` must survive to the terminal envelope --
|
|
1296
|
+
# an independent review that never ran must not read as a clean lane.
|
|
1297
|
+
reset_case unavailable unavailable unavailable
|
|
1298
|
+
out="$(CODE_REVIEW_CLIENT_ORDER=claude run_gate --implementer-family anthropic)"; rc=$?
|
|
1299
|
+
# client_sequence is appended only after the fake wrapper's argument loop, so on
|
|
1300
|
+
# its own it cannot tell "never spawned" from "spawned and died early". The
|
|
1301
|
+
# wrapper records ${client}_timeout inside that loop, strictly earlier, so
|
|
1302
|
+
# asserting both is what pins the skip to preflight rather than to a short-lived
|
|
1303
|
+
# provider process.
|
|
1304
|
+
check "an order without any cross-family reviewer fails closed instead of reporting a clean lane" \
|
|
1305
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && [ ! -e "$WORK/state/claude_timeout" ] && json_fields "$out" status=inconclusive reason_code=no_independent_reviewer_available next_action=stop_reviewer_lane skipped_clients.0.client=claude skipped_clients.0.stage=preflight skipped_clients.0.reason_code=same_family_as_implementer'
|
|
1306
|
+
|
|
1307
|
+
# Precision row for the guard above: skipping a same-family client must cost
|
|
1308
|
+
# that client only, not the lane. A tightening that turned the skip into a
|
|
1309
|
+
# terminal outcome would pass the case above and fail here.
|
|
1310
|
+
reset_case unavailable passed unavailable
|
|
1311
|
+
out="$(CODE_REVIEW_CLIENT_ORDER=claude,kimi run_gate --implementer-family anthropic)"; rc=$?
|
|
1312
|
+
check "a same-family skip still leaves a cross-family reviewer usable" \
|
|
1313
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/client_sequence")" = kimi ] && json_fields "$out" selected_client=kimi skipped_clients.0.reason_code=same_family_as_implementer'
|
|
1314
|
+
|
|
1315
|
+
printf '' >"$WORK/empty-diff.patch"
|
|
1316
|
+
reset_case passed unavailable unavailable
|
|
1317
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1318
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/empty-diff.patch" \
|
|
1319
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json")"; rc=$?
|
|
1320
|
+
# No ${client}_timeout assertion here: freeze_packet raises before the client
|
|
1321
|
+
# loop exists, so no wrapper is ever spawned on this path and the clause would
|
|
1322
|
+
# be unreachable decoration. reason_code plus the absent client_sequence is the
|
|
1323
|
+
# whole reachable surface.
|
|
1324
|
+
check "an empty candidate fails before provider execution" \
|
|
1325
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" status=inconclusive reason_code=empty_diff fallback_eligible=false next_action=stop_reviewer_lane'
|
|
1326
|
+
|
|
1327
|
+
# --challenge-index is derived, not defaulted: in each shape exactly one value is
|
|
1328
|
+
# legal, so an omitted flag resolves to that value instead of to a constant that
|
|
1329
|
+
# is illegal in the mode the flag exists for. Explicit values keep their old
|
|
1330
|
+
# validation, which is what the last three rows here pin.
|
|
1331
|
+
challenge_gate_no_index() {
|
|
1332
|
+
REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1333
|
+
--mode challenge --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
1334
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
1335
|
+
--challenge-budget 1 --focus fallback-contract "$@"
|
|
1336
|
+
}
|
|
1337
|
+
|
|
1338
|
+
reset_case quota passed unavailable
|
|
1339
|
+
out="$(challenge_gate_no_index --allow-fallback-egress)"; rc=$?
|
|
1340
|
+
check "an untracked challenge without --challenge-index resolves to its only legal index" \
|
|
1341
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge challenge_index=1'
|
|
1342
|
+
|
|
1343
|
+
reset_case quota passed unavailable
|
|
1344
|
+
out="$(challenge_gate_no_index --allow-fallback-egress --challenge-index 0)"; rc=$?
|
|
1345
|
+
check "an explicit --challenge-index 0 is still rejected in challenge mode" \
|
|
1346
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
1347
|
+
|
|
1348
|
+
# Budget 2 on purpose: with budget 1 an index of 2 is caught by the range check
|
|
1349
|
+
# first, so it would never reach the tracked-chain requirement this row pins.
|
|
1350
|
+
reset_case quota passed unavailable
|
|
1351
|
+
out="$(challenge_gate_no_index --allow-fallback-egress --challenge-budget 2 --challenge-index 2)"; rc=$?
|
|
1352
|
+
check "an explicit later untracked challenge index still demands a tracked chain" \
|
|
1353
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_required'
|
|
1354
|
+
|
|
1355
|
+
reset_case passed unavailable unavailable
|
|
1356
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
1357
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
1358
|
+
--implementer-family openai --review-plan-file "$WORK/review-plan.json" \
|
|
1359
|
+
--challenge-index 1)"; rc=$?
|
|
1360
|
+
check "--challenge-index outside challenge mode is still rejected" \
|
|
1361
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
1362
|
+
|
|
1363
|
+
# The guard reads `challenge_index != 0`, so it rejects a nonzero index outside
|
|
1364
|
+
# challenge mode and an explicit 0 keeps running. Pinned because the resolver now
|
|
1365
|
+
# makes explicit-0 and omitted distinguishable, so a later presence-sensitive
|
|
1366
|
+
# tightening would be a silent CLI break rather than a caught one.
|
|
1367
|
+
reset_case passed unavailable unavailable
|
|
1368
|
+
out="$(run_gate --challenge-index 0)"; rc=$?
|
|
1369
|
+
check "an explicit --challenge-index 0 outside challenge mode is not rejected on the index" \
|
|
1370
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=review challenge_index=0'
|
|
1371
|
+
|
|
1372
|
+
reset_case passed unavailable unavailable
|
|
1373
|
+
out="$(run_gate)"; rc=$?
|
|
1374
|
+
check "review mode without --challenge-index still carries index 0" \
|
|
1375
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=review challenge_index=0'
|
|
1376
|
+
|
|
1377
|
+
# An orphan --autonomous-review-index (no --review-chain-id) is rejected before
|
|
1378
|
+
# the derivation could matter. Pinned so that if that guard ever moves, the
|
|
1379
|
+
# resolver is already keyed off chain presence rather than silently deriving a
|
|
1380
|
+
# tracked value for an untracked invocation.
|
|
1381
|
+
reset_case passed unavailable unavailable
|
|
1382
|
+
out="$(challenge_gate_no_index --challenge-budget 2 --autonomous-review-index 3)"; rc=$?
|
|
1383
|
+
check "an untracked challenge carrying an orphan Agent review index is rejected before derivation" \
|
|
1384
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
1385
|
+
|
|
1386
|
+
reset_case passed unavailable unavailable
|
|
1387
|
+
out="$(run_completion_gate --completion-review-result-file "$WORK/completion-review.json")"; rc=$?
|
|
1388
|
+
check "complete mode without --challenge-index still carries index 0" \
|
|
1389
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=complete challenge_index=0'
|
|
1390
|
+
|
|
1391
|
+
reset_case legacy passed passed
|
|
1392
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
1393
|
+
check "legacy inconclusive without machine fields fails closed" \
|
|
1394
|
+
'[ "$rc" = 2 ] && [ "$(cat "$WORK/state/client_sequence")" = claude ]'
|
|
1395
|
+
|
|
1396
|
+
reset_case quota passed unavailable
|
|
1397
|
+
out="$(run_challenge_gate --allow-fallback-egress --focus fallback-contract)"; rc=$?
|
|
1398
|
+
check "challenge mode remains challenge across clients" \
|
|
1399
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/claude_mode")" = challenge ] && [ "$(cat "$WORK/state/kimi_mode")" = challenge ] && json_fields "$out" mode=challenge challenge_budget=1 challenge_index=1 challenge_focus=fallback-contract challenge_rounds_remaining=0 autonomous_review_budget=2 autonomous_review_index=2 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=false review_state=reviewed next_action=deep_self_review_before_completion completion_gated=true reviewed_concerns.0=challenge_focus self_review_gate.required=true self_review_gate.required_triggers.0=before_completion_claim && ! grep -q challenges_used <<<"$out"'
|
|
1400
|
+
|
|
1401
|
+
printf '%s\n' "$out" >"$WORK/untracked-challenge.json"
|
|
1402
|
+
reset_case passed unavailable unavailable
|
|
1403
|
+
out="$(run_completion_gate --challenge-budget 1 --completion-review-result-file "$WORK/untracked-challenge.json")"; rc=$?
|
|
1404
|
+
check "an untracked standalone challenge cannot become Agent completion evidence" \
|
|
1405
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1406
|
+
|
|
1407
|
+
reset_case passed unavailable unavailable
|
|
1408
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus missing-chain-history)"; rc=$?
|
|
1409
|
+
check "a later challenge requires the single tracked Agent review chain" \
|
|
1410
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_required'
|
|
1411
|
+
|
|
1412
|
+
reset_case passed unavailable unavailable
|
|
1413
|
+
out="$(run_challenge_gate)"; rc=$?
|
|
1414
|
+
check "every challenge requires an explicit focus before provider execution" \
|
|
1415
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
1416
|
+
|
|
1417
|
+
reset_case passed unavailable unavailable
|
|
1418
|
+
out="$(run_challenge_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --focus trust-boundary)"; rc=$?
|
|
1419
|
+
check "high-risk challenge keeps its focus plus the high-risk boundary only" \
|
|
1420
|
+
'[ "$rc" = 0 ] && json_fields "$out" reviewed_concerns.0=challenge_focus reviewed_concerns.1=high_risk_boundary && ! grep -q correctness <<<"$out"'
|
|
1421
|
+
|
|
1422
|
+
reset_case passed unavailable unavailable
|
|
1423
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus trust-boundary)"; rc=$?
|
|
1424
|
+
check "a clean challenge may stop before exhausting its maximum budget" \
|
|
1425
|
+
'[ "$rc" = 0 ] && json_fields "$out" mode=challenge challenge_budget=2 challenge_index=1 challenge_rounds_remaining=1 autonomous_review_budget=3 autonomous_review_index=2 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=orchestrator_verify_history'
|
|
1426
|
+
|
|
1427
|
+
for stage in explore build release; do
|
|
1428
|
+
reset_case passed unavailable unavailable
|
|
1429
|
+
if [ "$stage" = release ]; then
|
|
1430
|
+
out="$(run_gate --stage "$stage" --review-chain-id release-task --autonomous-review-index 1)"; rc=$?
|
|
1431
|
+
else
|
|
1432
|
+
out="$(run_gate --stage "$stage")"; rc=$?
|
|
1433
|
+
fi
|
|
1434
|
+
check "stage $stage selects and records its minimum review depth" \
|
|
1435
|
+
'[ "$rc" = 0 ] && json_fields "$out" stage='"$stage"' review_depth='"$stage"''
|
|
1436
|
+
done
|
|
1437
|
+
|
|
1438
|
+
reset_case passed unavailable unavailable
|
|
1439
|
+
out="$(run_gate --challenge-budget 2)"; rc=$?
|
|
1440
|
+
check "an initial review with challenge capacity requires a tracked Agent chain" \
|
|
1441
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_required next_action=stop_reviewer_lane'
|
|
1442
|
+
|
|
1443
|
+
reset_case passed unavailable unavailable
|
|
1444
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --review-chain-id high-risk-task --autonomous-review-index 1)"; rc=$?
|
|
1445
|
+
check "high-risk tags raise explore to release depth and default one challenge" \
|
|
1446
|
+
'[ "$rc" = 0 ] && json_fields "$out" stage=explore stage_source=caller-declared review_depth=release risk_tags_source=caller-declared challenge_budget=1 challenge_rounds_remaining=1 review_chain_tracked=true review_chain_id=high-risk-task autonomous_review_budget=2 autonomous_review_index=1 autonomous_reviews_remaining=1 autonomous_review_allowed=true next_action=run_challenge completion_gated=true risk_tags.0=shared-gate reviewed_concerns.7=high_risk_boundary self_review_gate.required=false self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=risk_or_scope_escalation'
|
|
1447
|
+
|
|
1448
|
+
reset_case passed unavailable unavailable
|
|
1449
|
+
out="$(run_gate --stage release --challenge-budget 0)"; rc=$?
|
|
1450
|
+
check "release review cannot explicitly waive its first challenge" \
|
|
1451
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input next_action=stop_reviewer_lane'
|
|
1452
|
+
|
|
1453
|
+
reset_case passed unavailable unavailable
|
|
1454
|
+
out="$(run_gate --stage explore --risk-tag shared-gate --review-plan-file "$WORK/high-risk-plan.json" --challenge-budget 0)"; rc=$?
|
|
1455
|
+
check "high-risk depth cannot explicitly waive its first challenge" \
|
|
1456
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input next_action=stop_reviewer_lane'
|
|
1457
|
+
|
|
1458
|
+
reset_case findings unavailable unavailable
|
|
1459
|
+
chain_round_one="$(run_gate --challenge-budget 2 --review-chain-id long-task --autonomous-review-index 1)"; chain_round_one_rc=$?
|
|
1460
|
+
printf '%s\n' "$chain_round_one" >"$WORK/chain-round-one.json"
|
|
1461
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-b\n+c\n' >"$WORK/diff.patch"
|
|
1462
|
+
python3 - "$WORK/chain-round-one.json" \
|
|
1463
|
+
"$WORK/chain-round-one-forged-controller.json" \
|
|
1464
|
+
"$WORK/chain-round-one-forged-skills.json" \
|
|
1465
|
+
"$WORK/chain-round-one-forged-skills-hash.json" <<'PY'
|
|
1466
|
+
import json
|
|
1467
|
+
from pathlib import Path
|
|
1468
|
+
import sys
|
|
1469
|
+
|
|
1470
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1471
|
+
mutations = (
|
|
1472
|
+
("review_controller_sha256", "0" * 64),
|
|
1473
|
+
("selected_skills", ["testing-strategy"]),
|
|
1474
|
+
("selected_skills_sha256", "0" * 64),
|
|
1475
|
+
)
|
|
1476
|
+
for target, (field, value) in zip(sys.argv[2:], mutations):
|
|
1477
|
+
result = json.loads(json.dumps(source))
|
|
1478
|
+
result[field] = value
|
|
1479
|
+
Path(target).write_text(json.dumps(result, separators=(",", ":")))
|
|
1480
|
+
PY
|
|
1481
|
+
for forged_case in controller skills skills-hash; do
|
|
1482
|
+
reset_case passed unavailable unavailable
|
|
1483
|
+
forged_file="$WORK/chain-round-one-forged-${forged_case}.json"
|
|
1484
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus "forged-${forged_case}" --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$forged_file")"; rc=$?
|
|
1485
|
+
check "a tracked round rejects forged prior ${forged_case} binding before provider execution" \
|
|
1486
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
1487
|
+
done
|
|
1488
|
+
reset_case passed unavailable unavailable
|
|
1489
|
+
chain_round_two="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus corrected-path --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$WORK/chain-round-one.json")"; chain_round_two_rc=$?
|
|
1490
|
+
printf '%s\n' "$chain_round_two" >"$WORK/chain-round-two.json"
|
|
1491
|
+
check "a finding-bearing older candidate remains consumed in the same Agent review chain" \
|
|
1492
|
+
'[ "$chain_round_one_rc" = 0 ] && [ "$chain_round_two_rc" = 0 ] && json_fields "$chain_round_two" review_chain_tracked=true review_chain_id=long-task autonomous_review_index=2 autonomous_reviews_remaining=1 prior_review_result_sha256.0='"$(shasum -a 256 "$WORK/chain-round-one.json" | awk '{print $1}')"' self_review_gate.satisfied_triggers.0=before_external_review self_review_gate.satisfied_triggers.1=material_candidate_change'
|
|
1493
|
+
|
|
1494
|
+
reset_case passed unavailable unavailable
|
|
1495
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
1496
|
+
check "a clean tracked challenge may close before exhausting its maximum review budget" \
|
|
1497
|
+
'[ "$rc" = 0 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" mode=complete status=passed review_chain_tracked=true review_chain_id=long-task autonomous_review_budget=3 autonomous_review_index=2 autonomous_reviews_remaining=1 autonomous_review_allowed=false completion_gated=false'
|
|
1498
|
+
|
|
1499
|
+
python3 - "$WORK/chain-round-two.json" \
|
|
1500
|
+
"$WORK/chain-round-two-forged-controller.json" \
|
|
1501
|
+
"$WORK/chain-round-two-forged-skills.json" \
|
|
1502
|
+
"$WORK/chain-round-two-forged-skills-hash.json" \
|
|
1503
|
+
"$WORK/chain-round-two-forged-selection-source.json" <<'PY'
|
|
1504
|
+
import json
|
|
1505
|
+
from pathlib import Path
|
|
1506
|
+
import sys
|
|
1507
|
+
|
|
1508
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1509
|
+
mutations = (
|
|
1510
|
+
("review_controller_sha256", "0" * 64),
|
|
1511
|
+
("selected_skills", ["testing-strategy"]),
|
|
1512
|
+
("selected_skills_sha256", "0" * 64),
|
|
1513
|
+
("owner_selection_source", "spoofed-source"),
|
|
1514
|
+
)
|
|
1515
|
+
for target, (field, value) in zip(sys.argv[2:], mutations):
|
|
1516
|
+
result = json.loads(json.dumps(source))
|
|
1517
|
+
result[field] = value
|
|
1518
|
+
Path(target).write_text(json.dumps(result, separators=(",", ":")))
|
|
1519
|
+
PY
|
|
1520
|
+
for forged_case in controller skills skills-hash selection-source; do
|
|
1521
|
+
reset_case passed unavailable unavailable
|
|
1522
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/chain-round-two-forged-${forged_case}.json")"; rc=$?
|
|
1523
|
+
check "the completion checkpoint rejects a forged ${forged_case} binding" \
|
|
1524
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1525
|
+
done
|
|
1526
|
+
|
|
1527
|
+
reset_case passed unavailable unavailable
|
|
1528
|
+
out="$(run_completion_gate --review-plan-file "$WORK/changed-acceptance-plan.json" --challenge-budget 2 --completion-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
1529
|
+
check "a tracked completion checkpoint rejects changed acceptance" \
|
|
1530
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1531
|
+
|
|
1532
|
+
reset_case passed unavailable unavailable
|
|
1533
|
+
out="$(run_challenge_gate --review-plan-file "$WORK/changed-review-plan.json" --challenge-budget 2 --challenge-index 1 --focus changed-scope --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$WORK/chain-round-one.json")"; rc=$?
|
|
1534
|
+
check "tracked scope escalation requires deep self-review and a task reframe before another external review" \
|
|
1535
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_scope_changed next_action=deep_self_review_and_request_task_reframe self_review_gate.required=true self_review_gate.required_triggers.0=risk_or_scope_escalation self_review_gate.blocks.0=external_review self_review_gate.allowed_next_actions.2=request_human_decision'
|
|
1536
|
+
|
|
1537
|
+
reset_case passed unavailable unavailable
|
|
1538
|
+
passed_round_one="$(run_gate --challenge-budget 2 --review-chain-id passed-task --autonomous-review-index 1)"; passed_round_one_rc=$?
|
|
1539
|
+
printf '%s\n' "$passed_round_one" >"$WORK/passed-round-one.json"
|
|
1540
|
+
check "a passed first tracked round still owes its challenge before completion" \
|
|
1541
|
+
'[ "$passed_round_one_rc" = 0 ] && json_fields "$passed_round_one" status=passed autonomous_review_index=1 autonomous_reviews_remaining=2 autonomous_review_allowed=true next_action=run_challenge completion_gated=true'
|
|
1542
|
+
|
|
1543
|
+
python3 - "$WORK/passed-round-one.json" "$WORK/chain-round-two.json" \
|
|
1544
|
+
"$WORK/round-one-forged-final-shape.json" \
|
|
1545
|
+
"$WORK/round-two-forged-final-shape.json" <<'PY'
|
|
1546
|
+
import json
|
|
1547
|
+
from pathlib import Path
|
|
1548
|
+
import sys
|
|
1549
|
+
|
|
1550
|
+
final_gate = {
|
|
1551
|
+
"required": True,
|
|
1552
|
+
"required_triggers": ["before_completion_claim"],
|
|
1553
|
+
"blocks": ["completion_claim"],
|
|
1554
|
+
"allowed_next_actions": ["deep_self_review", "continue_implementation"],
|
|
1555
|
+
}
|
|
1556
|
+
for source_path, target_path in zip(sys.argv[1:3], sys.argv[3:5]):
|
|
1557
|
+
source = json.loads(Path(source_path).read_text())
|
|
1558
|
+
result = json.loads(json.dumps(source))
|
|
1559
|
+
result["next_action"] = "deep_self_review_before_completion"
|
|
1560
|
+
gate = dict(final_gate)
|
|
1561
|
+
gate["satisfied_triggers"] = source["self_review_gate"]["satisfied_triggers"]
|
|
1562
|
+
result["self_review_gate"] = gate
|
|
1563
|
+
Path(target_path).write_text(json.dumps(result, separators=(",", ":")))
|
|
1564
|
+
PY
|
|
1565
|
+
for forged_round in one two; do
|
|
1566
|
+
reset_case passed unavailable unavailable
|
|
1567
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/round-${forged_round}-forged-final-shape.json")"; rc=$?
|
|
1568
|
+
check "the completion checkpoint rejects a final-round shape while autonomous reviews remain (round ${forged_round})" \
|
|
1569
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1570
|
+
done
|
|
1571
|
+
|
|
1572
|
+
python3 - "$WORK/chain-round-one.json" \
|
|
1573
|
+
"$WORK/round-one-forged-scope-depth.json" \
|
|
1574
|
+
"$WORK/round-one-forged-scope-tags.json" \
|
|
1575
|
+
"$WORK/round-one-forged-scope-stage.json" \
|
|
1576
|
+
"$WORK/round-one-forged-scope-budget.json" <<'PY'
|
|
1577
|
+
import json
|
|
1578
|
+
from pathlib import Path
|
|
1579
|
+
import sys
|
|
1580
|
+
|
|
1581
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1582
|
+
mutations = (
|
|
1583
|
+
("review_depth", "release"),
|
|
1584
|
+
("risk_tags", ["shared-gate"]),
|
|
1585
|
+
("stage", "explore"),
|
|
1586
|
+
("challenge_budget", 5),
|
|
1587
|
+
)
|
|
1588
|
+
for target, (field, value) in zip(sys.argv[2:], mutations):
|
|
1589
|
+
result = json.loads(json.dumps(source))
|
|
1590
|
+
result[field] = value
|
|
1591
|
+
Path(target).write_text(json.dumps(result, separators=(",", ":")))
|
|
1592
|
+
PY
|
|
1593
|
+
for forged_scope in depth tags stage budget; do
|
|
1594
|
+
reset_case passed unavailable unavailable
|
|
1595
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus "scope-${forged_scope}" --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$WORK/round-one-forged-scope-${forged_scope}.json")"; rc=$?
|
|
1596
|
+
check "a tracked round rejects a prior scope field contradicting the copied scope digest (${forged_scope})" \
|
|
1597
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
1598
|
+
done
|
|
1599
|
+
|
|
1600
|
+
python3 - "$WORK/chain-round-one.json" \
|
|
1601
|
+
"$WORK/round-one-forged-inner-intent.json" \
|
|
1602
|
+
"$WORK/round-one-forged-inner-acceptance.json" \
|
|
1603
|
+
"$WORK/round-one-forged-inner-schema.json" \
|
|
1604
|
+
"$WORK/round-one-forged-inner-missing.json" <<'PY'
|
|
1605
|
+
import json
|
|
1606
|
+
from pathlib import Path
|
|
1607
|
+
import sys
|
|
1608
|
+
|
|
1609
|
+
source = json.loads(Path(sys.argv[1]).read_text())
|
|
1610
|
+
mutations = (
|
|
1611
|
+
("intent_sha256", "f" * 64),
|
|
1612
|
+
("acceptance_sha256", "e" * 64),
|
|
1613
|
+
("schema_version", 99),
|
|
1614
|
+
(None, None),
|
|
1615
|
+
)
|
|
1616
|
+
for target, (field, value) in zip(sys.argv[2:], mutations):
|
|
1617
|
+
result = json.loads(json.dumps(source))
|
|
1618
|
+
if field is None:
|
|
1619
|
+
result.pop("review_scope", None)
|
|
1620
|
+
else:
|
|
1621
|
+
result["review_scope"][field] = value
|
|
1622
|
+
Path(target).write_text(json.dumps(result, separators=(",", ":")))
|
|
1623
|
+
PY
|
|
1624
|
+
for forged_inner in intent acceptance schema missing; do
|
|
1625
|
+
reset_case passed unavailable unavailable
|
|
1626
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus "inner-${forged_inner}" --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$WORK/round-one-forged-inner-${forged_inner}.json")"; rc=$?
|
|
1627
|
+
check "a tracked round rejects a prior whose recorded scope does not produce its own digest (${forged_inner})" \
|
|
1628
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
1629
|
+
done
|
|
1630
|
+
|
|
1631
|
+
# Built from chain-round-two, the prior the completion checkpoint otherwise ACCEPTS
|
|
1632
|
+
# (early-challenge close, remaining=1). Forging a findings-bearing or
|
|
1633
|
+
# challenge-owing prior instead would be rejected by an unrelated existing check,
|
|
1634
|
+
# leaving this assertion vacuous.
|
|
1635
|
+
python3 - "$WORK/chain-round-two.json" "$WORK/round-two-forged-inner-intent.json" <<'PY'
|
|
1636
|
+
import json
|
|
1637
|
+
from pathlib import Path
|
|
1638
|
+
import sys
|
|
1639
|
+
|
|
1640
|
+
result = json.loads(Path(sys.argv[1]).read_text())
|
|
1641
|
+
result["review_scope"]["intent_sha256"] = "f" * 64
|
|
1642
|
+
Path(sys.argv[2]).write_text(json.dumps(result, separators=(",", ":")))
|
|
1643
|
+
PY
|
|
1644
|
+
reset_case passed unavailable unavailable
|
|
1645
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/round-two-forged-inner-intent.json")"; rc=$?
|
|
1646
|
+
check "the completion checkpoint rejects a prior whose recorded scope does not produce its own digest" \
|
|
1647
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1648
|
+
|
|
1649
|
+
# Both samples are built from priors the gate otherwise ACCEPTS, so removing the
|
|
1650
|
+
# guard under test actually turns these red rather than tripping an unrelated check.
|
|
1651
|
+
python3 - "$WORK/chain-round-two.json" "$WORK/round-two-legacy-envelope.json" <<'PY'
|
|
1652
|
+
import json
|
|
1653
|
+
from pathlib import Path
|
|
1654
|
+
import sys
|
|
1655
|
+
|
|
1656
|
+
# A legacy envelope: previous schema version, and no recorded review scope.
|
|
1657
|
+
two = json.loads(Path(sys.argv[1]).read_text())
|
|
1658
|
+
two["schema_version"] = 2
|
|
1659
|
+
two.pop("review_scope", None)
|
|
1660
|
+
Path(sys.argv[2]).write_text(json.dumps(two, separators=(",", ":")))
|
|
1661
|
+
PY
|
|
1662
|
+
|
|
1663
|
+
# The boolean impersonation only bites when the CURRENT budget is 1 (True == 1),
|
|
1664
|
+
# so the prior must come from a budget-1 chain; reusing the budget-2 prior would
|
|
1665
|
+
# change the current scope digest and trip review_scope_changed first, leaving
|
|
1666
|
+
# the type guard untested.
|
|
1667
|
+
reset_case passed unavailable unavailable
|
|
1668
|
+
budget_one_round_one="$(run_gate --challenge-budget 1 --review-chain-id bool-task --autonomous-review-index 1)"; budget_one_rc=$?
|
|
1669
|
+
printf '%s\n' "$budget_one_round_one" >"$WORK/budget-one-round-one.json"
|
|
1670
|
+
check "a budget-one first tracked round renders before the boolean probe" \
|
|
1671
|
+
'[ "$budget_one_rc" = 0 ] && json_fields "$budget_one_round_one" status=passed challenge_budget=1 autonomous_review_index=1'
|
|
1672
|
+
|
|
1673
|
+
python3 - "$WORK/budget-one-round-one.json" "$WORK/budget-one-bool-budget.json" <<'PY'
|
|
1674
|
+
import json
|
|
1675
|
+
from pathlib import Path
|
|
1676
|
+
import sys
|
|
1677
|
+
|
|
1678
|
+
# A JSON boolean must not satisfy an integer budget: Python's True == 1. The
|
|
1679
|
+
# nested canonical scope is left intact so its digest still verifies.
|
|
1680
|
+
result = json.loads(Path(sys.argv[1]).read_text())
|
|
1681
|
+
result["challenge_budget"] = True
|
|
1682
|
+
Path(sys.argv[2]).write_text(json.dumps(result, separators=(",", ":")))
|
|
1683
|
+
PY
|
|
1684
|
+
reset_case passed unavailable unavailable
|
|
1685
|
+
out="$(run_challenge_gate --challenge-budget 1 --challenge-index 1 --focus bool-budget --review-chain-id bool-task --autonomous-review-index 2 --prior-review-result-file "$WORK/budget-one-bool-budget.json")"; rc=$?
|
|
1686
|
+
check "a JSON boolean cannot impersonate the integer challenge budget of a tracked prior" \
|
|
1687
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
1688
|
+
|
|
1689
|
+
reset_case passed unavailable unavailable
|
|
1690
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/round-two-legacy-envelope.json")"; rc=$?
|
|
1691
|
+
check "a legacy pre-scope envelope is rejected instead of being read as a malformed current result" \
|
|
1692
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=completion_checkpoint_invalid completion_gated=true'
|
|
1693
|
+
|
|
1694
|
+
# The per-invocation budget must cover a real reviewer run: measured lane costs
|
|
1695
|
+
# on this repository's own diffs were roughly 89s, 247s, and 419s, so a default
|
|
1696
|
+
# below those silently converts a working lane into an inconclusive timeout.
|
|
1697
|
+
# Intentionally omit both timeout flags: this is the end-to-end default path,
|
|
1698
|
+
# not only a parser-default assertion.
|
|
1699
|
+
reset_case passed unavailable unavailable
|
|
1700
|
+
out="$(run_gate)"; rc=$?
|
|
1701
|
+
check "the default cumulative budget leaves the default 600-second invocation uncapped" \
|
|
1702
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/claude_timeout")" = 600 ] && json_fields "$out" status=passed selected_client=claude'
|
|
1703
|
+
|
|
1704
|
+
reset_case passed unavailable unavailable
|
|
1705
|
+
out="$(run_gate --timeout 90)"; rc=$?
|
|
1706
|
+
check "an explicit budget still overrides the default" \
|
|
1707
|
+
'[ "$rc" = 0 ] && [ "$(cat "$WORK/state/claude_timeout")" = 90 ]'
|
|
1708
|
+
|
|
1709
|
+
# The gate always passes --timeout, which overrides each wrapper's own default,
|
|
1710
|
+
# so the assertions above cannot see the direct-invocation defaults at all. Pin
|
|
1711
|
+
# each declared default separately or three of the four silently regress.
|
|
1712
|
+
wrapper_defaults_ok=1
|
|
1713
|
+
wrapper_defaults_seen=0
|
|
1714
|
+
for wrapper_default in \
|
|
1715
|
+
'claude_review.sh:^timeout_s=600$' \
|
|
1716
|
+
'codex_review.sh:^TIMEOUT=600$' \
|
|
1717
|
+
'kimi_review.sh:^TIMEOUT=600$' \
|
|
1718
|
+
'opencode_review.sh:^BASE=.*; TIMEOUT="600"$'; do
|
|
1719
|
+
wrapper_file="${wrapper_default%%:*}"
|
|
1720
|
+
wrapper_pattern="${wrapper_default#*:}"
|
|
1721
|
+
if grep -qE "$wrapper_pattern" "$DIR/$wrapper_file"; then
|
|
1722
|
+
wrapper_defaults_seen=$((wrapper_defaults_seen + 1))
|
|
1723
|
+
else
|
|
1724
|
+
wrapper_defaults_ok=0
|
|
1725
|
+
printf 'wrapper default drift: %s does not declare %s\n' "$wrapper_file" "$wrapper_pattern" >&2
|
|
1726
|
+
fi
|
|
1727
|
+
done
|
|
1728
|
+
check "every wrapper declares the same direct-invocation budget as the gate default" \
|
|
1729
|
+
'[ "$wrapper_defaults_ok" = 1 ] && [ "$wrapper_defaults_seen" = 4 ]'
|
|
1730
|
+
|
|
1731
|
+
parser_total_timeout="$(python3 - "$DIR/review_gate.py" "$WORK" <<'PY'
|
|
1732
|
+
import importlib.util
|
|
1733
|
+
from pathlib import Path
|
|
1734
|
+
import sys
|
|
1735
|
+
import time
|
|
1736
|
+
|
|
1737
|
+
spec = importlib.util.spec_from_file_location("review_gate", sys.argv[1])
|
|
1738
|
+
module = importlib.util.module_from_spec(spec)
|
|
1739
|
+
spec.loader.exec_module(module)
|
|
1740
|
+
real_monotonic = module.time.monotonic
|
|
1741
|
+
args = module.build_parser().parse_args([
|
|
1742
|
+
"--mode", "review",
|
|
1743
|
+
"--cwd", "/tmp",
|
|
1744
|
+
"--implementer-family", "openai",
|
|
1745
|
+
"--review-plan-file", "/tmp/plan.json",
|
|
1746
|
+
])
|
|
1747
|
+
print(args.total_timeout)
|
|
1748
|
+
module.time.monotonic = lambda: 95.9
|
|
1749
|
+
print(module.remaining_gate_seconds(100.0))
|
|
1750
|
+
print(
|
|
1751
|
+
module.invocation_timeout_seconds(600, 90, "review"),
|
|
1752
|
+
module.invocation_timeout_seconds(600, 90, "challenge"),
|
|
1753
|
+
module.invocation_timeout_seconds(600, 20, "review"),
|
|
1754
|
+
module.invocation_timeout_seconds(600, 19, "review"),
|
|
1755
|
+
)
|
|
1756
|
+
print(
|
|
1757
|
+
module.reviewer_lane_timeout_seconds(5, 50, "review"),
|
|
1758
|
+
module.reviewer_lane_timeout_seconds(600, 90, "review"),
|
|
1759
|
+
)
|
|
1760
|
+
captured_git_timeout = []
|
|
1761
|
+
real_run = module.run
|
|
1762
|
+
module.run = lambda command, **kwargs: (
|
|
1763
|
+
captured_git_timeout.append(kwargs.get("timeout_seconds"))
|
|
1764
|
+
or module.subprocess.CompletedProcess(command, 0, b"ok", b"")
|
|
1765
|
+
)
|
|
1766
|
+
assert module.git_output(Path(sys.argv[2]), ["status"], deadline=100.0) == b"ok"
|
|
1767
|
+
assert captured_git_timeout == [4]
|
|
1768
|
+
module.time.monotonic = lambda: 100.0
|
|
1769
|
+
try:
|
|
1770
|
+
module.remaining_preflight_seconds(100.0)
|
|
1771
|
+
except module.GateError as exc:
|
|
1772
|
+
assert exc.reason_code == "gate_timeout"
|
|
1773
|
+
else:
|
|
1774
|
+
raise AssertionError("expired preflight budget was accepted")
|
|
1775
|
+
module.run = real_run
|
|
1776
|
+
module.time.monotonic = real_monotonic
|
|
1777
|
+
print("preflight_deadline_ok")
|
|
1778
|
+
|
|
1779
|
+
real_freeze_packet = module.freeze_packet
|
|
1780
|
+
real_emit = module.emit
|
|
1781
|
+
|
|
1782
|
+
def expired_freeze(*args, **kwargs):
|
|
1783
|
+
raise module.GateError(
|
|
1784
|
+
"review gate exhausted its total wall-clock budget", "gate_timeout"
|
|
1785
|
+
)
|
|
1786
|
+
|
|
1787
|
+
module.freeze_packet = expired_freeze
|
|
1788
|
+
module.emit = lambda payload, exit_code: (payload, exit_code)
|
|
1789
|
+
preflight_expired, preflight_expired_code = module.main([
|
|
1790
|
+
"--mode", "review",
|
|
1791
|
+
"--cwd", "/tmp",
|
|
1792
|
+
"--diff-file", "/tmp/diff.patch",
|
|
1793
|
+
"--implementer-family", "openai",
|
|
1794
|
+
"--review-plan-file", "/tmp/plan.json",
|
|
1795
|
+
])
|
|
1796
|
+
assert preflight_expired_code == 2
|
|
1797
|
+
assert preflight_expired["reason_code"] == "gate_timeout"
|
|
1798
|
+
assert preflight_expired["review_state"] == "self_reviewing"
|
|
1799
|
+
assert preflight_expired["completion_gated"] is True
|
|
1800
|
+
assert preflight_expired["self_review_gate"]["required_triggers"] == [
|
|
1801
|
+
"post_review_budget_checkpoint"
|
|
1802
|
+
]
|
|
1803
|
+
assert preflight_expired["self_review_gate"]["blocks"] == [
|
|
1804
|
+
"external_review",
|
|
1805
|
+
"completion_claim",
|
|
1806
|
+
]
|
|
1807
|
+
module.freeze_packet = real_freeze_packet
|
|
1808
|
+
module.emit = real_emit
|
|
1809
|
+
print("preflight_envelope_ok")
|
|
1810
|
+
|
|
1811
|
+
tree_pid_file = Path(sys.argv[2]) / "state" / "fallback_tree_child_pid"
|
|
1812
|
+
tree_process = module.subprocess.Popen(
|
|
1813
|
+
[
|
|
1814
|
+
sys.executable,
|
|
1815
|
+
"-c",
|
|
1816
|
+
"""
|
|
1817
|
+
import os
|
|
1818
|
+
import signal
|
|
1819
|
+
import sys
|
|
1820
|
+
|
|
1821
|
+
signal.signal(signal.SIGTERM, signal.SIG_IGN)
|
|
1822
|
+
child_pid = os.fork()
|
|
1823
|
+
if child_pid == 0:
|
|
1824
|
+
with open(sys.argv[1], "w", encoding="utf-8") as pid_file:
|
|
1825
|
+
pid_file.write(str(os.getpid()))
|
|
1826
|
+
while True:
|
|
1827
|
+
signal.pause()
|
|
1828
|
+
while True:
|
|
1829
|
+
signal.pause()
|
|
1830
|
+
""",
|
|
1831
|
+
str(tree_pid_file),
|
|
1832
|
+
],
|
|
1833
|
+
stdout=module.subprocess.PIPE,
|
|
1834
|
+
stderr=module.subprocess.PIPE,
|
|
1835
|
+
start_new_session=True,
|
|
1836
|
+
)
|
|
1837
|
+
tree_child_text = ""
|
|
1838
|
+
for _ in range(100):
|
|
1839
|
+
if tree_pid_file.exists():
|
|
1840
|
+
tree_child_text = tree_pid_file.read_text().strip()
|
|
1841
|
+
if tree_child_text:
|
|
1842
|
+
break
|
|
1843
|
+
time.sleep(0.01)
|
|
1844
|
+
assert tree_child_text
|
|
1845
|
+
tree_child_pid = int(tree_child_text)
|
|
1846
|
+
real_killpg = module.os.killpg
|
|
1847
|
+
killpg_calls = 0
|
|
1848
|
+
|
|
1849
|
+
def transient_killpg(pgid, signal_value):
|
|
1850
|
+
global killpg_calls
|
|
1851
|
+
killpg_calls += 1
|
|
1852
|
+
if killpg_calls <= 2:
|
|
1853
|
+
raise PermissionError()
|
|
1854
|
+
real_killpg(pgid, signal_value)
|
|
1855
|
+
|
|
1856
|
+
module.os.killpg = transient_killpg
|
|
1857
|
+
module.signal_reviewer_process_group(
|
|
1858
|
+
tree_process, module.signal.SIGKILL, tree_process.pid
|
|
1859
|
+
)
|
|
1860
|
+
tree_process.communicate(timeout=2)
|
|
1861
|
+
tree_child_gone = False
|
|
1862
|
+
for _ in range(100):
|
|
1863
|
+
try:
|
|
1864
|
+
module.os.kill(tree_child_pid, 0)
|
|
1865
|
+
except ProcessLookupError:
|
|
1866
|
+
tree_child_gone = True
|
|
1867
|
+
break
|
|
1868
|
+
time.sleep(0.01)
|
|
1869
|
+
assert tree_child_gone
|
|
1870
|
+
print("tree_fallback_ok")
|
|
1871
|
+
|
|
1872
|
+
class TimedOutProcess:
|
|
1873
|
+
pid = 12345
|
|
1874
|
+
returncode = None
|
|
1875
|
+
terminated = False
|
|
1876
|
+
killed = False
|
|
1877
|
+
polled = False
|
|
1878
|
+
waited = False
|
|
1879
|
+
communicate_calls = 0
|
|
1880
|
+
|
|
1881
|
+
class Pipe:
|
|
1882
|
+
closed = False
|
|
1883
|
+
|
|
1884
|
+
def close(self):
|
|
1885
|
+
self.closed = True
|
|
1886
|
+
|
|
1887
|
+
stdout = Pipe()
|
|
1888
|
+
stderr = Pipe()
|
|
1889
|
+
|
|
1890
|
+
def communicate(self, timeout):
|
|
1891
|
+
self.communicate_calls += 1
|
|
1892
|
+
if self.communicate_calls == 1:
|
|
1893
|
+
raise module.subprocess.TimeoutExpired(["reviewer"], timeout)
|
|
1894
|
+
raise OSError("cleanup pipe race")
|
|
1895
|
+
|
|
1896
|
+
def terminate(self):
|
|
1897
|
+
self.terminated = True
|
|
1898
|
+
|
|
1899
|
+
def kill(self):
|
|
1900
|
+
self.killed = True
|
|
1901
|
+
|
|
1902
|
+
def poll(self):
|
|
1903
|
+
self.polled = True
|
|
1904
|
+
return -9
|
|
1905
|
+
|
|
1906
|
+
def wait(self, timeout):
|
|
1907
|
+
self.waited = True
|
|
1908
|
+
raise module.subprocess.TimeoutExpired(["reviewer"], timeout)
|
|
1909
|
+
|
|
1910
|
+
timed_out_process = TimedOutProcess()
|
|
1911
|
+
module.subprocess.Popen = lambda *args, **kwargs: timed_out_process
|
|
1912
|
+
module.os.killpg = lambda *args, **kwargs: (_ for _ in ()).throw(PermissionError())
|
|
1913
|
+
try:
|
|
1914
|
+
module.run(["reviewer"], timeout_seconds=5)
|
|
1915
|
+
except module.GateError as exc:
|
|
1916
|
+
print(exc.reason_code)
|
|
1917
|
+
print(
|
|
1918
|
+
timed_out_process.terminated,
|
|
1919
|
+
timed_out_process.killed,
|
|
1920
|
+
timed_out_process.waited,
|
|
1921
|
+
timed_out_process.polled,
|
|
1922
|
+
timed_out_process.stdout.closed,
|
|
1923
|
+
timed_out_process.stderr.closed,
|
|
1924
|
+
)
|
|
1925
|
+
|
|
1926
|
+
class PostLaunchIoErrorProcess:
|
|
1927
|
+
pid = 12346
|
|
1928
|
+
returncode = None
|
|
1929
|
+
stdout = None
|
|
1930
|
+
stderr = None
|
|
1931
|
+
|
|
1932
|
+
def communicate(self, timeout):
|
|
1933
|
+
raise OSError("reviewer pipe failed")
|
|
1934
|
+
|
|
1935
|
+
module.subprocess.Popen = lambda *args, **kwargs: PostLaunchIoErrorProcess()
|
|
1936
|
+
try:
|
|
1937
|
+
module.run(["reviewer"], timeout_seconds=5)
|
|
1938
|
+
except module.GateError as exc:
|
|
1939
|
+
assert exc.reason_code == "local_process_io_failure"
|
|
1940
|
+
assert "subprocess I/O failed after starting" in exc.reason
|
|
1941
|
+
else:
|
|
1942
|
+
raise AssertionError("post-launch reviewer I/O failure was accepted")
|
|
1943
|
+
print("post_launch_io_ok")
|
|
1944
|
+
|
|
1945
|
+
class TermGraceProcess:
|
|
1946
|
+
pid = 12347
|
|
1947
|
+
returncode = None
|
|
1948
|
+
stdout = None
|
|
1949
|
+
stderr = None
|
|
1950
|
+
communicate_calls = 0
|
|
1951
|
+
|
|
1952
|
+
def communicate(self, timeout):
|
|
1953
|
+
self.communicate_calls += 1
|
|
1954
|
+
if self.communicate_calls == 1:
|
|
1955
|
+
raise module.subprocess.TimeoutExpired(["reviewer"], timeout)
|
|
1956
|
+
return b"", b""
|
|
1957
|
+
|
|
1958
|
+
term_grace_process = TermGraceProcess()
|
|
1959
|
+
signals = []
|
|
1960
|
+
module.subprocess.Popen = lambda *args, **kwargs: term_grace_process
|
|
1961
|
+
module.signal_reviewer_process_group = (
|
|
1962
|
+
lambda process, signal_value, recorded_pgid=None: signals.append(signal_value)
|
|
1963
|
+
)
|
|
1964
|
+
try:
|
|
1965
|
+
module.run(["reviewer"], timeout_seconds=5)
|
|
1966
|
+
except module.GateError as exc:
|
|
1967
|
+
assert exc.reason_code == "gate_timeout"
|
|
1968
|
+
else:
|
|
1969
|
+
raise AssertionError("timed-out reviewer was accepted")
|
|
1970
|
+
assert signals == [module.signal.SIGTERM, module.signal.SIGKILL]
|
|
1971
|
+
print("kill_after_term_grace_ok")
|
|
1972
|
+
|
|
1973
|
+
module.time.monotonic = lambda: 101.0
|
|
1974
|
+
module.emit = lambda payload, exit_code: (payload, exit_code)
|
|
1975
|
+
expired, expired_code = module.emit_with_gate_deadline(
|
|
1976
|
+
{
|
|
1977
|
+
"status": "findings",
|
|
1978
|
+
"selected_client": "claude",
|
|
1979
|
+
"selected_reviewer": "claude",
|
|
1980
|
+
"selected_attempt_index": 0,
|
|
1981
|
+
"findings": [{"severity": "P1"}],
|
|
1982
|
+
"concern_results": [{"concern": "correctness"}],
|
|
1983
|
+
"reviewed_concerns": ["correctness"],
|
|
1984
|
+
"reviewed_skills": ["code-review"],
|
|
1985
|
+
"findings_require_implementer_self_review": True,
|
|
1986
|
+
"human_decision_required": True,
|
|
1987
|
+
"review_state": "reviewed",
|
|
1988
|
+
"completion_gated": False,
|
|
1989
|
+
},
|
|
1990
|
+
0,
|
|
1991
|
+
100.0,
|
|
1992
|
+
)
|
|
1993
|
+
assert expired_code == 2
|
|
1994
|
+
assert expired["status"] == "inconclusive"
|
|
1995
|
+
assert expired["reason_code"] == "gate_timeout"
|
|
1996
|
+
assert expired["fallback_eligible"] is False
|
|
1997
|
+
assert expired["selected_client"] is None
|
|
1998
|
+
assert expired["findings"] == []
|
|
1999
|
+
assert expired["unbound_findings"] == [{"severity": "P1"}]
|
|
2000
|
+
assert expired["concern_results"] == []
|
|
2001
|
+
assert expired["findings_require_implementer_self_review"] is False
|
|
2002
|
+
assert expired["human_decision_required"] is False
|
|
2003
|
+
assert expired["review_state"] == "self_reviewing"
|
|
2004
|
+
assert expired["completion_gated"] is True
|
|
2005
|
+
assert expired["self_review_gate"]["required"] is True
|
|
2006
|
+
assert expired["self_review_gate"]["required_triggers"] == [
|
|
2007
|
+
"post_review_budget_checkpoint"
|
|
2008
|
+
]
|
|
2009
|
+
assert expired["self_review_gate"]["blocks"] == [
|
|
2010
|
+
"external_review",
|
|
2011
|
+
"completion_claim",
|
|
2012
|
+
]
|
|
2013
|
+
module.time.monotonic = lambda: 99.1
|
|
2014
|
+
near_deadline, near_deadline_code = module.emit_with_gate_deadline(
|
|
2015
|
+
{"status": "passed"}, 0, 100.0
|
|
2016
|
+
)
|
|
2017
|
+
assert near_deadline_code == 0
|
|
2018
|
+
assert near_deadline["status"] == "passed"
|
|
2019
|
+
print("post_deadline_ok")
|
|
2020
|
+
PY
|
|
2021
|
+
)"
|
|
2022
|
+
check "the gate declares a finite cumulative default and floors the remaining budget" \
|
|
2023
|
+
'[ "$parser_total_timeout" = "2400
|
|
2024
|
+
4
|
|
2025
|
+
40 80 5 4
|
|
2026
|
+
20 90
|
|
2027
|
+
preflight_deadline_ok
|
|
2028
|
+
preflight_envelope_ok
|
|
2029
|
+
tree_fallback_ok
|
|
2030
|
+
gate_timeout
|
|
2031
|
+
True True True True True True
|
|
2032
|
+
post_launch_io_ok
|
|
2033
|
+
kill_after_term_grace_ok
|
|
2034
|
+
post_deadline_ok" ]'
|
|
2035
|
+
|
|
2036
|
+
check "the staged contract declares the current result schema" \
|
|
2037
|
+
'grep -q "current result envelope is schema 3" "$DIR/../references/staged-review-contract.md"'
|
|
2038
|
+
|
|
2039
|
+
reset_case passed unavailable unavailable
|
|
2040
|
+
out="$(run_gate --timeout 600 --total-timeout 90)"; rc=$?
|
|
2041
|
+
claude_timeout="$(cat "$WORK/state/claude_timeout" 2>/dev/null || true)"
|
|
2042
|
+
check "the remaining total budget caps the wrapper timeout" \
|
|
2043
|
+
'[ "$rc" = 0 ] && [ "$claude_timeout" -ge 5 ] && [ "$claude_timeout" -le 40 ]'
|
|
2044
|
+
|
|
2045
|
+
reset_case quota_slow passed unavailable unavailable
|
|
2046
|
+
out="$(run_gate --allow-fallback-egress --total-timeout 25)"; rc=$?
|
|
2047
|
+
kimi_timeout="$(cat "$WORK/state/kimi_timeout" 2>/dev/null || true)"
|
|
2048
|
+
check "a fallback receives only the remaining cumulative budget" \
|
|
2049
|
+
'[ "$rc" = 0 ] && [ "$kimi_timeout" -ge 5 ] && [ "$kimi_timeout" -lt 12 ] && json_fields "$out" selected_client=kimi'
|
|
2050
|
+
|
|
2051
|
+
reset_case quota_slow hang unavailable unavailable
|
|
2052
|
+
out="$(run_gate --allow-fallback-egress --total-timeout 25)"; rc=$?
|
|
2053
|
+
kimi_timeout="$(cat "$WORK/state/kimi_timeout" 2>/dev/null || true)"
|
|
2054
|
+
check "a started fallback that exhausts the remaining budget cannot starve a later lane" \
|
|
2055
|
+
'[ "$rc" = 2 ] && [ "$kimi_timeout" -ge 5 ] && [ "$kimi_timeout" -lt 12 ] && [ ! -e "$WORK/state/opencode_timeout" ] && json_fields "$out" status=inconclusive fallback_eligible=false reason_code=gate_timeout next_action=stop_reviewer_lane'
|
|
2056
|
+
|
|
2057
|
+
reset_case quota_slow passed unavailable unavailable
|
|
2058
|
+
out="$(run_gate --allow-fallback-egress --total-timeout 5)"; rc=$?
|
|
2059
|
+
check "the gate does not start a fallback with less than the wrapper minimum remaining" \
|
|
2060
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/kimi_timeout" ] && json_fields "$out" status=inconclusive fallback_eligible=false reason_code=gate_timeout review_state=self_reviewing completion_gated=true self_review_gate.required=true self_review_gate.required_triggers.0=post_review_budget_checkpoint self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim next_action=stop_reviewer_lane'
|
|
2061
|
+
|
|
2062
|
+
reset_case quota_slow passed unavailable unavailable
|
|
2063
|
+
out="$(run_gate --allow-fallback-egress --total-timeout 22)"; rc=$?
|
|
2064
|
+
check "a primary may consume the usable budget without starting a doomed fallback" \
|
|
2065
|
+
'[ "$rc" = 2 ] && [ -e "$WORK/state/claude_timeout" ] && [ ! -e "$WORK/state/kimi_timeout" ] && json_fields "$out" status=inconclusive fallback_eligible=false reason_code=gate_timeout next_action=stop_reviewer_lane'
|
|
2066
|
+
|
|
2067
|
+
reset_case hang passed unavailable unavailable
|
|
2068
|
+
out="$(run_gate --diff-file "$WORK/secret-diff.patch" --timeout 5 --total-timeout 50)"; rc=$?
|
|
2069
|
+
hang_child_pid="$(cat "$WORK/state/hang_child_pid" 2>/dev/null || true)"
|
|
2070
|
+
hang_child_gone=0
|
|
2071
|
+
if [ -n "$hang_child_pid" ]; then
|
|
2072
|
+
if ! kill -0 "$hang_child_pid" 2>/dev/null; then
|
|
2073
|
+
hang_child_gone=1
|
|
2074
|
+
elif ps -o stat= -p "$hang_child_pid" 2>/dev/null | grep -q '^Z'; then
|
|
2075
|
+
hang_child_gone=1
|
|
2076
|
+
fi
|
|
2077
|
+
fi
|
|
2078
|
+
check "a timed-out primary cannot bypass fallback egress authorization" \
|
|
2079
|
+
'[ "$rc" = 2 ] && [ "$hang_child_gone" = 1 ] && [ ! -e "$WORK/state/kimi_timeout" ] && [ ! -e "$WORK/state/opencode_timeout" ] && json_fields "$out" status=inconclusive reason_code=egress_denied next_action=stop_reviewer_lane'
|
|
2080
|
+
|
|
2081
|
+
reset_case hang passed unavailable unavailable
|
|
2082
|
+
out="$(run_gate --allow-fallback-egress --timeout 5 --total-timeout 50)"; rc=$?
|
|
2083
|
+
hang_child_pid="$(cat "$WORK/state/hang_child_pid" 2>/dev/null || true)"
|
|
2084
|
+
hang_child_gone=0
|
|
2085
|
+
if [ -n "$hang_child_pid" ]; then
|
|
2086
|
+
if ! kill -0 "$hang_child_pid" 2>/dev/null; then
|
|
2087
|
+
hang_child_gone=1
|
|
2088
|
+
elif ps -o stat= -p "$hang_child_pid" 2>/dev/null | grep -q '^Z'; then
|
|
2089
|
+
hang_child_gone=1
|
|
2090
|
+
fi
|
|
2091
|
+
fi
|
|
2092
|
+
check "a timed-out primary lane leaves total budget for an independent fallback" \
|
|
2093
|
+
'[ "$rc" = 0 ] && [ "$hang_child_gone" = 1 ] && [ -e "$WORK/state/kimi_timeout" ] && json_fields "$out" status=passed selected_client=kimi && [ "$(printf "%s" "$out" | python3 -c "import json,sys; value=json.load(sys.stdin); print(len(value[\"attempts\"]), sum(item.get(\"reason_code\") == \"timeout\" for item in value[\"skipped_clients\"]))")" = "2 1" ]'
|
|
2094
|
+
|
|
2095
|
+
reset_case hang passed unavailable unavailable
|
|
2096
|
+
out="$(run_challenge_gate --allow-fallback-egress --focus process-group-timeout --total-timeout 16)"; rc=$?
|
|
2097
|
+
hang_child_pid="$(cat "$WORK/state/hang_child_pid" 2>/dev/null || true)"
|
|
2098
|
+
hang_child_gone=0
|
|
2099
|
+
if [ -n "$hang_child_pid" ]; then
|
|
2100
|
+
if ! kill -0 "$hang_child_pid" 2>/dev/null; then
|
|
2101
|
+
hang_child_gone=1
|
|
2102
|
+
elif ps -o stat= -p "$hang_child_pid" 2>/dev/null | grep -q '^Z'; then
|
|
2103
|
+
hang_child_gone=1
|
|
2104
|
+
fi
|
|
2105
|
+
fi
|
|
2106
|
+
hang_envelope_ok=0
|
|
2107
|
+
if json_fields "$out" status=inconclusive fallback_eligible=false reason_code=gate_timeout next_action=stop_reviewer_lane; then
|
|
2108
|
+
hang_envelope_ok=1
|
|
2109
|
+
fi
|
|
2110
|
+
if [ "$rc" != 2 ] || [ "$hang_child_gone" != 1 ] || [ -e "$WORK/state/kimi_timeout" ] || [ "$hang_envelope_ok" != 1 ]; then
|
|
2111
|
+
printf 'hang diagnostic: rc=%s child_pid=%s child_gone=%s kimi_started=%s envelope_ok=%s out=%s\n' \
|
|
2112
|
+
"$rc" "$hang_child_pid" "$hang_child_gone" "$([ -e "$WORK/state/kimi_timeout" ] && printf yes || printf no)" "$hang_envelope_ok" "$out" >&2
|
|
2113
|
+
fi
|
|
2114
|
+
check "the controller terminates an over-budget wrapper process group" \
|
|
2115
|
+
'[ "$rc" = 2 ] && [ "$hang_child_gone" = 1 ] && [ ! -e "$WORK/state/kimi_timeout" ] && [ "$hang_envelope_ok" = 1 ]'
|
|
2116
|
+
|
|
2117
|
+
reset_case escaped_hang passed unavailable unavailable
|
|
2118
|
+
escaped_started_at="$(date +%s)"
|
|
2119
|
+
# Run the gate asynchronously under a watchdog independent of it: a regressed
|
|
2120
|
+
# envelope that waits for the descendant's pipes would otherwise park this case
|
|
2121
|
+
# for that descendant's whole lifetime before any elapsed check could fail.
|
|
2122
|
+
escaped_out_file="$WORK/escaped-out.json"
|
|
2123
|
+
run_challenge_gate --allow-fallback-egress --focus escaped-pipe-timeout --total-timeout 16 >"$escaped_out_file" 2>/dev/null &
|
|
2124
|
+
escaped_gate_pid=$!
|
|
2125
|
+
escaped_gate_deadline="$(( escaped_started_at + 120 ))"
|
|
2126
|
+
escaped_watchdog_fired=0
|
|
2127
|
+
while kill -0 "$escaped_gate_pid" 2>/dev/null; do
|
|
2128
|
+
if [ "$(date +%s)" -ge "$escaped_gate_deadline" ]; then
|
|
2129
|
+
escaped_watchdog_fired=1
|
|
2130
|
+
kill -KILL "$escaped_gate_pid" 2>/dev/null || true
|
|
2131
|
+
break
|
|
2132
|
+
fi
|
|
2133
|
+
sleep 1
|
|
2134
|
+
done
|
|
2135
|
+
wait "$escaped_gate_pid"; rc=$?
|
|
2136
|
+
escaped_returned_at="$(date +%s)"
|
|
2137
|
+
# Drop the marker the descendant is watching for BEFORE anything else, so what it
|
|
2138
|
+
# witnesses is the gate's return and not this test's bookkeeping.
|
|
2139
|
+
: >"$WORK/state/escaped_gate_returned"
|
|
2140
|
+
out="$(cat "$escaped_out_file" 2>/dev/null || true)"
|
|
2141
|
+
escaped_elapsed="$(( $(date +%s) - escaped_started_at ))"
|
|
2142
|
+
escaped_child_pid="$(cat "$WORK/state/escaped_hang_child_pid" 2>/dev/null || true)"
|
|
2143
|
+
escaped_child_trap_pid="$escaped_child_pid"
|
|
2144
|
+
# Read the beat BEFORE the liveness probe so a descendant that dies between the two
|
|
2145
|
+
# still reports how far it got; empty means it never reached its own setsid.
|
|
2146
|
+
# Wait for the descendant to TESTIFY, not for time to pass: a live holder stamps
|
|
2147
|
+
# the witness within one beat interval, so this loop ends immediately on a healthy
|
|
2148
|
+
# run and only burns its bound when there is nothing alive to answer — which is
|
|
2149
|
+
# itself the failure this case must report.
|
|
2150
|
+
escaped_witness_deadline="$(( $(date +%s) + 5 ))"
|
|
2151
|
+
while :; do
|
|
2152
|
+
escaped_child_beat="$(tr -d '\n' <"$WORK/state/escaped_hang_child_beat" 2>/dev/null || true)"
|
|
2153
|
+
case "$escaped_child_beat" in *witnessed=1*) break ;; esac
|
|
2154
|
+
[ "$(date +%s)" -ge "$escaped_witness_deadline" ] && break
|
|
2155
|
+
sleep 1
|
|
2156
|
+
done
|
|
2157
|
+
escaped_test_sid="$(ps -o sess= -p $$ 2>/dev/null | tr -d ' ' || true)"
|
|
2158
|
+
escaped_wrapper_pid="$(cat "$WORK/state/escaped_hang_wrapper_pid" 2>/dev/null || true)"
|
|
2159
|
+
# What this case must prove is that the terminal envelope did not WAIT for the
|
|
2160
|
+
# escaped descendant to release the reviewer pipes, and two facts already prove it
|
|
2161
|
+
# without asking anything of the runner's timing: the descendant really escaped its
|
|
2162
|
+
# wrapper's session (so it still held those pipes), and the gate came back on its
|
|
2163
|
+
# own — watchdog_fired=0 means it returned inside 120s while the descendant's own
|
|
2164
|
+
# lifetime is 300s, so a gate that waited on the pipes could not have produced this
|
|
2165
|
+
# run. Detachment is read structurally from the beat the descendant wrote after its
|
|
2166
|
+
# own setsid: its session id equals its pid exactly when setsid took effect.
|
|
2167
|
+
#
|
|
2168
|
+
# The two oracles this replaces both asserted things no runner owes us. Probing
|
|
2169
|
+
# liveness *now* assumes nothing reaps a detached process between the gate's return
|
|
2170
|
+
# and this line — a host that reaps promptly then fails a gate that behaved
|
|
2171
|
+
# correctly (observed in CI: child gone, wrapper gone, envelope correct). Comparing
|
|
2172
|
+
# the last beat's clock against the return instant fails for a subtler reason: the
|
|
2173
|
+
# beat lags by up to its interval, so at one-second resolution a healthy run and a
|
|
2174
|
+
# descendant killed with its wrapper are indistinguishable — both land one second
|
|
2175
|
+
# before the return (measured: control at=...987 vs return ...988; mutant at=...817
|
|
2176
|
+
# vs return ...818). Sub-second stamps would only narrow that gap, not close it.
|
|
2177
|
+
escaped_child_beat_pid="${escaped_child_beat##* pid=}"
|
|
2178
|
+
escaped_child_beat_pid="${escaped_child_beat_pid%% *}"
|
|
2179
|
+
escaped_child_beat_sid="${escaped_child_beat##* sid=}"
|
|
2180
|
+
escaped_child_beat_sid="${escaped_child_beat_sid%% *}"
|
|
2181
|
+
escaped_child_detached=0
|
|
2182
|
+
case "$escaped_child_beat_pid$escaped_child_beat_sid" in
|
|
2183
|
+
''|*[!0-9]*) : ;;
|
|
2184
|
+
*) [ "$escaped_child_beat_sid" = "$escaped_child_beat_pid" ] && escaped_child_detached=1 ;;
|
|
2185
|
+
esac
|
|
2186
|
+
# Detachment alone would let the case pass without exercising itself: a controller
|
|
2187
|
+
# that kills the detached descendant and only THEN waits for the reviewer pipes
|
|
2188
|
+
# gets EOF immediately, returns well inside the watchdog, and satisfies every other
|
|
2189
|
+
# condition here — the exact waiting defect this case exists to catch, wearing a
|
|
2190
|
+
# green badge. The witness is what makes the pipes provably still held at return.
|
|
2191
|
+
escaped_child_witnessed=0
|
|
2192
|
+
case "$escaped_child_beat" in *witnessed=1*) escaped_child_witnessed=1 ;; esac
|
|
2193
|
+
escaped_child_alive=0
|
|
2194
|
+
escaped_child_stat=""
|
|
2195
|
+
if [ -n "$escaped_child_pid" ] && kill -0 "$escaped_child_pid" 2>/dev/null; then
|
|
2196
|
+
escaped_child_stat="$(ps -o stat= -p "$escaped_child_pid" 2>/dev/null || true)"
|
|
2197
|
+
case "$escaped_child_stat" in
|
|
2198
|
+
""|*Z*) : ;;
|
|
2199
|
+
*) escaped_child_alive=1 ;;
|
|
2200
|
+
esac
|
|
2201
|
+
fi
|
|
2202
|
+
# The controller signals the wrapper's process group, but reaping the corpse is
|
|
2203
|
+
# the OS's business and lags on a loaded runner: an unreaped zombie still answers
|
|
2204
|
+
# kill -0, so treat it as gone and give the reap a bounded grace period. Both stay
|
|
2205
|
+
# far below the descendant lifetime, so a wrapper the controller never killed
|
|
2206
|
+
# (it would outlive this loop) is still a hard failure.
|
|
2207
|
+
escaped_wrapper_gone=0
|
|
2208
|
+
escaped_wrapper_stat=""
|
|
2209
|
+
if [ -n "$escaped_wrapper_pid" ]; then
|
|
2210
|
+
escaped_wrapper_deadline="$(( $(date +%s) + 15 ))"
|
|
2211
|
+
while :; do
|
|
2212
|
+
if ! kill -0 "$escaped_wrapper_pid" 2>/dev/null; then
|
|
2213
|
+
escaped_wrapper_gone=1
|
|
2214
|
+
break
|
|
2215
|
+
fi
|
|
2216
|
+
escaped_wrapper_stat="$(ps -o stat= -p "$escaped_wrapper_pid" 2>/dev/null || true)"
|
|
2217
|
+
case "$escaped_wrapper_stat" in
|
|
2218
|
+
*Z*) escaped_wrapper_gone=1; break ;;
|
|
2219
|
+
esac
|
|
2220
|
+
[ "$(date +%s)" -ge "$escaped_wrapper_deadline" ] && break
|
|
2221
|
+
sleep 1
|
|
2222
|
+
done
|
|
2223
|
+
fi
|
|
2224
|
+
# Release the descendant only after that poll: while it runs the descendant must
|
|
2225
|
+
# still hold the reviewer pipes, or the wrapper would exit on its own and a
|
|
2226
|
+
# controller that never killed it would read as gone. Clearing the trap handle
|
|
2227
|
+
# right after keeps a much later EXIT from killing whatever inherited this pid.
|
|
2228
|
+
[ -z "$escaped_child_pid" ] || kill -KILL "$escaped_child_pid" 2>/dev/null || true
|
|
2229
|
+
escaped_child_trap_pid=""
|
|
2230
|
+
rm -f "$WORK/state/escaped_hang_child_pid" "$WORK/state/escaped_hang_child_beat" "$WORK/state/escaped_gate_returned"
|
|
2231
|
+
escaped_envelope_ok=0
|
|
2232
|
+
if json_fields "$out" status=inconclusive fallback_eligible=false reason_code=gate_timeout next_action=stop_reviewer_lane; then
|
|
2233
|
+
escaped_envelope_ok=1
|
|
2234
|
+
fi
|
|
2235
|
+
if [ "$rc" != 2 ] || [ "$escaped_child_detached" != 1 ] || [ "$escaped_child_witnessed" != 1 ] || [ "$escaped_watchdog_fired" != 0 ] || [ "$escaped_wrapper_gone" != 1 ] || [ -e "$WORK/state/kimi_timeout" ] || [ "$escaped_envelope_ok" != 1 ]; then
|
|
2236
|
+
printf 'escaped diagnostic: rc=%s elapsed=%s watchdog_fired=%s child_pid=%s child_alive=%s child_detached=%s child_witnessed=%s returned_at=%s child_stat=%s child_beat=%s test_sid=%s wrapper_pid=%s wrapper_gone=%s wrapper_stat=%s kimi_started=%s envelope_ok=%s out=%s\n' \
|
|
2237
|
+
"$rc" "$escaped_elapsed" "$escaped_watchdog_fired" "$escaped_child_pid" "$escaped_child_alive" "$escaped_child_detached" "$escaped_child_witnessed" "$escaped_returned_at" "[$escaped_child_stat]" "[$escaped_child_beat]" "$escaped_test_sid" "$escaped_wrapper_pid" "$escaped_wrapper_gone" "[$escaped_wrapper_stat]" \
|
|
2238
|
+
"$([ -e "$WORK/state/kimi_timeout" ] && printf yes || printf no)" "$escaped_envelope_ok" "$out" >&2
|
|
2239
|
+
fi
|
|
2240
|
+
check "an escaped descendant holding reviewer pipes cannot block the terminal envelope" \
|
|
2241
|
+
'[ "$rc" = 2 ] && [ "$escaped_child_detached" = 1 ] && [ "$escaped_child_witnessed" = 1 ] && [ "$escaped_watchdog_fired" = 0 ] && [ "$escaped_wrapper_gone" = 1 ] && [ ! -e "$WORK/state/kimi_timeout" ] && [ "$escaped_envelope_ok" = 1 ]'
|
|
2242
|
+
|
|
2243
|
+
reset_case passed_slow passed unavailable unavailable
|
|
2244
|
+
out="$(run_challenge_gate --allow-fallback-egress --focus deadline-edge-verdict --total-timeout 16)"; rc=$?
|
|
2245
|
+
check "controller headroom preserves a valid challenge verdict near the inner deadline" \
|
|
2246
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=passed selected_client=claude'
|
|
2247
|
+
|
|
2248
|
+
reset_case passed unavailable unavailable
|
|
2249
|
+
out="$(run_gate --total-timeout 4)"; rc=$?
|
|
2250
|
+
check "an out-of-range cumulative budget fails before provider execution" \
|
|
2251
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
2252
|
+
|
|
2253
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-c\n+d\n' >"$WORK/diff.patch"
|
|
2254
|
+
reset_case passed unavailable unavailable
|
|
2255
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus final-surface --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
2256
|
+
check "a changed third candidate reaches the last Agent round without resetting budget" \
|
|
2257
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_chain_tracked=true review_chain_id=long-task autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false prior_challenge_focuses.0=corrected-path'
|
|
2258
|
+
printf '%s\n' "$out" >"$WORK/chain-final.json"
|
|
2259
|
+
|
|
2260
|
+
reset_case passed unavailable unavailable
|
|
2261
|
+
out="$(run_completion_gate --challenge-budget 2 --completion-review-result-file "$WORK/chain-final.json")"; rc=$?
|
|
2262
|
+
check "completion preserves the verified tracked review-chain evidence" \
|
|
2263
|
+
'[ "$rc" = 0 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" mode=complete status=passed review_chain_tracked=true review_chain_id=long-task autonomous_review_budget=3 autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false prior_challenge_focuses.0=corrected-path completion_gated=false'
|
|
2264
|
+
|
|
2265
|
+
reset_case findings unavailable unavailable
|
|
2266
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus final-findings --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
2267
|
+
check "findings in the last tracked Agent round return to a post-budget checkpoint" \
|
|
2268
|
+
'[ "$rc" = 0 ] && json_fields "$out" status=findings review_chain_tracked=true autonomous_review_index=3 autonomous_reviews_remaining=0 autonomous_review_allowed=false human_decision_required=true review_state=post_review_budget findings_require_implementer_self_review=true next_action=triage_findings_and_continue_independent_work self_review_gate.required=true self_review_gate.required_triggers.0=findings_returned self_review_gate.required_triggers.1=post_review_budget_checkpoint self_review_gate.blocks.0=external_review self_review_gate.allowed_next_actions.2=continue_independent_work'
|
|
2269
|
+
|
|
2270
|
+
reset_case passed unavailable unavailable
|
|
2271
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus missing-history --review-chain-id long-task --autonomous-review-index 2)"; rc=$?
|
|
2272
|
+
check "a tracked Agent round rejects omitted prior results before provider execution" \
|
|
2273
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
2274
|
+
|
|
2275
|
+
# In a tracked chain the challenge index is not free: the chain requires
|
|
2276
|
+
# autonomous_review_index == challenge_index + 1, so an omitted flag has exactly
|
|
2277
|
+
# one legal value to resolve to, and a chain declared as Agent round 1 still has
|
|
2278
|
+
# none.
|
|
2279
|
+
reset_case passed unavailable unavailable
|
|
2280
|
+
out="$(challenge_gate_no_index --challenge-budget 2 --focus derived-tracked --review-chain-id long-task --autonomous-review-index 2 --prior-review-result-file "$WORK/chain-round-one.json")"; rc=$?
|
|
2281
|
+
check "a tracked challenge derives its index from the Agent review index" \
|
|
2282
|
+
'[ "$rc" = 0 ] && json_fields "$out" review_chain_tracked=true autonomous_review_index=2 challenge_index=1'
|
|
2283
|
+
|
|
2284
|
+
reset_case passed unavailable unavailable
|
|
2285
|
+
out="$(challenge_gate_no_index --challenge-budget 2 --focus derived-tracked-last --review-chain-id long-task --autonomous-review-index 3 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
2286
|
+
check "the derived tracked index follows the Agent review index past the first challenge" \
|
|
2287
|
+
'[ "$rc" = 0 ] && json_fields "$out" autonomous_review_index=3 challenge_index=2'
|
|
2288
|
+
|
|
2289
|
+
reset_case passed unavailable unavailable
|
|
2290
|
+
out="$(challenge_gate_no_index --challenge-budget 2 --focus derived-round-one --review-chain-id long-task --autonomous-review-index 1)"; rc=$?
|
|
2291
|
+
check "a tracked challenge declared as Agent round 1 still fails on the derived index" \
|
|
2292
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
2293
|
+
|
|
2294
|
+
reset_case passed unavailable unavailable
|
|
2295
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 1 --focus changed-chain --review-chain-id renamed-task --autonomous-review-index 2 --prior-review-result-file "$WORK/chain-round-one.json")"; rc=$?
|
|
2296
|
+
check "renaming the review chain cannot reset a consumed Agent round" \
|
|
2297
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
2298
|
+
|
|
2299
|
+
reset_case passed unavailable unavailable
|
|
2300
|
+
out="$(run_challenge_gate --challenge-budget 2 --challenge-index 2 --focus fourth-round --review-chain-id long-task --autonomous-review-index 4 --prior-review-result-file "$WORK/chain-round-one.json" --prior-review-result-file "$WORK/chain-round-two.json")"; rc=$?
|
|
2301
|
+
check "an Agent review beyond its configured chain budget is rejected" \
|
|
2302
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
2303
|
+
|
|
2304
|
+
reset_case passed unavailable unavailable
|
|
2305
|
+
max_five_round_one="$(run_gate --challenge-budget 4 --review-chain-id max-five-task --autonomous-review-index 1)"; max_five_round_one_rc=$?
|
|
2306
|
+
printf '%s\n' "$max_five_round_one" >"$WORK/max-five-round-one.json"
|
|
2307
|
+
check "a caller may configure four challenges after the initial Agent review" \
|
|
2308
|
+
'[ "$max_five_round_one_rc" = 0 ] && json_fields "$max_five_round_one" autonomous_review_budget=5 autonomous_review_index=1 autonomous_reviews_remaining=4 autonomous_review_allowed=true'
|
|
2309
|
+
|
|
2310
|
+
reset_case passed unavailable unavailable
|
|
2311
|
+
out="$(run_gate --challenge-budget 1 --review-chain-id per-budget-bound --autonomous-review-index 5)"; rc=$?
|
|
2312
|
+
check "the tracked index is bounded by this invocation's challenge budget" \
|
|
2313
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && [ "$(printf "%s" "$out" | python3 -c "import json,sys; print(json.load(sys.stdin).get(\"reason\"))")" = "a tracked Agent review requires --autonomous-review-index between 1 and 2" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
2314
|
+
|
|
2315
|
+
reset_case passed unavailable unavailable
|
|
2316
|
+
max_five_round_two="$(run_challenge_gate --challenge-budget 4 --challenge-index 1 --focus max-five-one --review-chain-id max-five-task --autonomous-review-index 2 --prior-review-result-file "$WORK/max-five-round-one.json")"; max_five_round_two_rc=$?
|
|
2317
|
+
printf '%s\n' "$max_five_round_two" >"$WORK/max-five-round-two.json"
|
|
2318
|
+
reset_case passed unavailable unavailable
|
|
2319
|
+
max_five_round_three="$(run_challenge_gate --challenge-budget 4 --challenge-index 2 --focus max-five-two --review-chain-id max-five-task --autonomous-review-index 3 --prior-review-result-file "$WORK/max-five-round-one.json" --prior-review-result-file "$WORK/max-five-round-two.json")"; max_five_round_three_rc=$?
|
|
2320
|
+
printf '%s\n' "$max_five_round_three" >"$WORK/max-five-round-three.json"
|
|
2321
|
+
reset_case passed unavailable unavailable
|
|
2322
|
+
max_five_round_four="$(run_challenge_gate --challenge-budget 4 --challenge-index 3 --focus max-five-three --review-chain-id max-five-task --autonomous-review-index 4 --prior-review-result-file "$WORK/max-five-round-one.json" --prior-review-result-file "$WORK/max-five-round-two.json" --prior-review-result-file "$WORK/max-five-round-three.json")"; max_five_round_four_rc=$?
|
|
2323
|
+
printf '%s\n' "$max_five_round_four" >"$WORK/max-five-round-four.json"
|
|
2324
|
+
reset_case passed unavailable unavailable
|
|
2325
|
+
max_five_round_five="$(run_challenge_gate --challenge-budget 4 --challenge-index 4 --focus max-five-four --review-chain-id max-five-task --autonomous-review-index 5 --prior-review-result-file "$WORK/max-five-round-one.json" --prior-review-result-file "$WORK/max-five-round-two.json" --prior-review-result-file "$WORK/max-five-round-three.json" --prior-review-result-file "$WORK/max-five-round-four.json")"; max_five_round_five_rc=$?
|
|
2326
|
+
printf '%s\n' "$max_five_round_five" >"$WORK/max-five-round-five.json"
|
|
2327
|
+
if [ "$max_five_round_two_rc" != 0 ] || [ "$max_five_round_three_rc" != 0 ] || [ "$max_five_round_four_rc" != 0 ] || [ "$max_five_round_five_rc" != 0 ]; then
|
|
2328
|
+
printf 'max-five diagnostic: rc=%s/%s/%s/%s round2=%s round3=%s round4=%s round5=%s\n' \
|
|
2329
|
+
"$max_five_round_two_rc" "$max_five_round_three_rc" "$max_five_round_four_rc" "$max_five_round_five_rc" \
|
|
2330
|
+
"$max_five_round_two" "$max_five_round_three" "$max_five_round_four" "$max_five_round_five" >&2
|
|
2331
|
+
fi
|
|
2332
|
+
check "one tracked Agent chain reaches five total external review rounds" \
|
|
2333
|
+
'[ "$max_five_round_two_rc" = 0 ] && [ "$max_five_round_three_rc" = 0 ] && [ "$max_five_round_four_rc" = 0 ] && [ "$max_five_round_five_rc" = 0 ] && json_fields "$max_five_round_five" autonomous_review_budget=5 autonomous_review_index=5 autonomous_reviews_remaining=0 autonomous_review_allowed=false prior_challenge_focuses.2=max-five-three next_action=deep_self_review_before_completion'
|
|
2334
|
+
|
|
2335
|
+
reset_case passed unavailable unavailable
|
|
2336
|
+
out="$(run_completion_gate --challenge-budget 4 --completion-review-result-file "$WORK/max-five-round-five.json")"; rc=$?
|
|
2337
|
+
check "completion accepts the fifth exact-candidate Agent round" \
|
|
2338
|
+
'[ "$rc" = 0 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" mode=complete status=passed autonomous_review_budget=5 autonomous_review_index=5 autonomous_reviews_remaining=0 completion_gated=false'
|
|
2339
|
+
|
|
2340
|
+
reset_case passed unavailable unavailable
|
|
2341
|
+
out="$(run_challenge_gate --challenge-budget 4 --challenge-index 5 --focus max-five-overflow --review-chain-id max-five-task --autonomous-review-index 6 --prior-review-result-file "$WORK/max-five-round-one.json" --prior-review-result-file "$WORK/max-five-round-two.json" --prior-review-result-file "$WORK/max-five-round-three.json" --prior-review-result-file "$WORK/max-five-round-four.json" --prior-review-result-file "$WORK/max-five-round-five.json")"; rc=$?
|
|
2342
|
+
check "a sixth Agent review is rejected before provider execution" \
|
|
2343
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=review_chain_invalid'
|
|
2344
|
+
|
|
2345
|
+
printf 'diff --git a/x b/x\n--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n' >"$WORK/diff.patch"
|
|
2346
|
+
|
|
2347
|
+
reset_case passed unavailable unavailable
|
|
2348
|
+
out="$(run_gate --challenge-budget 5)"; rc=$?
|
|
2349
|
+
check "Agent challenge budget above four fails closed before provider execution" \
|
|
2350
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=invalid_input'
|
|
2351
|
+
|
|
2352
|
+
reset_case passed unavailable unavailable
|
|
2353
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
2354
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
2355
|
+
--implementer-family openai --review-plan-file "$WORK/incomplete-plan.json")"; rc=$?
|
|
2356
|
+
check "incomplete owner-guided self-review fails before provider execution" \
|
|
2357
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete self_review_gate.required=true self_review_gate.required_triggers.0=before_external_review self_review_gate.blocks.0=external_review self_review_gate.blocks.1=completion_claim self_review_gate.allowed_next_actions.0=deep_self_review self_review_gate.allowed_next_actions.1=continue_implementation'
|
|
2358
|
+
|
|
2359
|
+
reset_case passed unavailable unavailable
|
|
2360
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
2361
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
2362
|
+
--implementer-family openai --review-plan-file "$WORK/placeholder-plan.json")"; rc=$?
|
|
2363
|
+
check "placeholder self-review conclusions fail before provider execution" \
|
|
2364
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete'
|
|
2365
|
+
|
|
2366
|
+
reset_case passed unavailable unavailable
|
|
2367
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
2368
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
2369
|
+
--implementer-family openai --review-plan-file "$WORK/placeholder-evidence-plan.json")"; rc=$?
|
|
2370
|
+
check "placeholder evidence results fail before provider execution" \
|
|
2371
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete'
|
|
2372
|
+
|
|
2373
|
+
for low_information_plan in low-information-self-review-plan low-information-evidence-plan; do
|
|
2374
|
+
reset_case passed unavailable unavailable
|
|
2375
|
+
out="$(REVIEW_GATE_TEST_STATE="$WORK/state" "$WORK/harness/scripts/review_gate.sh" \
|
|
2376
|
+
--mode review --cwd "$WORK/repo" --diff-file "$WORK/diff.patch" \
|
|
2377
|
+
--implementer-family openai --review-plan-file "$WORK/$low_information_plan.json")"; rc=$?
|
|
2378
|
+
check "$low_information_plan fails before provider execution" \
|
|
2379
|
+
'[ "$rc" = 2 ] && [ ! -e "$WORK/state/client_sequence" ] && json_fields "$out" reason_code=self_review_incomplete'
|
|
2380
|
+
done
|
|
2381
|
+
|
|
2382
|
+
reset_case quota spoof_controller unavailable
|
|
2383
|
+
out="$(run_gate --allow-fallback-egress)"; rc=$?
|
|
2384
|
+
check "reviewer cannot self-report controller-owned stage or coverage fields" \
|
|
2385
|
+
'[ "$rc" = 0 ] && json_fields "$out" stage=build review_depth=build owner_selection_source=implementer-declared selected_skills.0=code-review self_review_gate.required=true self_review_gate.required_triggers.0=before_completion_claim && [ "$(printf "%s" "$out" | python3 -c "import json,sys; print(len(json.load(sys.stdin).get(\"reviewed_skills\", [])))")" = 0 ] && ! grep -q spoofed <<<"$out"'
|
|
2386
|
+
|
|
2387
|
+
(
|
|
2388
|
+
cd "$WORK/repo"
|
|
2389
|
+
git init -q
|
|
2390
|
+
git config user.email test@example.invalid
|
|
2391
|
+
git config user.name 'Test User'
|
|
2392
|
+
printf 'before\n' >tracked.txt
|
|
2393
|
+
git add tracked.txt
|
|
2394
|
+
git commit -q -m initial
|
|
2395
|
+
printf 'after\n' >tracked.txt
|
|
2396
|
+
printf 'new\n' >untracked.txt
|
|
2397
|
+
)
|
|
2398
|
+
reset_case passed unavailable unavailable
|
|
2399
|
+
out="$(run_base_gate --allow-fallback-egress)"; rc=$?
|
|
2400
|
+
check "base packet freezes tracked and untracked changes once" \
|
|
2401
|
+
'[ "$rc" = 0 ] && grep -q tracked.txt "$WORK/state/claude_packet" && grep -q untracked.txt "$WORK/state/claude_packet" && grep -q "Untracked files" "$WORK/state/claude_packet"'
|
|
2402
|
+
|
|
2403
|
+
check "owner selection source has one controller-owned definition" \
|
|
2404
|
+
'[ "$(grep -c '\''"implementer-declared"'\'' "$DIR/review_gate.py")" = 1 ]'
|
|
2405
|
+
|
|
2406
|
+
printf '%s\n' '----'
|
|
2407
|
+
if [ "$fails" -eq 0 ]; then
|
|
2408
|
+
echo review_gate_tests_ok
|
|
2409
|
+
else
|
|
2410
|
+
echo "$fails FAILURES"
|
|
2411
|
+
exit 1
|
|
2412
|
+
fi
|