claude-dev-env 2.9.0 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +2 -2
- package/_shared/advisor/CLAUDE.md +3 -2
- package/_shared/advisor/advisor-protocol.md +74 -108
- package/_shared/advisor/reference/advisor-block.md +37 -0
- package/_shared/advisor/reference/cli-chain.md +45 -0
- package/_shared/advisor/reference/consult-format.md +41 -0
- package/_shared/advisor/reference/lifecycle.md +21 -0
- package/_shared/advisor/reference/sol-rung.md +31 -0
- package/_shared/advisor/reference/spawn-walk-log.md +31 -0
- package/_shared/advisor/reference/third-party-bind.md +30 -0
- package/_shared/advisor/reference/warm-up.md +33 -0
- package/_shared/advisor/scripts/codex_sol_advisor.py +449 -0
- package/_shared/advisor/scripts/config/advisor_scripts_constants/advisor_route_constants.py +21 -0
- package/_shared/advisor/scripts/config/advisor_scripts_constants/model_tier_run_validator_constants.py +19 -17
- package/_shared/advisor/scripts/config/advisor_scripts_constants/sol_advisor_constants.py +28 -0
- package/_shared/advisor/scripts/model_tier_run_validator.py +32 -9
- package/_shared/advisor/scripts/tests/test_codex_sol_advisor.py +474 -0
- package/_shared/advisor/scripts/tests/test_model_tier_run_validator.py +79 -0
- package/_shared/advisor/scripts/tests/test_tier_model_ids.py +39 -17
- package/_shared/advisor/scripts/tier_model_ids.py +24 -0
- package/_shared/pr-loop/CLAUDE.md +1 -1
- package/_shared/pr-loop/audit-contract.md +17 -6
- package/_shared/pr-loop/audit-reply-template.md +4 -4
- package/_shared/pr-loop/code-rules-gate.md +3 -5
- package/_shared/pr-loop/fix-protocol.md +2 -3
- package/_shared/pr-loop/gh-payloads.md +1 -1
- package/_shared/pr-loop/scripts/CLAUDE.md +1 -1
- package/_shared/pr-loop/scripts/README.md +1 -1
- package/_shared/pr-loop/scripts/code_rules_gate.py +2 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/gate_running.py +16 -1
- package/_shared/pr-loop/scripts/code_rules_gate_parts/git_blob_readers.py +11 -5
- package/_shared/pr-loop/scripts/preflight.py +9 -4
- package/_shared/pr-loop/scripts/reviews_disabled.py +50 -22
- package/_shared/pr-loop/scripts/tests/conftest.py +20 -0
- package/_shared/pr-loop/scripts/tests/test_claude_permissions_common.py +6 -6
- package/_shared/pr-loop/scripts/tests/test_reviews_disabled.py +50 -6
- package/_shared/pr-loop/scripts/tests/test_revoke_project_claude_permissions.py +1 -1
- package/_shared/pr-loop/state-schema.md +5 -14
- package/agents/CLAUDE.md +2 -2
- package/agents/clean-coder.md +58 -548
- package/agents/code-quality-agent.md +10 -2
- package/agents/code-verifier.md +1 -1
- package/agents/test_agent_frontmatter.py +32 -40
- package/audit-rubrics/CLAUDE.md +2 -1
- package/audit-rubrics/audit-categories.json +704 -0
- package/audit-rubrics/prompts/category-i-concurrency.md +1 -1
- package/bin/CLAUDE.md +16 -5
- package/bin/ever-shipped-skills.mjs +2 -0
- package/bin/install-plan.mjs +402 -0
- package/bin/install-transaction.mjs +455 -0
- package/bin/install.mjs +593 -147
- package/bin/install.plan.test.mjs +194 -0
- package/bin/install.profile-root.test.mjs +154 -0
- package/bin/install.profiles.test.mjs +253 -0
- package/bin/install.settings-defaults.test.mjs +200 -0
- package/bin/install.transaction.test.mjs +400 -0
- package/bin/install.uninstall-transaction.test.mjs +418 -0
- package/bin/merge_managed_permissions.mjs +130 -0
- package/bin/resolve-install-root.mjs +181 -0
- package/bin/select-install-targets.mjs +401 -0
- package/commands/CLAUDE.md +0 -2
- package/docs/references/CLAUDE.md +3 -1
- package/docs/references/advisor-tool.md +25 -7
- package/docs/references/prose-style-enforcement.md +25 -0
- package/docs/references/team-advisor-skill.md +3 -3
- package/docs/references/weak-executor-advisor.md +91 -0
- package/hooks/blocking/CLAUDE.md +6 -6
- package/hooks/blocking/_path_setup.py +9 -5
- package/hooks/blocking/code_rules_docstrings.py +124 -30
- package/hooks/blocking/code_rules_enforcer.py +161 -16
- package/hooks/blocking/code_rules_shared.py +40 -23
- package/hooks/blocking/config/CLAUDE.md +3 -5
- package/hooks/blocking/config/prose_style_enforcement_constants.py +38 -0
- package/hooks/blocking/config/test_prose_style_enforcement_constants.py +45 -0
- package/hooks/blocking/eli11_reply_enforcer.py +70 -113
- package/hooks/blocking/hedging_language_blocker.py +103 -20
- package/hooks/blocking/hook_prose_detector_consistency.py +6 -0
- package/hooks/blocking/intent_only_ending_blocker.py +6 -0
- package/hooks/blocking/plain_language_blocker.py +139 -20
- package/hooks/blocking/pre_tool_use_dispatcher.py +102 -20
- package/hooks/blocking/state_description_blocker.py +7 -1
- package/hooks/blocking/tdd_enforcer.py +8 -0
- package/hooks/blocking/test__path_setup.py +28 -0
- package/hooks/blocking/test_code_rules_enforcer_agent_home_tooling.py +99 -0
- package/hooks/blocking/test_code_rules_enforcer_docstring_args_span_scope.py +232 -10
- package/hooks/blocking/test_code_rules_enforcer_ephemeral.py +1 -1
- package/hooks/blocking/test_code_rules_enforcer_join_separator_magic.py +41 -0
- package/hooks/blocking/test_code_rules_enforcer_string_magic.py +98 -0
- package/hooks/blocking/test_eli11_reply_enforcer.py +98 -165
- package/hooks/blocking/test_fable_spawn_gate.py +18 -11
- package/hooks/blocking/test_hedging_language_blocker.py +120 -1
- package/hooks/blocking/test_hook_prose_detector_consistency.py +28 -8
- package/hooks/blocking/test_intent_only_ending_blocker.py +27 -2
- package/hooks/blocking/test_package_inventory_stale_blocker.py +11 -4
- package/hooks/blocking/test_plain_language_blocker.py +129 -19
- package/hooks/blocking/test_plain_language_blocker_allowlist.py +70 -26
- package/hooks/blocking/test_pre_tool_use_dispatcher.py +99 -26
- package/hooks/blocking/test_pre_tool_use_dispatcher_native.py +87 -50
- package/hooks/blocking/test_state_description_blocker.py +45 -2
- package/hooks/blocking/test_stop_dispatcher.py +11 -7
- package/hooks/blocking/test_volatile_path_in_post_blocker.py +12 -12
- package/hooks/blocking/volatile_path_in_post_blocker.py +2 -2
- package/hooks/hooks.json +15 -0
- package/hooks/hooks_constants/CLAUDE.md +14 -3
- package/hooks/hooks_constants/ask_user_question_shape.py +281 -0
- package/hooks/hooks_constants/code_rules_enforcer_constants.py +2 -1
- package/hooks/hooks_constants/eli11_reply_enforcer_constants.py +5 -12
- package/hooks/hooks_constants/hedging_uncertainty_constants.py +42 -0
- package/hooks/hooks_constants/issue_tracker_session_starter_constants.py +23 -0
- package/hooks/hooks_constants/orchestrator_auto_starter_constants.py +23 -0
- package/hooks/hooks_constants/piped_pytest_blocker_constants.py +4 -1
- package/hooks/hooks_constants/plain_language_blocker_constants.py +4 -1
- package/hooks/hooks_constants/pre_tool_use_dispatcher_constants.py +6 -0
- package/hooks/hooks_constants/project_paths_reader.py +31 -4
- package/hooks/hooks_constants/prose_matcher_precision_constants.py +40 -0
- package/hooks/hooks_constants/pytest_invocation.py +354 -0
- package/hooks/hooks_constants/session_start_injector.py +163 -0
- package/hooks/hooks_constants/session_start_injector_constants.py +46 -0
- package/hooks/hooks_constants/shell_command_pipeline.py +397 -0
- package/hooks/hooks_constants/shell_command_segments.py +5 -0
- package/hooks/hooks_constants/test_ask_user_question_shape.py +167 -0
- package/hooks/hooks_constants/test_project_paths_reader.py +29 -0
- package/hooks/hooks_constants/test_prose_metrics_parity.py +8 -0
- package/hooks/hooks_constants/test_pytest_invocation.py +130 -0
- package/hooks/hooks_constants/test_session_start_injector.py +168 -0
- package/hooks/hooks_constants/test_shell_command_pipeline.py +135 -0
- package/hooks/hooks_constants/volatile_path_in_post_blocker_constants.py +1 -1
- package/hooks/hooks_constants/working_style_prompt_constants.py +30 -0
- package/hooks/observability/CLAUDE.md +2 -0
- package/hooks/observability/prose_matcher_advisory.py +237 -0
- package/hooks/observability/test_prose_matcher_advisory.py +143 -0
- package/hooks/session/CLAUDE.md +9 -1
- package/hooks/session/_path_setup.py +13 -0
- package/hooks/session/issue_tracker_session_starter.py +135 -0
- package/hooks/session/orchestrator_auto_starter.py +100 -0
- package/hooks/session/test__path_setup.py +28 -0
- package/hooks/session/test_issue_tracker_session_starter.py +104 -0
- package/hooks/session/test_orchestrator_auto_starter.py +99 -0
- package/hooks/session/test_working_style_prompt.py +47 -0
- package/hooks/session/untracked_repo_detector.py +1 -24
- package/hooks/session/working_style_prompt.py +36 -0
- package/hooks/validators/_path_setup.py +19 -0
- package/hooks/validators/run_all_validators.py +8 -13
- package/installable-surfaces.manifest.json +21 -0
- package/output-styles/CLAUDE.md +1 -3
- package/package.json +4 -2
- package/rules/CLAUDE.md +1 -0
- package/rules/durable-post-artifacts.md +2 -2
- package/rules/eli11-replies.md +6 -1
- package/rules/hedging-claims.md +4 -2
- package/rules/long-horizon-autonomy.md +3 -1
- package/rules/opus5-communication-contract.md +45 -0
- package/rules/plain-language.md +2 -2
- package/rules/research-mode.md +1 -1
- package/scripts/CLAUDE.md +11 -0
- package/scripts/Sync-RepoMain.ps1 +215 -0
- package/scripts/active_capability_references.py +218 -0
- package/scripts/ci/windows-installer-lifecycle.ps1 +78 -0
- package/scripts/claude_chain_runner.py +394 -6
- package/scripts/claude_chain_usage.py +1 -1
- package/scripts/codex_compat_materializer.py +105 -85
- package/scripts/dev_env_scripts_constants/CLAUDE.md +2 -0
- package/scripts/dev_env_scripts_constants/active_capability_constants.py +46 -0
- package/scripts/dev_env_scripts_constants/claude_chain_constants.py +74 -0
- package/scripts/dev_env_scripts_constants/verify_installable_package_constants.py +116 -0
- package/scripts/profile-isolation-launchers/config/mcp-bundles.json +25 -0
- package/scripts/profile-isolation-launchers/config/profile-isolation-constants.mjs +60 -0
- package/scripts/profile-isolation-launchers/config/profiles.manifest.json +54 -0
- package/scripts/profile-isolation-launchers/config/shared-allowlist.json +64 -0
- package/scripts/profile-isolation-launchers/launcher-runtime.mjs +180 -0
- package/scripts/profile-isolation-launchers/lib/profile-manifest.mjs +288 -0
- package/scripts/profile-isolation-launchers/mcp-bundles.mjs +275 -0
- package/scripts/profile-isolation-launchers/profile-isolation-contract.test.mjs +221 -0
- package/scripts/profile-isolation-launchers/tests/launcher-runtime.test.mjs +108 -0
- package/scripts/profile-isolation-launchers/tests/mcp-bundles.test.mjs +147 -0
- package/scripts/profile-isolation-launchers/tests/shortcut-contract.test.ps1 +102 -0
- package/scripts/profile-isolation-launchers/tests/version-compatibility.test.mjs +210 -0
- package/scripts/profile-isolation-launchers/version-compatibility.mjs +299 -0
- package/scripts/profile-isolation-launchers/windows/shortcut-inventory.ps1 +127 -0
- package/scripts/profile-isolation-launchers/windows/shortcut-manifest.json +51 -0
- package/scripts/profile-isolation-launchers/windows/shortcut-reconcile.ps1 +77 -0
- package/scripts/spawn_grok_batch.py +3 -0
- package/scripts/test_active_capability_references.py +108 -0
- package/scripts/test_claude_chain_runner.py +414 -82
- package/scripts/test_claude_chain_usage.py +12 -12
- package/scripts/test_resolve_worker_spawn.py +2 -2
- package/scripts/test_verify_installable_package.py +208 -0
- package/scripts/tests/test_codex_compat_materializer.py +33 -0
- package/scripts/verify_installable_package.py +612 -0
- package/settings.json +10 -0
- package/skills/CLAUDE.md +2 -0
- package/skills/_shared/advisor/CLAUDE.md +1 -1
- package/skills/_shared/advisor/scripts/README.md +2 -0
- package/skills/_shared/pr-loop/scripts/CLAUDE.md +1 -0
- package/skills/_shared/pr-loop/scripts/audit_category_schema.py +355 -0
- package/skills/_shared/pr-loop/scripts/skills_pr_loop_constants/CLAUDE.md +1 -0
- package/skills/_shared/pr-loop/scripts/skills_pr_loop_constants/audit_category_schema_constants.py +32 -0
- package/skills/_shared/pr-loop/scripts/skills_pr_loop_constants/path_resolver_constants.py +7 -19
- package/skills/_shared/pr-loop/scripts/test_audit_category_schema.py +94 -0
- package/skills/_shared/pr-loop/scripts/test_build_audit_prompt.py +21 -0
- package/skills/autoconverge/reference/convergence.md +2 -1
- package/skills/autoconverge/reference/stop-conditions.md +5 -3
- package/skills/beat-sheet/SKILL.md +54 -0
- package/skills/beat-sheet/reference/visual-beats.md +29 -0
- package/skills/bugteam/CONSTRAINTS.md +4 -4
- package/skills/bugteam/EXAMPLES.md +1 -1
- package/skills/bugteam/reference/README.md +1 -1
- package/skills/e-code-review/SKILL.md +26 -5
- package/skills/e-code-review/reference/effort-evaluation.md +35 -0
- package/skills/e-code-review/reference/medium.md +15 -4
- package/skills/e-code-review/scripts/config/e_code_review_effort_constants/__init__.py +41 -0
- package/skills/e-code-review/scripts/config/e_code_review_effort_constants/effort_constants.py +40 -0
- package/skills/e-code-review/scripts/e_code_review_scripts_constants/finding_pipeline_constants.py +49 -0
- package/skills/e-code-review/scripts/effort_defaults_evidence.json +186 -0
- package/skills/e-code-review/scripts/effort_evaluation.py +362 -0
- package/skills/e-code-review/scripts/finding_pipeline.py +140 -0
- package/skills/e-code-review/scripts/fixtures/demanding.json +26 -0
- package/skills/e-code-review/scripts/fixtures/easy.json +14 -0
- package/skills/e-code-review/scripts/fixtures/medium.json +20 -0
- package/skills/e-code-review/scripts/grok_code_review.py +16 -7
- package/skills/e-code-review/scripts/test_effort_evaluation.py +180 -0
- package/skills/e-code-review/scripts/test_finding_pipeline.py +197 -0
- package/skills/e-code-review/scripts/test_grok_code_review.py +77 -0
- package/skills/grokify/SKILL.md +1 -1
- package/skills/grokify/templates/handoff-template.md +2 -2
- package/skills/orchestrator/SKILL.md +5 -4
- package/skills/plan-to-pr/scripts/create_packet.py +4 -4
- package/skills/plan-to-pr/scripts/load_skill_constants.py +41 -0
- package/skills/plan-to-pr/scripts/validate_packet.py +4 -4
- package/skills/plan-to-pr/scripts/validate_protocol.py +4 -1
- package/skills/plan-to-pr/scripts/validate_run.py +4 -1
- package/skills/pr-converge/scripts/check_convergence.py +21 -19
- package/skills/pr-converge/scripts/check_convergence_availability.py +50 -7
- package/skills/pr-converge/scripts/conftest.py +35 -0
- package/skills/pr-converge/scripts/test_check_convergence_availability.py +65 -0
- package/skills/pr-converge/scripts/test_check_convergence_codex.py +11 -1
- package/skills/pr-converge/scripts/test_check_convergence_contract.py +9 -2
- package/skills/pr-loop-cloud-transport/SKILL.md +1 -1
- package/skills/rebase/SKILL.md +15 -3
- package/skills/reviewer-gates/SKILL.md +2 -2
- package/skills/show/SKILL.md +51 -0
- package/skills/show/references/accessibility.md +7 -0
- package/skills/show/references/art.md +3 -0
- package/skills/show/references/charts.md +3 -0
- package/skills/show/references/core-design.md +14 -0
- package/skills/show/references/erds.md +3 -0
- package/skills/show/references/flowcharts.md +3 -0
- package/skills/show/references/host-and-html.md +3 -0
- package/skills/show/references/illustrative-diagrams.md +10 -0
- package/skills/show/references/interaction.md +3 -0
- package/skills/show/references/mockups.md +3 -0
- package/skills/show/references/quality-gates.md +7 -0
- package/skills/show/references/structural-diagrams.md +3 -0
- package/skills/show/references/subject-inventory.md +21 -0
- package/skills/show/references/svg-contract.md +22 -0
- package/skills/show/routing.yaml +30 -0
- package/skills/show/samples/pr1262-v2.svg +222 -0
- package/skills/show/scripts/README.md +6 -0
- package/skills/show/scripts/validate-artifact.py +91 -0
- package/skills/show/scripts/validate-package.py +18 -0
- package/skills/show/templates/html-widget.html +4 -0
- package/skills/show/templates/svg-base.svg +19 -0
- package/skills/show/tests/fixtures/css-var.svg +6 -0
- package/skills/show/tests/fixtures/dead-ref.svg +7 -0
- package/skills/show/tests/fixtures/filled-glyph.svg +8 -0
- package/skills/show/tests/fixtures/inherited-fill.svg +18 -0
- package/skills/show/tests/fixtures/invalid.svg +1 -0
- package/skills/show/tests/fixtures/large-canvas.svg +21 -0
- package/skills/show/tests/fixtures/unfilled-connector.svg +15 -0
- package/skills/show/tests/fixtures/valid.html +1 -0
- package/skills/show/tests/test_validate-artifact.py +74 -0
- package/skills/show/tests/test_validators.py +59 -0
- package/skills/show/workflows/create-visual.md +13 -0
- package/skills/show/workflows/review-visual.md +20 -0
- package/skills/split-pr/SKILL.md +85 -0
- package/skills/split-pr/reference/path-layers.md +16 -0
- package/skills/split-pr/reference/proposal-format.md +15 -0
- package/skills/split-pr/reference/split-further-loop.md +10 -0
- package/skills/split-pr/reference/splitting-principles.md +26 -0
- package/skills/split-pr/scripts/analyze_pr.py +279 -0
- package/skills/split-pr/scripts/categorize_files.py +106 -0
- package/skills/split-pr/scripts/config/__init__.py +1 -0
- package/skills/split-pr/scripts/config/dependency_constants.py +14 -0
- package/skills/split-pr/scripts/config/git_operations_constants.py +36 -0
- package/skills/split-pr/scripts/config/packing_constants.py +61 -0
- package/skills/split-pr/scripts/config/plan_constants.py +49 -0
- package/skills/split-pr/scripts/config/split_pr_constants.py +110 -0
- package/skills/split-pr/scripts/execute_split_slices.py +82 -0
- package/skills/split-pr/scripts/pack_files_into_slices.py +212 -0
- package/skills/split-pr/scripts/split_pr_dependency_graph.py +70 -0
- package/skills/split-pr/scripts/split_pr_git_operations.py +184 -0
- package/skills/split-pr/scripts/split_pr_layer_order.py +58 -0
- package/skills/split-pr/scripts/split_pr_paginate.py +119 -0
- package/skills/split-pr/scripts/split_pr_process_runner.py +52 -0
- package/skills/split-pr/scripts/split_pr_script_types.py +126 -0
- package/skills/split-pr/scripts/split_pr_title.py +41 -0
- package/skills/split-pr/scripts/test_analyze_pr.py +228 -0
- package/skills/split-pr/scripts/test_categorize_files.py +55 -0
- package/skills/split-pr/scripts/test_categorize_files_packing.py +59 -0
- package/skills/split-pr/scripts/test_execute_split_slices.py +99 -0
- package/skills/split-pr/scripts/test_split_pr_dependency_graph.py +47 -0
- package/skills/split-pr/scripts/test_split_pr_git_operations.py +125 -0
- package/skills/split-pr/scripts/test_split_pr_layer_order.py +36 -0
- package/skills/split-pr/scripts/test_split_pr_paginate.py +65 -0
- package/skills/split-pr/scripts/test_split_pr_script_types.py +73 -0
- package/skills/split-pr/scripts/test_split_pr_title.py +28 -0
- package/skills/split-pr/scripts/test_verify_dependency_graph.py +46 -0
- package/skills/split-pr/scripts/test_verify_plan.py +56 -0
- package/skills/split-pr/scripts/test_verify_plan_contract.py +50 -0
- package/skills/split-pr/scripts/test_verify_plan_path_normalization.py +45 -0
- package/skills/split-pr/scripts/verify_dependency_graph.py +111 -0
- package/skills/split-pr/scripts/verify_plan.py +139 -0
- package/skills/team-advisor/SKILL.md +7 -4
- package/skills/team-advisor/reference/advisor-docs-review.md +207 -0
- package/system-prompts/software-engineer.xml +11 -2
- package/commands/initialize.md +0 -90
- package/commands/stubcheck.md +0 -88
- package/output-styles/caveman-agent.md +0 -37
|
@@ -19,10 +19,10 @@
|
|
|
19
19
|
|
|
20
20
|
The three sibling skills compose, but `/bugteam` solves a problem they cannot solve in sequence:
|
|
21
21
|
|
|
22
|
-
-
|
|
23
|
-
- `/
|
|
24
|
-
- A human-driven
|
|
22
|
+
- `clean-room audit (code-quality-agent)` audits once and stops.
|
|
23
|
+
- `/pr-fix-protocol` fixes the findings of one audit and stops.
|
|
24
|
+
- A human-driven `clean-room audit (code-quality-agent)` → `/pr-fix-protocol` → `clean-room audit (code-quality-agent)` → `/pr-fix-protocol` cycle works but requires the user to drive it.
|
|
25
25
|
|
|
26
26
|
`/bugteam` automates that cycle. The clean-room property is preserved by spawning a fresh audit agent each loop with no inherited context — every audit is independent of the prior loop's verdict. The 20-loop cap is the safety: pathological cases (audit agent oscillating, fix agent regressing) cannot run away.
|
|
27
27
|
|
|
28
|
-
The single up-front confirmation is the explicit trade — `/bugteam` is more autonomous than
|
|
28
|
+
The single up-front confirmation is the explicit trade — `/bugteam` is more autonomous than `clean-room audit (code-quality-agent)`+`/pr-fix-protocol` chained manually. The user accepts that autonomy by typing the command. Stop conditions and the loop log give the user full visibility on exit.
|
|
@@ -20,7 +20,7 @@ The `pre-push-review` skill was retired. Its mechanical checks are now covered a
|
|
|
20
20
|
|
|
21
21
|
- **Mechanical pre-push checks** (magic values, boolean naming, imports, constants location, and other CODE_RULES checks) — handled by the `code_rules_enforcer.py` PreToolUse hook (blocks at write time) and by the git pre-push hook installed via `npx claude-dev-env`. The git pre-push hook is the gate that runs at `git push` time; no manual invocation is needed.
|
|
22
22
|
|
|
23
|
-
- **`/
|
|
23
|
+
- **`/bugteam`** — a full PR audit-fix cycle that spawns subagents, runs multiple audit loops, and produces a structured report. It is NOT a lightweight pre-push gate. Do not use `/bugteam` as a substitute for `git push` (the hook fires automatically). Use `/bugteam` when you want a thorough multi-loop review of a PR before requesting human review.
|
|
24
24
|
|
|
25
25
|
References:
|
|
26
26
|
- `hooks/git-hooks/pre_push.py` — the git pre-push hook that runs the CODE_RULES gate over the commits about to be pushed
|
|
@@ -14,8 +14,9 @@ description: >-
|
|
|
14
14
|
## Gotchas
|
|
15
15
|
|
|
16
16
|
- **`low` stays single-pass.** No subagents, no full-file reads: one read pass per target item, one findings pass.
|
|
17
|
-
-
|
|
18
|
-
-
|
|
17
|
+
- **Collection reports every real finding.** Keep every CONFIRMED or PLAUSIBLE finding at its assigned severity (`blocker`, `high`, `medium`, `low`, `nit`). Do not drop low or nit findings during collection. Severity or action filtering is a separate consumer stage after the collection record is complete (`scripts/finding_pipeline.py`).
|
|
18
|
+
- **`medium` favors precision, `xhigh` favors recall.** At `medium` (8 angles) assign severity carefully so a later filter can pick maintainer-action findings. At `xhigh` (10 angles plus a gap sweep) a single non-REFUTED vote carries the finding; do not drop on uncertainty.
|
|
19
|
+
- **Every retained finding carries `severity` and `verdict`.** Severity is one of `blocker`, `high`, `medium`, `low`, `nit`. Verdict is `CONFIRMED` or `PLAUSIBLE`. Drop REFUTED candidates only; never emit an unclassified retained finding.
|
|
19
20
|
- **`--fix` applies findings once.** Load `reference/fix.md` and follow it — it owns the fix agent, the code-rules gate, skip logging, and outcome reporting. Commits are lead-owned; fix agents never commit or push.
|
|
20
21
|
- **`loop` never asks.** A round with bug findings validates them with an advisor, fixes, and re-reviews. Terminals are exactly `clean`, `nits_fixed`, and `advisor_blocked`. There is no reviewed-head count limit; a new head increments the count once, a re-review of the same head does not. Load `reference/loop.md` and follow it.
|
|
21
22
|
- **`--fix` and `loop` combine.** With both, each loop round runs the level file, and the round's fixing happens inside `reference/loop.md`'s gate sequence, which loads `reference/fix.md` for the mechanics. There is no separate fix pass around the round.
|
|
@@ -28,6 +29,20 @@ Triggers: `/e-code-review <level> [--fix] [loop]`. `<level>` is `low`, `medium`,
|
|
|
28
29
|
|
|
29
30
|
- **No level, or an unknown level.** Respond exactly: `Which effort level — low, medium, or xhigh?`
|
|
30
31
|
|
|
32
|
+
## Evaluation-backed defaults (e-code-review family)
|
|
33
|
+
|
|
34
|
+
When the caller asks which level to pick for a known fixture band (easy / medium / demanding scope), use the committed OP-02B evaluation evidence — not an inherited hard-coded preference. Thinking stays on; **effort** is the only cost/latency lever.
|
|
35
|
+
|
|
36
|
+
| Fixture band | Skill effort | Source |
|
|
37
|
+
|---|---|---|
|
|
38
|
+
| easy | `medium` | `scripts/effort_defaults_evidence.json` → `skill_defaults.default_by_band.easy` |
|
|
39
|
+
| medium | `xhigh` | same file → `medium` (maps evaluation `high`) |
|
|
40
|
+
| demanding | `xhigh` | same file → `demanding` (maps evaluation `max`) |
|
|
41
|
+
|
|
42
|
+
Resolve programmatically with `scripts/effort_evaluation.py` → `resolve_skill_effort_for_band`. Every default cites a completed evaluation row. The user-facing command still requires an explicit level flag; these defaults answer "what should I pick?" for this workflow family only.
|
|
43
|
+
|
|
44
|
+
Detail: `reference/effort-evaluation.md`.
|
|
45
|
+
|
|
31
46
|
## The process
|
|
32
47
|
|
|
33
48
|
1. Read `<level>` and the optional `--fix` and `loop` flags. Apply the refusal first.
|
|
@@ -46,13 +61,19 @@ Triggers: `/e-code-review <level> [--fix] [loop]`. `<level>` is `low`, `medium`,
|
|
|
46
61
|
| `reference/xhigh.md` | xhigh review procedure — 10 angles, 1-vote verify, gap sweep |
|
|
47
62
|
| `reference/fix.md` | Fix application, code-rules gate, skip logging, outcome reporting |
|
|
48
63
|
| `reference/loop.md` | Repeat review/fix rounds until clean |
|
|
64
|
+
| `reference/effort-evaluation.md` | Effort evaluation fixtures, evidence, and skill defaults |
|
|
65
|
+
| `reference/runner-selection.md` | Runner selection map |
|
|
66
|
+
| `scripts/finding_pipeline.py` | Collect every real finding; filter severity only later |
|
|
67
|
+
| `scripts/test_finding_pipeline.py` | Collection and filter-stage behavioral tests |
|
|
68
|
+
| `scripts/e_code_review_scripts_constants/finding_pipeline_constants.py` | Named constants for the collect-then-filter pipeline |
|
|
69
|
+
| `scripts/effort_evaluation.py` | Effort rows, recommendation, skill default resolver |
|
|
70
|
+
| `scripts/effort_defaults_evidence.json` | Committed evaluation rows + e-code-review skill defaults |
|
|
49
71
|
| `scripts/grok_code_review.py` | Grok medium-review discovery and verification |
|
|
50
72
|
| `scripts/test_grok_code_review.py` | Behavioral tests for the Grok medium-review module |
|
|
51
73
|
| `scripts/e_code_review_scripts_constants/` | Skill-local constants (unique package name; avoids bare `config` import shadow) |
|
|
52
|
-
| `reference/runner-selection.md` | Runner selection map |
|
|
53
74
|
|
|
54
75
|
## Folder map
|
|
55
76
|
|
|
56
77
|
- `SKILL.md` — route and dispatch.
|
|
57
|
-
- `reference/` — level procedures, fix/loop, runner selection.
|
|
58
|
-
- `scripts/` — medium-review module, tests, `e_code_review_scripts_constants/`.
|
|
78
|
+
- `reference/` — level procedures, fix/loop, effort evaluation, runner selection.
|
|
79
|
+
- `scripts/` — collect-then-filter finding pipeline, effort evaluation, Grok medium-review module, tests, and `e_code_review_scripts_constants/`.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# Effort evaluation sweep (Opus)
|
|
2
|
+
|
|
3
|
+
Offline-first harness for choosing e-code-review effort and publishing cited skill defaults.
|
|
4
|
+
|
|
5
|
+
## What it holds
|
|
6
|
+
|
|
7
|
+
| Piece | Path |
|
|
8
|
+
|---|---|
|
|
9
|
+
| Fixtures | `scripts/fixtures/{easy,medium,demanding}.json` |
|
|
10
|
+
| Schema + recommend | `scripts/effort_evaluation.py` |
|
|
11
|
+
| Constants | `scripts/config/e_code_review_effort_constants/` |
|
|
12
|
+
| Tests | `scripts/test_effort_evaluation.py` |
|
|
13
|
+
|
|
14
|
+
## How to run offline tests
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
python -m pytest packages/claude-dev-env/skills/e-code-review/scripts/test_effort_evaluation.py -q
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Skill defaults (OP-02C — e-code-review family)
|
|
21
|
+
|
|
22
|
+
Committed evidence: `scripts/effort_defaults_evidence.json`.
|
|
23
|
+
|
|
24
|
+
- `resolve_skill_effort_for_band("easy"|"medium"|"demanding")` → `low`|`medium`|`xhigh`
|
|
25
|
+
- Evaluation `high` / `max` map to skill `xhigh` (skill surface has three levels only)
|
|
26
|
+
- Every skill default cites a completed evaluation row with `thinking_enabled: true`
|
|
27
|
+
|
|
28
|
+
## Paid / live runs (optional)
|
|
29
|
+
|
|
30
|
+
1. For each fixture, run the matching e-code-review level at each CLI-supported effort (`low` … `max`) with thinking on.
|
|
31
|
+
2. Score quality, finding recall, finding precision; record visible tokens and latency.
|
|
32
|
+
3. Feed completed rows into `recommend_effort_by_band` and `skill_defaults_from_recommendation`.
|
|
33
|
+
4. Replace `effort_defaults_evidence.json` when live rows supersede the offline baseline.
|
|
34
|
+
|
|
35
|
+
Stop if the Opus 5 CLI model is unavailable, or if results cannot separate quality from cost/latency.
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
`medium effort → 3+5 angles → 1-vote verify`
|
|
2
2
|
|
|
3
|
-
You are reviewing for **precision** at medium effort:
|
|
4
|
-
|
|
3
|
+
You are reviewing for **precision** at medium effort: assign severity carefully
|
|
4
|
+
so a later consumer can select maintainer-action findings. **Collection
|
|
5
|
+
reports every real finding** (including `low` and `nit`); do not drop by
|
|
6
|
+
severity during collection. Severity filtering is a separate consumer stage
|
|
7
|
+
after the collection record is complete.
|
|
5
8
|
|
|
6
9
|
## Phase 0 — Gather the diff
|
|
7
10
|
|
|
@@ -121,11 +124,19 @@ Keep candidates where the vote is CONFIRMED or PLAUSIBLE.
|
|
|
121
124
|
|
|
122
125
|
## Output
|
|
123
126
|
|
|
127
|
+
**Collection stage first.** Keep every CONFIRMED or PLAUSIBLE finding in the
|
|
128
|
+
collection record, including `low` and `nit`. Drop only REFUTED candidates.
|
|
129
|
+
Do not drop a real finding because its severity is low.
|
|
130
|
+
|
|
124
131
|
Report this review's results — `{level, findings}` — through the structured
|
|
125
132
|
findings-report call: the mechanism that renders a review's results as a typed
|
|
126
133
|
list in the host UI, ranked most-severe first. Each **retained** entry carries
|
|
127
|
-
every field below.
|
|
128
|
-
|
|
134
|
+
every field below. Do not emit a finding that lacks `severity` or `verdict`.
|
|
135
|
+
|
|
136
|
+
**Severity filter stage (later consumer).** After the collection record is
|
|
137
|
+
complete, a separate consumer may filter by minimum severity for action or
|
|
138
|
+
display. That filter must not rewrite or discard fields on the collection
|
|
139
|
+
record itself.
|
|
129
140
|
|
|
130
141
|
| Field | Required | Value |
|
|
131
142
|
|---|---|---|
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Constants for the e-code-review effort evaluation harness."""
|
|
2
|
+
|
|
3
|
+
from e_code_review_effort_constants.effort_constants import (
|
|
4
|
+
ALL_EFFORT_LEVELS,
|
|
5
|
+
ALL_EFFORT_RANK_BY_NAME,
|
|
6
|
+
ALL_FIXTURE_BANDS,
|
|
7
|
+
ALL_REQUIRED_ROW_KEYS,
|
|
8
|
+
ALL_SCORE_ROW_KEYS,
|
|
9
|
+
ALL_SKILL_EFFORT_FOR_EVALUATION_EFFORT,
|
|
10
|
+
ALL_SKILL_EFFORT_LEVELS,
|
|
11
|
+
COST_LATENCY_LEVER,
|
|
12
|
+
EVALUATION_EVIDENCE_FILENAME,
|
|
13
|
+
EVALUATION_SCHEMA_VERSION,
|
|
14
|
+
FIXTURES_DIRECTORY_NAME,
|
|
15
|
+
JSON_SUFFIX,
|
|
16
|
+
LATENCY_MS_ROW_KEY,
|
|
17
|
+
MINIMUM_QUALITY_HOLD_SCORE,
|
|
18
|
+
THINKING_ENABLED_DEFAULT,
|
|
19
|
+
VISIBLE_TOKENS_ROW_KEY,
|
|
20
|
+
WORKFLOW_FAMILY_E_CODE_REVIEW,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"ALL_EFFORT_LEVELS",
|
|
25
|
+
"ALL_EFFORT_RANK_BY_NAME",
|
|
26
|
+
"ALL_FIXTURE_BANDS",
|
|
27
|
+
"ALL_REQUIRED_ROW_KEYS",
|
|
28
|
+
"ALL_SCORE_ROW_KEYS",
|
|
29
|
+
"ALL_SKILL_EFFORT_FOR_EVALUATION_EFFORT",
|
|
30
|
+
"ALL_SKILL_EFFORT_LEVELS",
|
|
31
|
+
"COST_LATENCY_LEVER",
|
|
32
|
+
"EVALUATION_EVIDENCE_FILENAME",
|
|
33
|
+
"EVALUATION_SCHEMA_VERSION",
|
|
34
|
+
"FIXTURES_DIRECTORY_NAME",
|
|
35
|
+
"JSON_SUFFIX",
|
|
36
|
+
"LATENCY_MS_ROW_KEY",
|
|
37
|
+
"MINIMUM_QUALITY_HOLD_SCORE",
|
|
38
|
+
"THINKING_ENABLED_DEFAULT",
|
|
39
|
+
"VISIBLE_TOKENS_ROW_KEY",
|
|
40
|
+
"WORKFLOW_FAMILY_E_CODE_REVIEW",
|
|
41
|
+
]
|
package/skills/e-code-review/scripts/config/e_code_review_effort_constants/effort_constants.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Named values for the Opus effort evaluation sweep."""
|
|
2
|
+
|
|
3
|
+
EVALUATION_SCHEMA_VERSION: str = "1"
|
|
4
|
+
THINKING_ENABLED_DEFAULT: bool = True
|
|
5
|
+
MINIMUM_QUALITY_HOLD_SCORE: float = 0.8
|
|
6
|
+
COST_LATENCY_LEVER: str = "effort"
|
|
7
|
+
ALL_EFFORT_LEVELS: tuple[str, ...] = ("low", "medium", "high", "xhigh", "max")
|
|
8
|
+
ALL_FIXTURE_BANDS: tuple[str, ...] = ("easy", "medium", "demanding")
|
|
9
|
+
ALL_SCORE_ROW_KEYS: tuple[str, ...] = (
|
|
10
|
+
"quality_score",
|
|
11
|
+
"finding_recall",
|
|
12
|
+
"finding_precision",
|
|
13
|
+
)
|
|
14
|
+
FIXTURES_DIRECTORY_NAME: str = "fixtures"
|
|
15
|
+
JSON_SUFFIX: str = ".json"
|
|
16
|
+
VISIBLE_TOKENS_ROW_KEY: str = "visible_tokens"
|
|
17
|
+
LATENCY_MS_ROW_KEY: str = "latency_ms"
|
|
18
|
+
ALL_REQUIRED_ROW_KEYS: tuple[str, ...] = (
|
|
19
|
+
"fixture_id",
|
|
20
|
+
"fixture_band",
|
|
21
|
+
"effort",
|
|
22
|
+
*ALL_SCORE_ROW_KEYS,
|
|
23
|
+
VISIBLE_TOKENS_ROW_KEY,
|
|
24
|
+
LATENCY_MS_ROW_KEY,
|
|
25
|
+
"thinking_enabled",
|
|
26
|
+
)
|
|
27
|
+
ALL_EFFORT_RANK_BY_NAME: dict[str, int] = {
|
|
28
|
+
each_effort: each_index
|
|
29
|
+
for each_index, each_effort in enumerate(ALL_EFFORT_LEVELS)
|
|
30
|
+
}
|
|
31
|
+
ALL_SKILL_EFFORT_LEVELS: tuple[str, ...] = ("low", "medium", "xhigh")
|
|
32
|
+
EVALUATION_EVIDENCE_FILENAME: str = "effort_defaults_evidence.json"
|
|
33
|
+
WORKFLOW_FAMILY_E_CODE_REVIEW: str = "e-code-review"
|
|
34
|
+
ALL_SKILL_EFFORT_FOR_EVALUATION_EFFORT: dict[str, str] = {
|
|
35
|
+
"low": "low",
|
|
36
|
+
"medium": "medium",
|
|
37
|
+
"high": "xhigh",
|
|
38
|
+
"xhigh": "xhigh",
|
|
39
|
+
"max": "xhigh",
|
|
40
|
+
}
|
package/skills/e-code-review/scripts/e_code_review_scripts_constants/finding_pipeline_constants.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Named constants for the collect-then-filter finding pipeline.
|
|
2
|
+
|
|
3
|
+
::
|
|
4
|
+
|
|
5
|
+
collect_findings([...low-severity finding...])
|
|
6
|
+
ok: collection keeps every seeded severity including low and nit
|
|
7
|
+
filter_findings_by_severity(collection, minimum_severity="medium")
|
|
8
|
+
ok: consumer stage drops lower severities; collection record is unchanged
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
COLLECTION_STAGE_NAME: str = "collection"
|
|
14
|
+
"""Stage that retains every real finding with its evidence fields."""
|
|
15
|
+
|
|
16
|
+
FILTER_STAGE_NAME: str = "severity_filter"
|
|
17
|
+
"""Later consumer stage that may drop findings by severity for action."""
|
|
18
|
+
|
|
19
|
+
FINDING_FIELD_FILE: str = "file"
|
|
20
|
+
FINDING_FIELD_LINE: str = "line"
|
|
21
|
+
FINDING_FIELD_SEVERITY: str = "severity"
|
|
22
|
+
FINDING_FIELD_CATEGORY: str = "category"
|
|
23
|
+
FINDING_FIELD_EVIDENCE: str = "evidence"
|
|
24
|
+
|
|
25
|
+
SEVERITY_BLOCKER: str = "blocker"
|
|
26
|
+
SEVERITY_HIGH: str = "high"
|
|
27
|
+
SEVERITY_MEDIUM: str = "medium"
|
|
28
|
+
SEVERITY_LOW: str = "low"
|
|
29
|
+
SEVERITY_NIT: str = "nit"
|
|
30
|
+
|
|
31
|
+
ALL_SEVERITY_RANK_BY_TOKEN: dict[str, int] = {
|
|
32
|
+
SEVERITY_BLOCKER: 5,
|
|
33
|
+
SEVERITY_HIGH: 4,
|
|
34
|
+
SEVERITY_MEDIUM: 3,
|
|
35
|
+
SEVERITY_LOW: 2,
|
|
36
|
+
SEVERITY_NIT: 1,
|
|
37
|
+
}
|
|
38
|
+
"""Higher rank is more severe; used only by the filter stage."""
|
|
39
|
+
|
|
40
|
+
ALL_COLLECTION_SEVERITIES: tuple[str, ...] = tuple(ALL_SEVERITY_RANK_BY_TOKEN)
|
|
41
|
+
"""Severity tokens collection accepts; derived from the rank map keys."""
|
|
42
|
+
|
|
43
|
+
REPORT_EVERY_FINDING_INSTRUCTION: str = (
|
|
44
|
+
"Report every real finding. Collection retains all severities "
|
|
45
|
+
f"({', '.join(ALL_COLLECTION_SEVERITIES)}). Do not drop findings "
|
|
46
|
+
"by severity during collection; severity or action filtering is a "
|
|
47
|
+
"separate consumer stage after the collection record is complete."
|
|
48
|
+
)
|
|
49
|
+
"""Reviewer-prompt instruction that separates collection from filtering."""
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schema_version": "1",
|
|
3
|
+
"workflow_family": "e-code-review",
|
|
4
|
+
"evidence_source": "offline_harness_baseline",
|
|
5
|
+
"thinking_enabled": true,
|
|
6
|
+
"cost_latency_lever": "effort",
|
|
7
|
+
"notes": "Completed evaluation rows from the OP-02B offline harness. Each skill default cites a holding row. Paid live sweeps may replace rows without changing the resolver contract.",
|
|
8
|
+
"all_rows": [
|
|
9
|
+
{
|
|
10
|
+
"fixture_id": "easy-comment-preservation",
|
|
11
|
+
"fixture_band": "easy",
|
|
12
|
+
"effort": "low",
|
|
13
|
+
"quality_score": 0.5,
|
|
14
|
+
"finding_recall": 0.5,
|
|
15
|
+
"finding_precision": 0.5,
|
|
16
|
+
"visible_tokens": 800,
|
|
17
|
+
"latency_ms": 2000,
|
|
18
|
+
"thinking_enabled": true
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"fixture_id": "easy-comment-preservation",
|
|
22
|
+
"fixture_band": "easy",
|
|
23
|
+
"effort": "medium",
|
|
24
|
+
"quality_score": 0.95,
|
|
25
|
+
"finding_recall": 0.95,
|
|
26
|
+
"finding_precision": 0.95,
|
|
27
|
+
"visible_tokens": 1600,
|
|
28
|
+
"latency_ms": 4000,
|
|
29
|
+
"thinking_enabled": true
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
"fixture_id": "medium-constants-and-types",
|
|
33
|
+
"fixture_band": "medium",
|
|
34
|
+
"effort": "medium",
|
|
35
|
+
"quality_score": 0.7,
|
|
36
|
+
"finding_recall": 0.65,
|
|
37
|
+
"finding_precision": 0.75,
|
|
38
|
+
"visible_tokens": 2200,
|
|
39
|
+
"latency_ms": 5000,
|
|
40
|
+
"thinking_enabled": true
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"fixture_id": "medium-constants-and-types",
|
|
44
|
+
"fixture_band": "medium",
|
|
45
|
+
"effort": "high",
|
|
46
|
+
"quality_score": 0.9,
|
|
47
|
+
"finding_recall": 0.9,
|
|
48
|
+
"finding_precision": 0.9,
|
|
49
|
+
"visible_tokens": 3200,
|
|
50
|
+
"latency_ms": 7000,
|
|
51
|
+
"thinking_enabled": true
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"fixture_id": "demanding-multi-surface",
|
|
55
|
+
"fixture_band": "demanding",
|
|
56
|
+
"effort": "xhigh",
|
|
57
|
+
"quality_score": 0.75,
|
|
58
|
+
"finding_recall": 0.7,
|
|
59
|
+
"finding_precision": 0.8,
|
|
60
|
+
"visible_tokens": 5000,
|
|
61
|
+
"latency_ms": 10000,
|
|
62
|
+
"thinking_enabled": true
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"fixture_id": "demanding-multi-surface",
|
|
66
|
+
"fixture_band": "demanding",
|
|
67
|
+
"effort": "max",
|
|
68
|
+
"quality_score": 0.92,
|
|
69
|
+
"finding_recall": 0.92,
|
|
70
|
+
"finding_precision": 0.9,
|
|
71
|
+
"visible_tokens": 8000,
|
|
72
|
+
"latency_ms": 15000,
|
|
73
|
+
"thinking_enabled": true
|
|
74
|
+
}
|
|
75
|
+
],
|
|
76
|
+
"recommendation": {
|
|
77
|
+
"schema_version": "1",
|
|
78
|
+
"thinking_enabled": true,
|
|
79
|
+
"cost_latency_lever": "effort",
|
|
80
|
+
"defaults_unchanged": false,
|
|
81
|
+
"minimum_quality": 0.8,
|
|
82
|
+
"recommendation_by_band": {
|
|
83
|
+
"easy": {
|
|
84
|
+
"recommended_effort": "medium",
|
|
85
|
+
"cited_row": {
|
|
86
|
+
"fixture_id": "easy-comment-preservation",
|
|
87
|
+
"fixture_band": "easy",
|
|
88
|
+
"effort": "medium",
|
|
89
|
+
"quality_score": 0.95,
|
|
90
|
+
"finding_recall": 0.95,
|
|
91
|
+
"finding_precision": 0.95,
|
|
92
|
+
"visible_tokens": 1600,
|
|
93
|
+
"latency_ms": 4000,
|
|
94
|
+
"thinking_enabled": true
|
|
95
|
+
},
|
|
96
|
+
"blocker": null
|
|
97
|
+
},
|
|
98
|
+
"medium": {
|
|
99
|
+
"recommended_effort": "high",
|
|
100
|
+
"cited_row": {
|
|
101
|
+
"fixture_id": "medium-constants-and-types",
|
|
102
|
+
"fixture_band": "medium",
|
|
103
|
+
"effort": "high",
|
|
104
|
+
"quality_score": 0.9,
|
|
105
|
+
"finding_recall": 0.9,
|
|
106
|
+
"finding_precision": 0.9,
|
|
107
|
+
"visible_tokens": 3200,
|
|
108
|
+
"latency_ms": 7000,
|
|
109
|
+
"thinking_enabled": true
|
|
110
|
+
},
|
|
111
|
+
"blocker": null
|
|
112
|
+
},
|
|
113
|
+
"demanding": {
|
|
114
|
+
"recommended_effort": "max",
|
|
115
|
+
"cited_row": {
|
|
116
|
+
"fixture_id": "demanding-multi-surface",
|
|
117
|
+
"fixture_band": "demanding",
|
|
118
|
+
"effort": "max",
|
|
119
|
+
"quality_score": 0.92,
|
|
120
|
+
"finding_recall": 0.92,
|
|
121
|
+
"finding_precision": 0.9,
|
|
122
|
+
"visible_tokens": 8000,
|
|
123
|
+
"latency_ms": 15000,
|
|
124
|
+
"thinking_enabled": true
|
|
125
|
+
},
|
|
126
|
+
"blocker": null
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
},
|
|
130
|
+
"skill_defaults": {
|
|
131
|
+
"workflow_family": "e-code-review",
|
|
132
|
+
"thinking_enabled": true,
|
|
133
|
+
"cost_latency_lever": "effort",
|
|
134
|
+
"skill_levels": ["low", "medium", "xhigh"],
|
|
135
|
+
"default_by_band": {
|
|
136
|
+
"easy": {
|
|
137
|
+
"skill_effort": "medium",
|
|
138
|
+
"evaluation_effort": "medium",
|
|
139
|
+
"cited_row": {
|
|
140
|
+
"fixture_id": "easy-comment-preservation",
|
|
141
|
+
"fixture_band": "easy",
|
|
142
|
+
"effort": "medium",
|
|
143
|
+
"quality_score": 0.95,
|
|
144
|
+
"finding_recall": 0.95,
|
|
145
|
+
"finding_precision": 0.95,
|
|
146
|
+
"visible_tokens": 1600,
|
|
147
|
+
"latency_ms": 4000,
|
|
148
|
+
"thinking_enabled": true
|
|
149
|
+
},
|
|
150
|
+
"blocker": null
|
|
151
|
+
},
|
|
152
|
+
"medium": {
|
|
153
|
+
"skill_effort": "xhigh",
|
|
154
|
+
"evaluation_effort": "high",
|
|
155
|
+
"cited_row": {
|
|
156
|
+
"fixture_id": "medium-constants-and-types",
|
|
157
|
+
"fixture_band": "medium",
|
|
158
|
+
"effort": "high",
|
|
159
|
+
"quality_score": 0.9,
|
|
160
|
+
"finding_recall": 0.9,
|
|
161
|
+
"finding_precision": 0.9,
|
|
162
|
+
"visible_tokens": 3200,
|
|
163
|
+
"latency_ms": 7000,
|
|
164
|
+
"thinking_enabled": true
|
|
165
|
+
},
|
|
166
|
+
"blocker": null
|
|
167
|
+
},
|
|
168
|
+
"demanding": {
|
|
169
|
+
"skill_effort": "xhigh",
|
|
170
|
+
"evaluation_effort": "max",
|
|
171
|
+
"cited_row": {
|
|
172
|
+
"fixture_id": "demanding-multi-surface",
|
|
173
|
+
"fixture_band": "demanding",
|
|
174
|
+
"effort": "max",
|
|
175
|
+
"quality_score": 0.92,
|
|
176
|
+
"finding_recall": 0.92,
|
|
177
|
+
"finding_precision": 0.9,
|
|
178
|
+
"visible_tokens": 8000,
|
|
179
|
+
"latency_ms": 15000,
|
|
180
|
+
"thinking_enabled": true
|
|
181
|
+
},
|
|
182
|
+
"blocker": null
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|