claude-dev-env 1.95.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/_shared/advisor/CLAUDE.md +2 -2
  2. package/_shared/advisor/advisor-protocol.md +20 -20
  3. package/_shared/advisor/scripts/config/advisor_scripts_constants/model_tier_run_validator_constants.py +15 -12
  4. package/_shared/advisor/scripts/model_tier_run_validator.py +11 -10
  5. package/_shared/advisor/scripts/tests/test_model_tier_run_validator.py +25 -19
  6. package/_shared/advisor/scripts/tests/test_tier_model_ids.py +17 -17
  7. package/_shared/advisor/scripts/tier_model_ids.py +18 -18
  8. package/_shared/pr-loop/CLAUDE.md +1 -0
  9. package/_shared/pr-loop/scripts/CLAUDE.md +2 -1
  10. package/_shared/pr-loop/scripts/README.md +1 -0
  11. package/_shared/pr-loop/scripts/code_rules_gate.py +253 -1980
  12. package/_shared/pr-loop/scripts/code_rules_gate_parts/CLAUDE.md +32 -0
  13. package/_shared/pr-loop/scripts/code_rules_gate_parts/__init__.py +7 -0
  14. package/_shared/pr-loop/scripts/code_rules_gate_parts/added_line_maps.py +268 -0
  15. package/_shared/pr-loop/scripts/code_rules_gate_parts/enforcer_loading.py +172 -0
  16. package/_shared/pr-loop/scripts/code_rules_gate_parts/gate_arguments.py +70 -0
  17. package/_shared/pr-loop/scripts/code_rules_gate_parts/gate_running.py +326 -0
  18. package/_shared/pr-loop/scripts/code_rules_gate_parts/git_blob_readers.py +85 -0
  19. package/_shared/pr-loop/scripts/code_rules_gate_parts/git_file_sets.py +331 -0
  20. package/_shared/pr-loop/scripts/code_rules_gate_parts/staged_test_running.py +369 -0
  21. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/conftest.py +14 -0
  22. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_added_line_maps.py +118 -0
  23. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_enforcer_loading.py +17 -0
  24. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_gate_arguments.py +29 -0
  25. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_gate_running.py +99 -0
  26. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_git_blob_readers.py +69 -0
  27. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_git_file_sets.py +137 -0
  28. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_staged_test_running.py +116 -0
  29. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_violation_scoping.py +75 -0
  30. package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_wrapper_plumb_check.py +49 -0
  31. package/_shared/pr-loop/scripts/code_rules_gate_parts/violation_scoping.py +328 -0
  32. package/_shared/pr-loop/scripts/code_rules_gate_parts/wrapper_plumb_check.py +206 -0
  33. package/_shared/pr-loop/scripts/pr_loop_shared_constants/code_rules_gate_constants.py +24 -17
  34. package/_shared/pr-loop/scripts/pr_loop_shared_constants/reviews_disabled_constants.py +1 -0
  35. package/_shared/pr-loop/scripts/reviews_disabled.py +19 -2
  36. package/_shared/pr-loop/scripts/test_code_rules_gate.py +278 -0
  37. package/_shared/pr-loop/scripts/tests/test_code_rules_gate_constants.py +6 -39
  38. package/_shared/pr-loop/scripts/tests/test_reviews_disabled.py +43 -0
  39. package/_shared/pr-loop/worker-spawn.md +186 -0
  40. package/agents/code-verifier.md +1 -1
  41. package/bin/ever-shipped-skills.mjs +3 -0
  42. package/bin/expand_home_directory_tokens.mjs +1 -1
  43. package/bin/install.mjs +5 -2
  44. package/hooks/advisory/refactor_guard.py +3 -4
  45. package/hooks/blocking/CLAUDE.md +7 -1
  46. package/hooks/blocking/block_main_commit.py +2 -2
  47. package/hooks/blocking/claude_md_orphan_file_blocker.py +75 -699
  48. package/hooks/blocking/claude_md_orphan_file_blocker_parts/CLAUDE.md +28 -0
  49. package/hooks/blocking/claude_md_orphan_file_blocker_parts/__init__.py +1 -0
  50. package/hooks/blocking/claude_md_orphan_file_blocker_parts/config/__init__.py +1 -0
  51. package/hooks/blocking/claude_md_orphan_file_blocker_parts/config/orphan_blocker_constants.py +18 -0
  52. package/hooks/blocking/claude_md_orphan_file_blocker_parts/decision.py +81 -0
  53. package/hooks/blocking/claude_md_orphan_file_blocker_parts/references.py +307 -0
  54. package/hooks/blocking/claude_md_orphan_file_blocker_parts/scan_plan.py +124 -0
  55. package/hooks/blocking/claude_md_orphan_file_blocker_parts/subtree_scan.py +179 -0
  56. package/hooks/blocking/claude_md_orphan_file_blocker_parts/tests/conftest.py +10 -0
  57. package/hooks/blocking/claude_md_orphan_file_blocker_parts/tests/test_decision.py +34 -0
  58. package/hooks/blocking/claude_md_orphan_file_blocker_parts/tests/test_references.py +42 -0
  59. package/hooks/blocking/claude_md_orphan_file_blocker_parts/tests/test_scan_plan.py +27 -0
  60. package/hooks/blocking/claude_md_orphan_file_blocker_parts/tests/test_subtree_scan.py +30 -0
  61. package/hooks/blocking/code_rules_boolean_mustcheck.py +1 -1
  62. package/hooks/blocking/code_rules_mock_completeness.py +1 -1
  63. package/hooks/blocking/code_rules_optional_params.py +2 -2
  64. package/hooks/blocking/code_rules_shared.py +1 -1
  65. package/hooks/blocking/code_rules_test_assertions.py +1 -1
  66. package/hooks/blocking/code_rules_typeddict_stub.py +1 -1
  67. package/hooks/blocking/gh_pr_author_enforcer.py +1 -1
  68. package/hooks/blocking/inventory_intent_records/CLAUDE.md +26 -0
  69. package/hooks/blocking/inventory_intent_records/__init__.py +1 -0
  70. package/hooks/blocking/inventory_intent_records/config/__init__.py +1 -0
  71. package/hooks/blocking/inventory_intent_records/config/intent_records_constants.py +20 -0
  72. package/hooks/blocking/inventory_intent_records/records.py +271 -0
  73. package/hooks/blocking/inventory_intent_records/tests/conftest.py +10 -0
  74. package/hooks/blocking/inventory_intent_records/tests/test_records.py +80 -0
  75. package/hooks/blocking/package_inventory_stale_blocker.py +54 -384
  76. package/hooks/blocking/package_inventory_stale_blocker_parts/CLAUDE.md +26 -0
  77. package/hooks/blocking/package_inventory_stale_blocker_parts/__init__.py +1 -0
  78. package/hooks/blocking/package_inventory_stale_blocker_parts/config/__init__.py +1 -0
  79. package/hooks/blocking/package_inventory_stale_blocker_parts/config/inventory_blocker_constants.py +16 -0
  80. package/hooks/blocking/package_inventory_stale_blocker_parts/decision.py +84 -0
  81. package/hooks/blocking/package_inventory_stale_blocker_parts/inventory_detection.py +307 -0
  82. package/hooks/blocking/package_inventory_stale_blocker_parts/tests/conftest.py +10 -0
  83. package/hooks/blocking/package_inventory_stale_blocker_parts/tests/test_decision.py +38 -0
  84. package/hooks/blocking/package_inventory_stale_blocker_parts/tests/test_inventory_detection.py +61 -0
  85. package/hooks/blocking/pii_payload_scan.py +138 -42
  86. package/hooks/blocking/pii_prevention_blocker.py +185 -291
  87. package/hooks/blocking/pii_prevention_blocker_parts/CLAUDE.md +24 -0
  88. package/hooks/blocking/pii_prevention_blocker_parts/__init__.py +1 -0
  89. package/hooks/blocking/pii_prevention_blocker_parts/config/__init__.py +1 -0
  90. package/hooks/blocking/pii_prevention_blocker_parts/config/repository_resolution_constants.py +28 -0
  91. package/hooks/blocking/pii_prevention_blocker_parts/repository_exemption.py +214 -0
  92. package/hooks/blocking/pii_prevention_blocker_parts/repository_resolution.py +208 -0
  93. package/hooks/blocking/pr_description_command_parser.py +8 -4
  94. package/hooks/blocking/precommit_code_rules_gate.py +3 -3
  95. package/hooks/blocking/tdd_enforcer.py +97 -608
  96. package/hooks/blocking/tdd_enforcer_parts/CLAUDE.md +30 -0
  97. package/hooks/blocking/tdd_enforcer_parts/__init__.py +1 -0
  98. package/hooks/blocking/tdd_enforcer_parts/candidate_paths.py +142 -0
  99. package/hooks/blocking/tdd_enforcer_parts/config/__init__.py +1 -0
  100. package/hooks/blocking/tdd_enforcer_parts/config/tdd_enforcer_constants.py +32 -0
  101. package/hooks/blocking/tdd_enforcer_parts/content_analysis.py +268 -0
  102. package/hooks/blocking/tdd_enforcer_parts/decisions.py +92 -0
  103. package/hooks/blocking/tdd_enforcer_parts/freshness.py +80 -0
  104. package/hooks/blocking/tdd_enforcer_parts/git_tracking.py +63 -0
  105. package/hooks/blocking/tdd_enforcer_parts/path_classification.py +119 -0
  106. package/hooks/blocking/tdd_enforcer_parts/tests/conftest.py +10 -0
  107. package/hooks/blocking/tdd_enforcer_parts/tests/test_candidate_paths.py +31 -0
  108. package/hooks/blocking/tdd_enforcer_parts/tests/test_content_analysis.py +30 -0
  109. package/hooks/blocking/tdd_enforcer_parts/tests/test_decisions.py +34 -0
  110. package/hooks/blocking/tdd_enforcer_parts/tests/test_freshness.py +28 -0
  111. package/hooks/blocking/tdd_enforcer_parts/tests/test_git_tracking.py +48 -0
  112. package/hooks/blocking/tdd_enforcer_parts/tests/test_path_classification.py +36 -0
  113. package/hooks/blocking/test_inventory_deadlock_resolution.py +154 -0
  114. package/hooks/blocking/test_pii_payload_scan.py +168 -0
  115. package/hooks/blocking/test_tdd_enforcer_restore.py +108 -0
  116. package/hooks/blocking/tests/conftest.py +10 -0
  117. package/hooks/blocking/tests/test_pii_prevention_blocker.py +260 -0
  118. package/hooks/blocking/tests/test_repository_exemption.py +105 -0
  119. package/hooks/blocking/tests/test_repository_resolution.py +108 -0
  120. package/hooks/diagnostic/hook_log_extractor.py +12 -10
  121. package/hooks/git-hooks/post_commit.py +3 -4
  122. package/hooks/hooks_constants/CLAUDE.md +2 -2
  123. package/hooks/hooks_constants/banned_identifiers_constants.py +0 -1
  124. package/hooks/hooks_constants/code_rules_path_utils_constants.py +1 -1
  125. package/hooks/hooks_constants/local_identity.py +59 -8
  126. package/hooks/hooks_constants/pii_prevention_constants.py +0 -6
  127. package/hooks/hooks_constants/test_local_identity.py +105 -3
  128. package/hooks/pyproject.toml +13 -36
  129. package/hooks/session/plugin_data_dir_cleanup.py +0 -1
  130. package/hooks/validation/mypy_validator.py +2 -2
  131. package/hooks/validators/health_check.py +1 -0
  132. package/hooks/validators/mypy_integration.py +2 -0
  133. package/hooks/validators/ruff_integration.py +3 -0
  134. package/hooks/workflow/auto_formatter.py +5 -4
  135. package/package.json +1 -1
  136. package/scripts/CLAUDE.md +4 -0
  137. package/scripts/dev_env_scripts_constants/CLAUDE.md +6 -4
  138. package/scripts/dev_env_scripts_constants/code_review_constants.py +71 -0
  139. package/scripts/dev_env_scripts_constants/grok_worker_constants.py +435 -0
  140. package/scripts/dev_env_scripts_constants/timing.py +7 -1
  141. package/scripts/grok_headless_runner.py +294 -0
  142. package/scripts/grok_worker_preflight.py +410 -0
  143. package/scripts/invoke_code_review.py +463 -0
  144. package/scripts/resolve_worker_spawn.py +619 -0
  145. package/scripts/spawn_grok_batch.py +672 -0
  146. package/scripts/test_grok_headless_runner.py +626 -0
  147. package/scripts/test_grok_worker_preflight.py +1054 -0
  148. package/scripts/test_invoke_code_review.py +672 -0
  149. package/scripts/test_resolve_worker_spawn.py +1014 -0
  150. package/scripts/test_spawn_grok_batch.py +1017 -0
  151. package/skills/CLAUDE.md +5 -3
  152. package/skills/_shared/pr-loop/scripts/build_audit_prompt.py +72 -13
  153. package/skills/_shared/pr-loop/scripts/build_fix_prompt.py +121 -14
  154. package/skills/_shared/pr-loop/scripts/skills_pr_loop_constants/path_resolver_constants.py +78 -0
  155. package/skills/_shared/pr-loop/scripts/test_build_audit_prompt.py +121 -0
  156. package/skills/_shared/pr-loop/scripts/test_build_fix_prompt.py +196 -6
  157. package/skills/autoconverge/CLAUDE.md +3 -3
  158. package/skills/autoconverge/SKILL.md +9 -3
  159. package/skills/autoconverge/reference/CLAUDE.md +2 -2
  160. package/skills/autoconverge/reference/convergence.md +33 -11
  161. package/skills/autoconverge/reference/stop-conditions.md +16 -5
  162. package/skills/autoconverge/workflow/CLAUDE.md +2 -1
  163. package/skills/autoconverge/workflow/converge.clean-audit.test.mjs +7 -2
  164. package/skills/autoconverge/workflow/converge.codex-gate.test.mjs +300 -0
  165. package/skills/autoconverge/workflow/converge.contract.test.mjs +5 -5
  166. package/skills/autoconverge/workflow/converge.copilot-gate.test.mjs +29 -29
  167. package/skills/autoconverge/workflow/converge.fix-progress.test.mjs +1 -1
  168. package/skills/autoconverge/workflow/converge.mjs +200 -16
  169. package/skills/bugteam/CLAUDE.md +2 -2
  170. package/skills/bugteam/CONSTRAINTS.md +3 -2
  171. package/skills/bugteam/PROMPTS.md +7 -6
  172. package/skills/bugteam/SKILL.md +18 -13
  173. package/skills/bugteam/reference/audit-and-teammates.md +215 -35
  174. package/skills/bugteam/reference/design-rationale.md +1 -1
  175. package/skills/bugteam/reference/obstacles/CLAUDE.md +1 -1
  176. package/skills/bugteam/reference/team-setup.md +8 -2
  177. package/skills/codex-review/CLAUDE.md +46 -0
  178. package/skills/codex-review/SKILL.md +181 -0
  179. package/skills/codex-review/reference/CLAUDE.md +15 -0
  180. package/skills/codex-review/reference/cli-contract.md +253 -0
  181. package/skills/codex-review/reference/loop-integration.md +118 -0
  182. package/skills/codex-review/scripts/codex_down_classifier.py +98 -0
  183. package/skills/codex-review/scripts/codex_review_scripts_constants/CLAUDE.md +18 -0
  184. package/skills/codex-review/scripts/codex_review_scripts_constants/__init__.py +1 -0
  185. package/skills/codex-review/scripts/codex_review_scripts_constants/classifier_constants.py +35 -0
  186. package/skills/codex-review/scripts/codex_review_scripts_constants/codex_usage_probe_constants.py +86 -0
  187. package/skills/codex-review/scripts/codex_review_scripts_constants/findings_constants.py +18 -0
  188. package/skills/codex-review/scripts/codex_review_scripts_constants/run_constants.py +45 -0
  189. package/skills/codex-review/scripts/codex_usage_probe.py +573 -0
  190. package/skills/codex-review/scripts/fixtures/auth_failure_synthetic.txt +1 -0
  191. package/skills/codex-review/scripts/fixtures/config_load_failure_v0.125.0.txt +1 -0
  192. package/skills/codex-review/scripts/fixtures/freeform_findings_v0.144.3.txt +6 -0
  193. package/skills/codex-review/scripts/fixtures/model_rejection_v0.125.0.jsonl +5 -0
  194. package/skills/codex-review/scripts/fixtures/structured_findings.txt +13 -0
  195. package/skills/codex-review/scripts/fixtures/success_stream_v0.144.3.jsonl +6 -0
  196. package/skills/codex-review/scripts/fixtures/unknown_failure_synthetic.txt +1 -0
  197. package/skills/codex-review/scripts/fixtures/usage_limit_synthetic.txt +1 -0
  198. package/skills/codex-review/scripts/parse_codex_findings.py +207 -0
  199. package/skills/codex-review/scripts/run_codex_review.py +415 -0
  200. package/skills/codex-review/scripts/test_codex_down_classifier.py +143 -0
  201. package/skills/codex-review/scripts/test_codex_usage_probe.py +678 -0
  202. package/skills/codex-review/scripts/test_parse_codex_findings.py +130 -0
  203. package/skills/codex-review/scripts/test_run_codex_review.py +812 -0
  204. package/skills/codex-review/test_skill_scaffold.py +192 -0
  205. package/skills/grok-spawn/CLAUDE.md +28 -0
  206. package/skills/grok-spawn/SKILL.md +226 -0
  207. package/skills/grok-spawn/reference/flag-profiles.md +132 -0
  208. package/skills/grok-spawn/reference/worker-briefs.md +152 -0
  209. package/skills/grokify/SKILL.md +9 -1
  210. package/skills/grokify/capability-claims.test.mjs +28 -0
  211. package/skills/grokify/evals/README.md +72 -0
  212. package/skills/grokify/evals/parse-payload.test.mjs +171 -0
  213. package/skills/grokify/evals/run-capability-evals.mjs +545 -0
  214. package/skills/orchestrator/SKILL.md +15 -11
  215. package/skills/orchestrator-refresh/SKILL.md +5 -5
  216. package/skills/pr-converge/SKILL.md +34 -13
  217. package/skills/pr-converge/reference/convergence-gates.md +42 -15
  218. package/skills/pr-converge/reference/fix-protocol.md +1 -1
  219. package/skills/pr-converge/reference/ground-rules.md +1 -1
  220. package/skills/pr-converge/reference/per-tick.md +130 -42
  221. package/skills/pr-converge/reference/state-schema.md +10 -0
  222. package/skills/pr-converge/scripts/CLAUDE.md +2 -0
  223. package/skills/pr-converge/scripts/_pr_converge_path_setup.py +5 -1
  224. package/skills/pr-converge/scripts/check_convergence.py +605 -29
  225. package/skills/pr-converge/scripts/check_convergence_availability.py +232 -0
  226. package/skills/pr-converge/scripts/check_convergence_gates.py +279 -235
  227. package/skills/pr-converge/scripts/check_convergence_thread_gates.py +1 -1
  228. package/skills/pr-converge/scripts/pr_converge_scripts_constants/convergence_gate_constants.py +36 -2
  229. package/skills/pr-converge/scripts/test__pr_converge_path_setup.py +4 -0
  230. package/skills/pr-converge/scripts/test_check_convergence.py +71 -3
  231. package/skills/pr-converge/scripts/test_check_convergence_availability.py +326 -0
  232. package/skills/pr-converge/scripts/test_check_convergence_codex.py +507 -0
  233. package/skills/pr-converge/scripts/test_check_convergence_contract.py +89 -17
  234. package/skills/pr-converge/scripts/test_check_convergence_fixture.py +179 -0
  235. package/skills/pr-converge/scripts/test_check_convergence_gates.py +84 -68
  236. package/skills/pr-converge/scripts/test_check_convergence_thread_gates.py +24 -0
  237. package/skills/pr-converge/test_step5_host_branch.py +106 -0
  238. package/skills/pr-loop-cloud-transport/SKILL.md +2 -0
  239. package/skills/reviewer-gates/SKILL.md +7 -5
  240. package/skills/team-advisor/SKILL.md +7 -7
@@ -0,0 +1,152 @@
1
+ # Worker brief templates
2
+
3
+ Prompt-part bodies for headless grok workers. Copy a template into a part file,
4
+ fill every bracket, and list the part paths on the worker's `prompt_parts`.
5
+
6
+ The batch launcher prepends a tool-profile header before the joined parts:
7
+
8
+ - `readonly` — no write, edit, or shell
9
+ - `build` — full tools; never commit, push, or call `gh`
10
+
11
+ Put the role brief first, the task body next, the report contract last.
12
+
13
+ ---
14
+
15
+ ## Read-only investigation brief
16
+
17
+ Use with `tool_profile: "readonly"`. For repo-only scans set `is_repo_only: true`
18
+ so the launcher also passes `--disable-web-search`.
19
+
20
+ ```markdown
21
+ # Role: read-only investigation
22
+
23
+ You investigate only. You do not write files, edit files, or run shell commands.
24
+
25
+ ## Scope
26
+
27
+ - Working directory: [absolute path]
28
+ - Paths or symbols in scope: [list]
29
+ - Question to answer: [one clear question]
30
+ - Out of scope: [list]
31
+
32
+ ## Method
33
+
34
+ 1. Read the named paths and their direct callers or callees as needed.
35
+ 2. Cite every claim with `file:line`.
36
+ 3. Prefer measured facts (names, signatures, call sites) over guesses.
37
+ 4. When a fact is unproven, label it `unverified`.
38
+
39
+ ## Hard stops
40
+
41
+ - No Write, Edit, or Bash.
42
+ - No commits, pushes, or `gh`.
43
+ - No expanding scope past the listed paths without noting it as an open question.
44
+
45
+ ## Done when
46
+
47
+ You can answer the question with file:line evidence, or you can name the exact
48
+ gap that blocks an answer.
49
+ ```
50
+
51
+ ---
52
+
53
+ ## Build brief
54
+
55
+ Use with `tool_profile: "build"`. The worker may edit and run tests. The lead
56
+ session owns every git and GitHub step.
57
+
58
+ ```markdown
59
+ # Role: build worker
60
+
61
+ You edit code and run tests for one closed task. You never commit, push, or call
62
+ `gh`. The lead session stages, verifies, commits, pushes, and posts.
63
+
64
+ ## Scope
65
+
66
+ - Working directory: [absolute path]
67
+ - Task: [one sentence]
68
+ - Files you may touch: [list]
69
+ - Files you must not touch: [list]
70
+ - Acceptance lines (each must map to evidence in the report):
71
+ 1. [line]
72
+ 2. [line]
73
+
74
+ ## Method
75
+
76
+ 1. Read the in-scope files before editing.
77
+ 2. Write or update tests that pin the acceptance lines (red first when you add
78
+ behavior).
79
+ 3. Make the smallest edit that satisfies the acceptance lines.
80
+ 4. Run the named test commands and capture pass/fail output.
81
+ 5. Stop with a stage-ready tree and a full report. Do not commit.
82
+
83
+ ## Hard stops
84
+
85
+ - Never `git commit`, `git push`, or `gh`.
86
+ - Never force-push, rewrite shared history, or change git config.
87
+ - Never expand into files outside the allow list without an open question.
88
+ - Never mark acceptance done without command output or a file:line proof.
89
+
90
+ ## Done when
91
+
92
+ Every acceptance line has evidence, tests you own are green (or failures are
93
+ listed with command output), and the report lists every changed file.
94
+ ```
95
+
96
+ ---
97
+
98
+ ## Report contract
99
+
100
+ Append this part (or paste its sections into the task body) for every worker —
101
+ readonly and build alike. The lead session uses it to verify and to relay open
102
+ questions to an advisor when needed.
103
+
104
+ ```markdown
105
+ # Report contract
106
+
107
+ End your turn with exactly these sections, in this order. Use plain markdown.
108
+ No tool calls after the report.
109
+
110
+ ## Changed files
111
+
112
+ - List every path you wrote or edited, one per line.
113
+ - Write `none` when the role is read-only or no edit landed.
114
+
115
+ ## Red-green evidence
116
+
117
+ - For each test you added or changed: the failing command/output before the
118
+ fix (red), then the passing command/output after (green).
119
+ - Write `n/a — investigation only` for read-only workers.
120
+
121
+ ## Acceptance mapping
122
+
123
+ For each acceptance line from the brief:
124
+
125
+ - **Acceptance:** [quote the line]
126
+ - **Status:** met | not met | blocked
127
+ - **Evidence:** [command output summary or `file:line` proof]
128
+
129
+ ## Test results
130
+
131
+ - Commands run (full command lines)
132
+ - Exit codes
133
+ - Short pass/fail summary (paste key lines; do not dump huge logs)
134
+
135
+ ## Open questions
136
+
137
+ - Questions for the lead session or advisor (empty list if none)
138
+ - Unverified claims that need a second look
139
+ - Scope gaps that blocked a full answer
140
+ ```
141
+
142
+ ---
143
+
144
+ ## Part assembly tips
145
+
146
+ 1. Keep task-specific detail in its own part file so the brief templates stay
147
+ reusable.
148
+ 2. Absolute paths only in `prompt_parts` — the launcher reads them as given.
149
+ 3. One worker, one closed scope. Split large work into more workers rather than
150
+ one long brief.
151
+ 4. The lead session fills bracketed fields before launch; workers never see the
152
+ skill folder unless you copy text into their part files.
@@ -12,7 +12,9 @@ One paste-ready handoff turns this session's work into a plan a Grok Build sessi
12
12
 
13
13
  ## Gotchas
14
14
 
15
- - Grok Build cannot spawn Claude subagents. The advisor is an out-of-process `claude -p` session never write Agent-tool, `session-advisor`, or SendMessage instructions into the handoff.
15
+ - Grok Build can use `spawn_subagent`, `--agent` / agent definitions, and can read skills under the user's Claude config paths. Skill evals measure these when `GROK_CAPABILITY_EVALS=1` (see `evals/README.md`).
16
+ - The **grokify handoff's Claude-tier advisor** is an out-of-process `claude -p` bind/resume. Never write Claude Agent-tool, `session-advisor`, or SendMessage protocol into the handoff for that advisor path — product design for a Claude-model advisor, not a claim that Grok lacks agents.
17
+ - Grok does not expose a Claude/GSD Workflow tool. When that tool is required, workflow orchestration stays with Claude.
16
18
  - A `--resume` after a usage-limit failover to another binary fails, because a session store belongs to the binary that minted it. The handoff must tell Grok to treat that failure as starting over: re-send the charter plus a compact recap, capture the new `session_id`.
17
19
  - Conversation-relative phrases ("as discussed", "the plan above", "the earlier choice") are dead text to Grok — every statement stands on its own.
18
20
  - Copy findings' measured numbers and `file:line` citations into the handoff exactly, and label each figure measured, bounded, or unverified.
@@ -51,8 +53,14 @@ The user types `/grokify`, alone or with guidance.
51
53
  |---|---|
52
54
  | `SKILL.md` | Trigger, process, fixed advisor structure. |
53
55
  | `templates/handoff-template.md` | Section-by-section skeleton of the handoff prompt. |
56
+ | `capability-claims.test.mjs` | Offline static guards on capability wording. |
57
+ | `evals/README.md` | How to run opt-in live capability evals. |
58
+ | `evals/run-capability-evals.mjs` | Live E1–E5 runner (manual / opt-in only). |
59
+ | `evals/parse-payload.test.mjs` | Offline unit tests for the eval output parser, run under `npm test`. |
54
60
 
55
61
  ## Folder map
56
62
 
57
63
  - `SKILL.md` — the whole workflow.
58
64
  - `templates/` — the handoff skeleton.
65
+ - `evals/` — the opt-in live runner plus its offline parser unit tests.
66
+ - `capability-claims.test.mjs` — offline claim guards for `npm test`.
@@ -0,0 +1,28 @@
1
+ import { test } from 'node:test';
2
+ import { strict as assert } from 'node:assert';
3
+ import { readFileSync } from 'node:fs';
4
+ import { fileURLToPath } from 'node:url';
5
+ import { dirname, join } from 'node:path';
6
+
7
+ const skillDirectory = dirname(fileURLToPath(import.meta.url));
8
+ const skillSource = readFileSync(join(skillDirectory, 'SKILL.md'), 'utf8');
9
+ const handoffSource = readFileSync(
10
+ join(skillDirectory, 'templates', 'handoff-template.md'),
11
+ 'utf8',
12
+ );
13
+
14
+ test('SKILL.md does not claim Grok cannot spawn Claude subagents', () => {
15
+ assert.equal(
16
+ skillSource.includes('cannot spawn Claude subagents'),
17
+ false,
18
+ 'SKILL.md must not contain the false substring "cannot spawn Claude subagents"',
19
+ );
20
+ });
21
+
22
+ test('SKILL.md documents the out-of-process claude -p advisor path', () => {
23
+ assert.match(skillSource, /claude -p/);
24
+ });
25
+
26
+ test('handoff template keeps the claude -p advisor bind path', () => {
27
+ assert.match(handoffSource, /claude -p/);
28
+ });
@@ -0,0 +1,72 @@
1
+ # Grok capability evals (opt-in only)
2
+
3
+ Live measurements of Grok Build capabilities used by the `grokify` skill wording.
4
+
5
+ **Manual / opt-in only.** These evals never gate default `npm test`, pr-check, or CI.
6
+
7
+ ## Requirements
8
+
9
+ - `grok` on `PATH` and authenticated (`grok models` exits 0). The runner calls `grok` by default; set `GROK_BIN` to point it at another binary.
10
+ - Network access for the Grok API
11
+ - A writable platform temp directory (`$env:TEMP` on Windows, `$TMPDIR` / `/tmp` elsewhere)
12
+ - For E3, an agent `grok` can load under `--agent`. The runner passes the agent named by `GROK_CAPABILITY_EVAL_AGENT` (default `Explore`); set that variable to an agent your `grok` resolves when the default does not exist for you.
13
+
14
+ ## Run
15
+
16
+ From the package root (`packages/claude-dev-env`):
17
+
18
+ ```powershell
19
+ # Windows PowerShell
20
+ $env:GROK_CAPABILITY_EVALS = "1"
21
+ node skills/grokify/evals/run-capability-evals.mjs
22
+ ```
23
+
24
+ ```bash
25
+ # POSIX
26
+ GROK_CAPABILITY_EVALS=1 node skills/grokify/evals/run-capability-evals.mjs
27
+ ```
28
+
29
+ From this skill folder (`packages/claude-dev-env/skills/grokify`):
30
+
31
+ ```powershell
32
+ # Windows PowerShell
33
+ $env:GROK_CAPABILITY_EVALS = "1"
34
+ node evals/run-capability-evals.mjs
35
+ ```
36
+
37
+ ```bash
38
+ # POSIX
39
+ GROK_CAPABILITY_EVALS=1 node evals/run-capability-evals.mjs
40
+ ```
41
+
42
+ Or pass `--run` without the env var (paths match the cwd above):
43
+
44
+ ```powershell
45
+ # from package root
46
+ node skills/grokify/evals/run-capability-evals.mjs --run
47
+ # from this skill folder
48
+ node evals/run-capability-evals.mjs --run
49
+ ```
50
+
51
+ Without `GROK_CAPABILITY_EVALS=1` or `--run`, the script prints how to opt in and exits 0 (no live calls).
52
+
53
+ ## What it measures
54
+
55
+ | Eval | Assertion |
56
+ |------|-----------|
57
+ | E1 | `can_spawn_subagent_tool === true` and tool list includes `spawn_subagent` |
58
+ | E2 | `spawn_succeeded === true` (child reports `SPAWN_OK`) |
59
+ | E3 | `skill_read_ok === true` and `agent_definition_loaded === true` (skill file read under `--agent`) |
60
+ | E4 | Probe write under the eval cwd succeeds (soft: hooks log may show `global/settings` + `pre_tool_use`) |
61
+ | E5 | `has_workflow_tool === false` and `result === "no_tool"` (both fields agree on absence) |
62
+
63
+ Each live `grok` call uses a unique `--leader-socket` under a fresh temp directory so concurrent runs do not share a socket.
64
+
65
+ ## Offline static guard
66
+
67
+ `../capability-claims.test.mjs` runs under package `npm test` (no live `grok`):
68
+
69
+ - `SKILL.md` must not contain `cannot spawn Claude subagents`
70
+ - `SKILL.md` and the handoff template must still mention `claude -p` for the advisor path
71
+
72
+ `parse-payload.test.mjs` also runs under package `npm test` (no live `grok`): it unit-tests the runner's output parser (`tryParseJsonObject`, `extractResultText`, `parsePayload`, `isWorkflowToolAbsent`).
@@ -0,0 +1,171 @@
1
+ import { test } from 'node:test';
2
+ import { strict as assert } from 'node:assert';
3
+ import {
4
+ tryParseJsonObject,
5
+ extractResultText,
6
+ parsePayload,
7
+ isWorkflowToolAbsent,
8
+ } from './run-capability-evals.mjs';
9
+
10
+ const E1_CLAIM_PAYLOAD = {
11
+ can_spawn_subagent_tool: true,
12
+ tool_names: ['spawn_subagent'],
13
+ agents_dir_exists: true,
14
+ skills_dir_exists: true,
15
+ };
16
+
17
+ test('tryParseJsonObject parses a pure JSON object string', () => {
18
+ const parsed = tryParseJsonObject(JSON.stringify(E1_CLAIM_PAYLOAD));
19
+ assert.deepEqual(parsed, E1_CLAIM_PAYLOAD);
20
+ });
21
+
22
+ test('tryParseJsonObject strips leading prose before the first brace', () => {
23
+ const withProse = `Here is the inventory:\n${JSON.stringify(E1_CLAIM_PAYLOAD)}`;
24
+ const parsed = tryParseJsonObject(withProse);
25
+ assert.deepEqual(parsed, E1_CLAIM_PAYLOAD);
26
+ });
27
+
28
+ test('tryParseJsonObject extracts the object when trailing prose carries braces', () => {
29
+ const withTrailingBraces = `${JSON.stringify(E1_CLAIM_PAYLOAD)}\nNote: see {run-directory} for logs.`;
30
+ const parsed = tryParseJsonObject(withTrailingBraces);
31
+ assert.deepEqual(parsed, E1_CLAIM_PAYLOAD);
32
+ });
33
+
34
+ test('tryParseJsonObject ignores a stray brace pair in leading prose', () => {
35
+ const withLeadingBraces = `Here is the result (see {} for details):\n${JSON.stringify(E1_CLAIM_PAYLOAD)}`;
36
+ const parsed = tryParseJsonObject(withLeadingBraces);
37
+ assert.deepEqual(parsed, E1_CLAIM_PAYLOAD);
38
+ });
39
+
40
+ test('tryParseJsonObject returns null for empty input', () => {
41
+ assert.equal(tryParseJsonObject(''), null);
42
+ assert.equal(tryParseJsonObject(' '), null);
43
+ });
44
+
45
+ test('extractResultText reads Grok top-level text field', () => {
46
+ const nestedJson = JSON.stringify(E1_CLAIM_PAYLOAD);
47
+ const envelope = JSON.stringify({ text: nestedJson, model: 'grok' });
48
+ assert.equal(extractResultText(envelope), nestedJson);
49
+ });
50
+
51
+ test('extractResultText prefers Claude result over Grok text', () => {
52
+ const envelope = JSON.stringify({
53
+ result: '{"from":"result"}',
54
+ text: '{"from":"text"}',
55
+ });
56
+ assert.equal(extractResultText(envelope), '{"from":"result"}');
57
+ });
58
+
59
+ test('extractResultText prefers Grok text over a conversational message', () => {
60
+ const envelope = JSON.stringify({
61
+ message: 'working',
62
+ text: '{"from":"text"}',
63
+ });
64
+ assert.equal(extractResultText(envelope), '{"from":"text"}');
65
+ });
66
+
67
+ test('extractResultText reads Claude message field', () => {
68
+ const envelope = JSON.stringify({ message: '{"from":"message"}' });
69
+ assert.equal(extractResultText(envelope), '{"from":"message"}');
70
+ });
71
+
72
+ test('extractResultText reads Claude stream result events', () => {
73
+ const envelope = JSON.stringify([
74
+ { type: 'assistant', content: 'thinking' },
75
+ {
76
+ type: 'result',
77
+ result: JSON.stringify(E1_CLAIM_PAYLOAD),
78
+ },
79
+ ]);
80
+ assert.equal(extractResultText(envelope), JSON.stringify(E1_CLAIM_PAYLOAD));
81
+ });
82
+
83
+ test('extractResultText prefers the last result event in a stream array', () => {
84
+ const intermediateClaim = { can_spawn_subagent_tool: false, tool_names: [] };
85
+ const envelope = JSON.stringify([
86
+ { type: 'assistant', content: 'thinking' },
87
+ { type: 'result', result: JSON.stringify(intermediateClaim) },
88
+ { type: 'assistant', content: 'retry' },
89
+ { type: 'result', result: JSON.stringify(E1_CLAIM_PAYLOAD) },
90
+ ]);
91
+ assert.equal(extractResultText(envelope), JSON.stringify(E1_CLAIM_PAYLOAD));
92
+ });
93
+
94
+ test('parsePayload extracts nested claim JSON from Grok text envelope', () => {
95
+ const envelope = JSON.stringify({
96
+ text: JSON.stringify(E1_CLAIM_PAYLOAD),
97
+ });
98
+ const payload = parsePayload(envelope);
99
+ assert.equal(payload.can_spawn_subagent_tool, true);
100
+ assert.deepEqual(payload.tool_names, ['spawn_subagent']);
101
+ });
102
+
103
+ test('parsePayload extracts nested claim JSON when Grok text has leading prose', () => {
104
+ const envelope = JSON.stringify({
105
+ text: `Capability inventory complete.\n${JSON.stringify(E1_CLAIM_PAYLOAD)}`,
106
+ });
107
+ const payload = parsePayload(envelope);
108
+ assert.equal(payload.can_spawn_subagent_tool, true);
109
+ assert.equal(payload.agents_dir_exists, true);
110
+ });
111
+
112
+ test('parsePayload still reads Claude result envelopes', () => {
113
+ const envelope = JSON.stringify({
114
+ result: JSON.stringify(E1_CLAIM_PAYLOAD),
115
+ });
116
+ const payload = parsePayload(envelope);
117
+ assert.equal(payload.can_spawn_subagent_tool, true);
118
+ });
119
+
120
+ test('parsePayload still reads Claude message envelopes', () => {
121
+ const envelope = JSON.stringify({
122
+ message: JSON.stringify(E1_CLAIM_PAYLOAD),
123
+ });
124
+ const payload = parsePayload(envelope);
125
+ assert.equal(payload.can_spawn_subagent_tool, true);
126
+ });
127
+
128
+ test('parsePayload still reads Claude stream array envelopes', () => {
129
+ const envelope = JSON.stringify([
130
+ { type: 'message', content: 'working' },
131
+ { type: 'result', result: JSON.stringify(E1_CLAIM_PAYLOAD) },
132
+ ]);
133
+ const payload = parsePayload(envelope);
134
+ assert.equal(payload.can_spawn_subagent_tool, true);
135
+ assert.deepEqual(payload.tool_names, ['spawn_subagent']);
136
+ });
137
+
138
+ test('isWorkflowToolAbsent accepts consistent absence', () => {
139
+ assert.equal(
140
+ isWorkflowToolAbsent({ has_workflow_tool: false, result: 'no_tool' }),
141
+ true,
142
+ );
143
+ });
144
+
145
+ test('isWorkflowToolAbsent rejects contradictory true and no_tool', () => {
146
+ assert.equal(
147
+ isWorkflowToolAbsent({ has_workflow_tool: true, result: 'no_tool' }),
148
+ false,
149
+ );
150
+ });
151
+
152
+ test('isWorkflowToolAbsent rejects contradictory false and has_tool', () => {
153
+ assert.equal(
154
+ isWorkflowToolAbsent({ has_workflow_tool: false, result: 'has_tool' }),
155
+ false,
156
+ );
157
+ });
158
+
159
+ test('isWorkflowToolAbsent rejects consistent presence', () => {
160
+ assert.equal(
161
+ isWorkflowToolAbsent({ has_workflow_tool: true, result: 'has_tool' }),
162
+ false,
163
+ );
164
+ });
165
+
166
+ test('isWorkflowToolAbsent rejects partial or missing fields', () => {
167
+ assert.equal(isWorkflowToolAbsent({ has_workflow_tool: false }), false);
168
+ assert.equal(isWorkflowToolAbsent({ result: 'no_tool' }), false);
169
+ assert.equal(isWorkflowToolAbsent(null), false);
170
+ assert.equal(isWorkflowToolAbsent({}), false);
171
+ });