forge-workflow 0.0.3 → 0.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/.claude/commands/dev.md +340 -314
  2. package/.claude/commands/plan.md +521 -478
  3. package/.claude/commands/premerge.md +176 -179
  4. package/.claude/commands/research.md +42 -42
  5. package/.claude/commands/review.md +442 -442
  6. package/.claude/commands/rollback.md +721 -721
  7. package/.claude/commands/ship.md +164 -134
  8. package/.claude/commands/sonarcloud.md +152 -152
  9. package/.claude/commands/status.md +48 -77
  10. package/.claude/commands/validate.md +282 -237
  11. package/.claude/commands/verify.md +221 -221
  12. package/.claude/rules/greptile-review-process.md +285 -285
  13. package/.claude/rules/workflow.md +105 -105
  14. package/.claude/scripts/greptile-resolve.sh +526 -526
  15. package/.claude/scripts/load-env.sh +32 -32
  16. package/.cline/workflows/dev.md +337 -311
  17. package/.cline/workflows/plan.md +518 -475
  18. package/.cline/workflows/premerge.md +173 -176
  19. package/.cline/workflows/research.md +39 -39
  20. package/.cline/workflows/review.md +439 -439
  21. package/.cline/workflows/rollback.md +718 -718
  22. package/.cline/workflows/ship.md +161 -131
  23. package/.cline/workflows/sonarcloud.md +146 -146
  24. package/.cline/workflows/status.md +45 -74
  25. package/.cline/workflows/validate.md +279 -234
  26. package/.cline/workflows/verify.md +218 -218
  27. package/.codex/config.toml +11 -11
  28. package/.codex/skills/dev/SKILL.md +340 -314
  29. package/.codex/skills/plan/SKILL.md +521 -478
  30. package/.codex/skills/premerge/SKILL.md +176 -179
  31. package/.codex/skills/research/SKILL.md +42 -42
  32. package/.codex/skills/review/SKILL.md +442 -442
  33. package/.codex/skills/rollback/SKILL.md +721 -721
  34. package/.codex/skills/ship/SKILL.md +164 -134
  35. package/.codex/skills/sonarcloud/SKILL.md +149 -149
  36. package/.codex/skills/status/SKILL.md +48 -77
  37. package/.codex/skills/validate/SKILL.md +282 -237
  38. package/.codex/skills/verify/SKILL.md +221 -221
  39. package/.cursor/commands/dev.md +337 -311
  40. package/.cursor/commands/plan.md +518 -475
  41. package/.cursor/commands/premerge.md +173 -176
  42. package/.cursor/commands/research.md +39 -39
  43. package/.cursor/commands/review.md +439 -439
  44. package/.cursor/commands/rollback.md +718 -718
  45. package/.cursor/commands/ship.md +161 -131
  46. package/.cursor/commands/sonarcloud.md +146 -146
  47. package/.cursor/commands/status.md +45 -74
  48. package/.cursor/commands/validate.md +279 -234
  49. package/.cursor/commands/verify.md +218 -218
  50. package/.cursor/rules/permissions-guidance.mdc +37 -37
  51. package/.forge/hooks/check-tdd.js +240 -240
  52. package/.github/PLUGIN_TEMPLATE.json +32 -32
  53. package/.github/prompts/dev.prompt.md +342 -316
  54. package/.github/prompts/plan.prompt.md +523 -480
  55. package/.github/prompts/premerge.prompt.md +178 -181
  56. package/.github/prompts/research.prompt.md +44 -44
  57. package/.github/prompts/review.prompt.md +444 -444
  58. package/.github/prompts/rollback.prompt.md +723 -723
  59. package/.github/prompts/ship.prompt.md +166 -136
  60. package/.github/prompts/sonarcloud.prompt.md +151 -151
  61. package/.github/prompts/status.prompt.md +50 -79
  62. package/.github/prompts/validate.prompt.md +284 -239
  63. package/.github/prompts/verify.prompt.md +223 -223
  64. package/.github/workflows/beads-to-github.yml +56 -0
  65. package/.github/workflows/github-to-beads.yml +97 -0
  66. package/.kilocode/workflows/dev.md +341 -315
  67. package/.kilocode/workflows/plan.md +522 -479
  68. package/.kilocode/workflows/premerge.md +177 -180
  69. package/.kilocode/workflows/research.md +43 -43
  70. package/.kilocode/workflows/review.md +443 -443
  71. package/.kilocode/workflows/rollback.md +722 -722
  72. package/.kilocode/workflows/ship.md +165 -135
  73. package/.kilocode/workflows/sonarcloud.md +150 -150
  74. package/.kilocode/workflows/status.md +49 -78
  75. package/.kilocode/workflows/validate.md +283 -238
  76. package/.kilocode/workflows/verify.md +222 -222
  77. package/.mcp.json.example +12 -12
  78. package/.opencode/commands/dev.md +340 -314
  79. package/.opencode/commands/plan.md +521 -478
  80. package/.opencode/commands/premerge.md +176 -179
  81. package/.opencode/commands/research.md +42 -42
  82. package/.opencode/commands/review.md +442 -442
  83. package/.opencode/commands/rollback.md +721 -721
  84. package/.opencode/commands/ship.md +164 -134
  85. package/.opencode/commands/sonarcloud.md +149 -149
  86. package/.opencode/commands/status.md +48 -77
  87. package/.opencode/commands/validate.md +282 -237
  88. package/.opencode/commands/verify.md +221 -221
  89. package/.roo/commands/dev.md +341 -315
  90. package/.roo/commands/plan.md +522 -479
  91. package/.roo/commands/premerge.md +177 -180
  92. package/.roo/commands/research.md +43 -43
  93. package/.roo/commands/review.md +443 -443
  94. package/.roo/commands/rollback.md +722 -722
  95. package/.roo/commands/ship.md +165 -135
  96. package/.roo/commands/sonarcloud.md +150 -150
  97. package/.roo/commands/status.md +49 -78
  98. package/.roo/commands/validate.md +283 -238
  99. package/.roo/commands/verify.md +222 -222
  100. package/AGENTS.md +175 -169
  101. package/CLAUDE.md +100 -99
  102. package/LICENSE +21 -21
  103. package/README.md +429 -414
  104. package/bin/forge-cmd.js +313 -313
  105. package/bin/{forge-validate.js → forge-preflight.js} +309 -303
  106. package/bin/forge.js +4596 -4232
  107. package/docs/AGENT_INSTALL_PROMPT.md +342 -342
  108. package/docs/BEADS_GITHUB_SYNC.md +251 -0
  109. package/docs/ENHANCED_ONBOARDING.md +602 -602
  110. package/docs/EXAMPLES.md +482 -482
  111. package/docs/GREPTILE_SETUP.md +400 -400
  112. package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
  113. package/docs/ROADMAP.md +359 -359
  114. package/docs/SETUP.md +663 -632
  115. package/docs/TOOLCHAIN.md +630 -630
  116. package/docs/VALIDATION.md +363 -363
  117. package/install.sh +40 -1058
  118. package/lefthook.yml +39 -39
  119. package/lib/agents/README.md +198 -198
  120. package/lib/agents/claude.plugin.json +28 -28
  121. package/lib/agents/cline.plugin.json +22 -22
  122. package/lib/agents/codex.plugin.json +19 -19
  123. package/lib/agents/copilot.plugin.json +24 -24
  124. package/lib/agents/cursor.plugin.json +25 -25
  125. package/lib/agents/kilocode.plugin.json +22 -22
  126. package/lib/agents/opencode.plugin.json +20 -20
  127. package/lib/agents/roo.plugin.json +23 -23
  128. package/lib/agents-config.js +2112 -2112
  129. package/lib/beads-health-check.js +143 -0
  130. package/lib/beads-setup.js +341 -0
  131. package/lib/beads-sync-scaffold.js +260 -0
  132. package/lib/commands/dev.js +513 -513
  133. package/lib/commands/plan.js +692 -692
  134. package/lib/commands/recommend.js +119 -119
  135. package/lib/commands/ship.js +377 -377
  136. package/lib/commands/status.js +378 -378
  137. package/lib/commands/validate.js +602 -602
  138. package/lib/context-merge.js +359 -359
  139. package/lib/dep-guard/analyzer.js +294 -294
  140. package/lib/dep-guard/behavior-detector.js +98 -98
  141. package/lib/dep-guard/contract-detector.js +162 -162
  142. package/lib/dep-guard/import-detector.js +498 -498
  143. package/lib/dep-guard/path-utils.js +13 -13
  144. package/lib/dep-guard/rubric.js +120 -120
  145. package/lib/dep-guard/task-parser.js +318 -318
  146. package/lib/detect-agent.js +191 -0
  147. package/lib/detect-worktree.js +47 -0
  148. package/lib/file-hash.js +26 -0
  149. package/lib/husky-migration.js +450 -0
  150. package/lib/lefthook-check.js +65 -0
  151. package/lib/pat-setup.js +207 -0
  152. package/lib/plugin-catalog.js +350 -350
  153. package/lib/plugin-manager.js +166 -166
  154. package/lib/plugin-recommender.js +141 -141
  155. package/lib/project-discovery.js +491 -491
  156. package/lib/setup-action-log.js +139 -0
  157. package/lib/setup-summary-renderer.js +106 -0
  158. package/lib/setup-utils.js +96 -0
  159. package/lib/setup.js +192 -118
  160. package/lib/smart-merge.js +64 -0
  161. package/lib/symlink-utils.js +81 -0
  162. package/lib/workflow-profiles.js +197 -197
  163. package/package.json +131 -129
  164. package/scripts/beads-context.sh +291 -0
  165. package/scripts/beads-context.test.js +563 -0
  166. package/scripts/behavioral-judge.sh +378 -0
  167. package/scripts/benchmark.js +85 -0
  168. package/scripts/branch-protection.js +183 -0
  169. package/scripts/check-agents.js +172 -0
  170. package/scripts/commitlint.js +42 -0
  171. package/scripts/conflict-detect.sh +323 -0
  172. package/scripts/dep-guard-analyze.js +71 -0
  173. package/scripts/dep-guard.sh +811 -0
  174. package/scripts/eval_win.py +249 -0
  175. package/scripts/file-index.sh +399 -0
  176. package/scripts/github-beads-sync/comment.mjs +64 -0
  177. package/scripts/github-beads-sync/config.mjs +148 -0
  178. package/scripts/github-beads-sync/github-api.mjs +131 -0
  179. package/scripts/github-beads-sync/index.mjs +332 -0
  180. package/scripts/github-beads-sync/label-mapper.mjs +54 -0
  181. package/scripts/github-beads-sync/mapping.mjs +78 -0
  182. package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
  183. package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
  184. package/scripts/github-beads-sync/run-bd.mjs +159 -0
  185. package/scripts/github-beads-sync/sanitize.mjs +121 -0
  186. package/scripts/github-beads-sync.config.json +26 -0
  187. package/scripts/improve-command.js +375 -0
  188. package/scripts/lib/eval-runner.js +229 -0
  189. package/scripts/lib/eval-schema.js +135 -0
  190. package/scripts/lib/eval-storage.js +78 -0
  191. package/scripts/lib/grading.js +203 -0
  192. package/scripts/lib/transcript-parser.js +63 -0
  193. package/scripts/lint.js +47 -0
  194. package/scripts/migrate-to-bun-test.js +412 -0
  195. package/scripts/run-command-eval.js +236 -0
  196. package/scripts/smart-status.sh +782 -0
  197. package/scripts/sync-commands.js +571 -0
  198. package/scripts/sync-utils.sh +460 -0
  199. package/scripts/test-dashboard.js +123 -0
  200. package/scripts/test.js +44 -0
  201. package/scripts/validate.sh +94 -0
  202. package/skills/parallel-deep-research/SKILL.md +108 -108
  203. package/skills/parallel-deep-research/evals/README.md +27 -27
  204. package/skills/parallel-deep-research/evals/evals.json +62 -62
  205. package/skills/sonarcloud-analysis/SKILL.md +171 -171
  206. package/skills/sonarcloud-analysis/evals/README.md +27 -27
  207. package/skills/sonarcloud-analysis/evals/evals.json +50 -50
  208. package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
  209. package/docs/WORKFLOW.md +0 -400
@@ -0,0 +1,94 @@
1
+ #!/usr/bin/env bash
2
+ # Unified validation script for Forge project
3
+ # Runs all quality checks in sequence
4
+ # Exits on first failure
5
+
6
+ set -e # Exit on first error
7
+ set -o pipefail # Catch errors in pipes
8
+
9
+ # Colors for output
10
+ if [ -t 1 ]; then
11
+ RED='\033[0;31m'
12
+ GREEN='\033[0;32m'
13
+ YELLOW='\033[1;33m'
14
+ BLUE='\033[0;34m'
15
+ NC='\033[0m' # No Color
16
+ else
17
+ RED=''
18
+ GREEN=''
19
+ YELLOW=''
20
+ BLUE=''
21
+ NC=''
22
+ fi
23
+
24
+ # Print section header
25
+ print_header() {
26
+ echo ""
27
+ echo -e "${BLUE}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
28
+ echo -e "${BLUE}▶ $1${NC}"
29
+ echo -e "${BLUE}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
30
+ }
31
+
32
+ # Print success message
33
+ print_success() {
34
+ echo -e "${GREEN}✓ $1${NC}"
35
+ }
36
+
37
+ # Print error message
38
+ print_error() {
39
+ echo -e "${RED}✗ $1${NC}"
40
+ }
41
+
42
+ # Print warning message
43
+ print_warning() {
44
+ echo -e "${YELLOW}⚠ $1${NC}"
45
+ }
46
+
47
+ echo ""
48
+ echo -e "${BLUE}╔═══════════════════════════════════════════╗${NC}"
49
+ echo -e "${BLUE}║ Forge Quality Gate - Running Checks ║${NC}"
50
+ echo -e "${BLUE}╚═══════════════════════════════════════════╝${NC}"
51
+
52
+ # Step 1: Type Check
53
+ print_header "1/4: Type Check"
54
+ print_warning "SKIPPED (no TypeScript in project)"
55
+
56
+ # Step 2: Lint
57
+ print_header "2/4: Lint"
58
+ if bun run lint; then
59
+ print_success "Lint passed"
60
+ else
61
+ print_error "Lint failed"
62
+ exit 1
63
+ fi
64
+
65
+ # Step 3: Security Audit
66
+ print_header "3/4: Security Audit"
67
+ AUDIT_OUTPUT=$(bun audit 2>&1 || true)
68
+ if echo "$AUDIT_OUTPUT" | grep -qiE 'critical|high'; then
69
+ print_error "Security audit found critical/high vulnerabilities"
70
+ echo "$AUDIT_OUTPUT"
71
+ exit 1
72
+ fi
73
+ if bun audit; then
74
+ print_success "Security audit passed"
75
+ else
76
+ print_warning "Security audit found issues (moderate/low — non-blocking)"
77
+ fi
78
+
79
+ # Step 4: Tests
80
+ print_header "4/4: Tests"
81
+ if bun test; then
82
+ print_success "All tests passed"
83
+ else
84
+ print_error "Tests failed"
85
+ exit 1
86
+ fi
87
+
88
+ echo ""
89
+ echo -e "${GREEN}╔═══════════════════════════════════════════╗${NC}"
90
+ echo -e "${GREEN}║ ✓ All Checks Passed Successfully ║${NC}"
91
+ echo -e "${GREEN}╚═══════════════════════════════════════════╝${NC}"
92
+ echo ""
93
+
94
+ exit 0
@@ -1,108 +1,108 @@
1
- ---
2
- name: parallel-deep-research
3
- description: >
4
- Produces comprehensive research reports that go far beyond what built-in web
5
- search can achieve. Sends research tasks to Parallel AI's pro/ultra processors
6
- which spend 3-25 minutes autonomously crawling, reading, and synthesizing dozens
7
- of sources — returning structured reports with citations. Built-in WebSearch
8
- can only run a few queries; this skill runs an entire research pipeline externally.
9
- No binary install — requires PARALLEL_API_KEY in .env.local. ALWAYS use this
10
- skill instead of doing multiple WebSearch calls when the user needs a comprehensive
11
- report, market analysis, competitive landscape, industry deep-dive, strategic
12
- recommendations, or multi-source synthesis. This is the RIGHT tool for any
13
- research task that would require more than 3-4 web searches to answer properly.
14
- Also trigger during /plan Phase 2 research and /research workflows.
15
- compatibility: Requires PARALLEL_API_KEY in .env.local. Uses curl. Takes 3-25 minutes.
16
- metadata:
17
- author: harshanandak
18
- version: "1.0.0"
19
- ---
20
-
21
- # Parallel Deep Research
22
-
23
- Comprehensive research reports with multi-source synthesis. Use `pro` (3-9 min, $0.10) or `ultra` (5-25 min, $0.30) for deep analysis.
24
-
25
- > **CLI alternative (recommended)**: Install `parallel-cli` for official skill:
26
- > `npx skills add parallel-web/parallel-agent-skills --skill parallel-deep-research`
27
-
28
- ## Setup
29
-
30
- ```bash
31
- API_KEY=$(grep "^PARALLEL_API_KEY=" .env.local | cut -d= -f2)
32
- ```
33
-
34
- ## Create Research Task
35
-
36
- ```bash
37
- curl -s -X POST "https://api.parallel.ai/v1/tasks/runs" \
38
- -H "x-api-key: $API_KEY" \
39
- -H "Content-Type: application/json" \
40
- -d '{
41
- "input": "Analyze the AI chatbot market. Include: size, growth, key players, trends, competitive threats",
42
- "processor": "pro"
43
- }'
44
- ```
45
-
46
- Response: `{"run_id": "trun_abc123...", "status": "queued"}`
47
-
48
- ## Get Result (polling)
49
-
50
- The result endpoint returns both status and output in one call. Poll until `status` is `completed`.
51
-
52
- ```bash
53
- RUN_ID="trun_abc123..."
54
- MAX_POLLS=180 # 30 min max (180 × 10s)
55
-
56
- for i in $(seq 1 $MAX_POLLS); do
57
- RESULT=$(curl -s "https://api.parallel.ai/v1/tasks/runs/$RUN_ID/result" \
58
- -H "x-api-key: $API_KEY")
59
-
60
- STATUS=$(echo "$RESULT" | grep -o '"status":"[^"]*"' | head -1 | cut -d'"' -f4)
61
-
62
- if [ "$STATUS" = "completed" ]; then
63
- echo "$RESULT"
64
- break
65
- elif [ "$STATUS" = "failed" ]; then
66
- echo "Task failed: $RESULT"
67
- break
68
- fi
69
-
70
- echo "Poll $i/$MAX_POLLS — Status: $STATUS"
71
- sleep 10
72
- done
73
-
74
- if [ "$i" = "$MAX_POLLS" ] && [ "$STATUS" != "completed" ] && [ "$STATUS" != "failed" ]; then
75
- echo "Timeout: task did not complete within 30 minutes"
76
- fi
77
- ```
78
-
79
- ## Processors
80
-
81
- | Processor | Speed | Cost | Use For |
82
- |-----------|-------|------|---------|
83
- | pro | 3-9 min | $0.10/task | Market analysis, strategic reports |
84
- | ultra | 5-25 min | $0.30/task | Comprehensive deep research |
85
-
86
- ## Example: Market Analysis
87
-
88
- ```json
89
- {
90
- "input": "Analyze the AI chip market in 2024. Include market size, growth rate, key players (NVIDIA, AMD, Intel), emerging competitors, and 2025 outlook.",
91
- "processor": "pro"
92
- }
93
- ```
94
-
95
- Result: Markdown report with citations.
96
-
97
- ## When to Use
98
-
99
- - Market research and competitive analysis
100
- - Strategic reports requiring multiple sources
101
- - Research that needs synthesis across many documents
102
- - Any task that would need more than 3-4 web searches to answer properly
103
-
104
- For quick facts or single-source lookups, use built-in WebSearch instead.
105
-
106
- ## Timeout
107
-
108
- Set polling timeout to 1800s (30 min) for ultra tasks. Pro tasks typically complete in 3-9 min.
1
+ ---
2
+ name: parallel-deep-research
3
+ description: >
4
+ Produces comprehensive research reports that go far beyond what built-in web
5
+ search can achieve. Sends research tasks to Parallel AI's pro/ultra processors
6
+ which spend 3-25 minutes autonomously crawling, reading, and synthesizing dozens
7
+ of sources — returning structured reports with citations. Built-in WebSearch
8
+ can only run a few queries; this skill runs an entire research pipeline externally.
9
+ No binary install — requires PARALLEL_API_KEY in .env.local. ALWAYS use this
10
+ skill instead of doing multiple WebSearch calls when the user needs a comprehensive
11
+ report, market analysis, competitive landscape, industry deep-dive, strategic
12
+ recommendations, or multi-source synthesis. This is the RIGHT tool for any
13
+ research task that would require more than 3-4 web searches to answer properly.
14
+ Also trigger during /plan Phase 2 research and /research workflows.
15
+ compatibility: Requires PARALLEL_API_KEY in .env.local. Uses curl. Takes 3-25 minutes.
16
+ metadata:
17
+ author: harshanandak
18
+ version: "1.0.0"
19
+ ---
20
+
21
+ # Parallel Deep Research
22
+
23
+ Comprehensive research reports with multi-source synthesis. Use `pro` (3-9 min, $0.10) or `ultra` (5-25 min, $0.30) for deep analysis.
24
+
25
+ > **CLI alternative (recommended)**: Install `parallel-cli` for official skill:
26
+ > `npx skills add parallel-web/parallel-agent-skills --skill parallel-deep-research`
27
+
28
+ ## Setup
29
+
30
+ ```bash
31
+ API_KEY=$(grep "^PARALLEL_API_KEY=" .env.local | cut -d= -f2)
32
+ ```
33
+
34
+ ## Create Research Task
35
+
36
+ ```bash
37
+ curl -s -X POST "https://api.parallel.ai/v1/tasks/runs" \
38
+ -H "x-api-key: $API_KEY" \
39
+ -H "Content-Type: application/json" \
40
+ -d '{
41
+ "input": "Analyze the AI chatbot market. Include: size, growth, key players, trends, competitive threats",
42
+ "processor": "pro"
43
+ }'
44
+ ```
45
+
46
+ Response: `{"run_id": "trun_abc123...", "status": "queued"}`
47
+
48
+ ## Get Result (polling)
49
+
50
+ The result endpoint returns both status and output in one call. Poll until `status` is `completed`.
51
+
52
+ ```bash
53
+ RUN_ID="trun_abc123..."
54
+ MAX_POLLS=180 # 30 min max (180 × 10s)
55
+
56
+ for i in $(seq 1 $MAX_POLLS); do
57
+ RESULT=$(curl -s "https://api.parallel.ai/v1/tasks/runs/$RUN_ID/result" \
58
+ -H "x-api-key: $API_KEY")
59
+
60
+ STATUS=$(echo "$RESULT" | grep -o '"status":"[^"]*"' | head -1 | cut -d'"' -f4)
61
+
62
+ if [ "$STATUS" = "completed" ]; then
63
+ echo "$RESULT"
64
+ break
65
+ elif [ "$STATUS" = "failed" ]; then
66
+ echo "Task failed: $RESULT"
67
+ break
68
+ fi
69
+
70
+ echo "Poll $i/$MAX_POLLS — Status: $STATUS"
71
+ sleep 10
72
+ done
73
+
74
+ if [ "$i" = "$MAX_POLLS" ] && [ "$STATUS" != "completed" ] && [ "$STATUS" != "failed" ]; then
75
+ echo "Timeout: task did not complete within 30 minutes"
76
+ fi
77
+ ```
78
+
79
+ ## Processors
80
+
81
+ | Processor | Speed | Cost | Use For |
82
+ |-----------|-------|------|---------|
83
+ | pro | 3-9 min | $0.10/task | Market analysis, strategic reports |
84
+ | ultra | 5-25 min | $0.30/task | Comprehensive deep research |
85
+
86
+ ## Example: Market Analysis
87
+
88
+ ```json
89
+ {
90
+ "input": "Analyze the AI chip market in 2024. Include market size, growth rate, key players (NVIDIA, AMD, Intel), emerging competitors, and 2025 outlook.",
91
+ "processor": "pro"
92
+ }
93
+ ```
94
+
95
+ Result: Markdown report with citations.
96
+
97
+ ## When to Use
98
+
99
+ - Market research and competitive analysis
100
+ - Strategic reports requiring multiple sources
101
+ - Research that needs synthesis across many documents
102
+ - Any task that would need more than 3-4 web searches to answer properly
103
+
104
+ For quick facts or single-source lookups, use built-in WebSearch instead.
105
+
106
+ ## Timeout
107
+
108
+ Set polling timeout to 1800s (30 min) for ultra tasks. Pro tasks typically complete in 3-9 min.
@@ -1,27 +1,27 @@
1
- # Eval Sets
2
-
3
- These eval files use a flat JSON array format targeting `scripts/eval_win.py`:
4
-
5
- ```json
6
- [{"query": "...", "should_trigger": true}]
7
- ```
8
-
9
- This is **not** compatible with the skill-creator plugin's `run_loop.py` which expects:
10
-
11
- ```json
12
- {"skill_name": "...", "evals": [{"id": "...", "prompt": "...", "should_trigger": true}]}
13
- ```
14
-
15
- To convert for `run_loop.py`:
16
-
17
- ```bash
18
- python3 -c "
19
- import json, pathlib, uuid
20
- data = json.loads(pathlib.Path('evals.json').read_text())
21
- out = {'skill_name': 'SKILL-NAME-HERE', 'evals': [
22
- {'id': str(uuid.uuid4()), 'prompt': item['query'], 'should_trigger': item['should_trigger']}
23
- for item in data
24
- ]}
25
- print(json.dumps(out, indent=2))
26
- " > evals_skill_creator.json
27
- ```
1
+ # Eval Sets
2
+
3
+ These eval files use a flat JSON array format targeting `scripts/eval_win.py`:
4
+
5
+ ```json
6
+ [{"query": "...", "should_trigger": true}]
7
+ ```
8
+
9
+ This is **not** compatible with the skill-creator plugin's `run_loop.py` which expects:
10
+
11
+ ```json
12
+ {"skill_name": "...", "evals": [{"id": "...", "prompt": "...", "should_trigger": true}]}
13
+ ```
14
+
15
+ To convert for `run_loop.py`:
16
+
17
+ ```bash
18
+ python3 -c "
19
+ import json, pathlib, uuid
20
+ data = json.loads(pathlib.Path('evals.json').read_text())
21
+ out = {'skill_name': 'SKILL-NAME-HERE', 'evals': [
22
+ {'id': str(uuid.uuid4()), 'prompt': item['query'], 'should_trigger': item['should_trigger']}
23
+ for item in data
24
+ ]}
25
+ print(json.dumps(out, indent=2))
26
+ " > evals_skill_creator.json
27
+ ```
@@ -1,62 +1,62 @@
1
- [
2
- {
3
- "query": "I need a comprehensive market analysis of the developer tools space — cover the top 20 players, their funding, market positioning, and where the industry is heading over the next 3-5 years. Include data from Gartner, Forrester, and any recent VC investment reports",
4
- "should_trigger": true
5
- },
6
- {
7
- "query": "Write a deep research report on the state of WebAssembly adoption in production — synthesize case studies from companies using WASM, performance benchmarks vs native code, ecosystem maturity analysis, and strategic recommendations for when to adopt it",
8
- "should_trigger": true
9
- },
10
- {
11
- "query": "Our CTO wants a competitive landscape analysis comparing Convex, Supabase, Firebase, and PlanetScale for our backend rewrite. Need comprehensive feature comparison, pricing analysis at our scale (50K DAU), community health metrics, and risk assessment for each",
12
- "should_trigger": true
13
- },
14
- {
15
- "query": "Create a strategic report on the AI code generation market — analyze GitHub Copilot, Cursor, Claude Code, and emerging competitors. Cover adoption rates, developer satisfaction surveys, enterprise pricing trends, and predictions for 2027",
16
- "should_trigger": true
17
- },
18
- {
19
- "query": "I'm doing /plan Phase 2 research — produce a comprehensive analysis of event-driven architecture patterns in microservices, covering Kafka vs RabbitMQ vs NATS, with production case studies, failure mode analysis, and recommendations for our 100K events/sec throughput requirement",
20
- "should_trigger": true
21
- },
22
- {
23
- "query": "Write an industry deep-dive into the observability market — compare Datadog, Grafana Cloud, and New Relic across features, pricing, scalability, and OpenTelemetry support. Include customer migration stories and TCO analysis for a 500-node cluster",
24
- "should_trigger": true
25
- },
26
- {
27
- "query": "Research and synthesize the current state of edge computing for real-time AI inference — cover hardware options, cloud provider edge offerings, latency benchmarks, and case studies from autonomous vehicles and IoT deployments",
28
- "should_trigger": true
29
- },
30
- {
31
- "query": "Quick search: what's the current price of Bitcoin and what were Anthropic's latest announcements?",
32
- "should_trigger": false
33
- },
34
- {
35
- "query": "Find the official migration guide URL for ESLint's new flat config format",
36
- "should_trigger": false
37
- },
38
- {
39
- "query": "Search for recent blog posts about Bun 2.0 features and release date",
40
- "should_trigger": false
41
- },
42
- {
43
- "query": "Go to https://stripe.com/docs/api and extract the authentication section with all the code examples",
44
- "should_trigger": false
45
- },
46
- {
47
- "query": "Scrape https://openai.com/pricing and extract the per-token costs for each model tier",
48
- "should_trigger": false
49
- },
50
- {
51
- "query": "Build a structured company profile for Databricks — return JSON with founding year, funding rounds, valuation, employee count, and tech stack",
52
- "should_trigger": false
53
- },
54
- {
55
- "query": "Fix the TypeScript compilation error in src/services/auth.ts — it's complaining about missing type for the session token",
56
- "should_trigger": false
57
- },
58
- {
59
- "query": "Add unit tests for the rate limiter middleware in src/middleware/rateLimit.ts",
60
- "should_trigger": false
61
- }
62
- ]
1
+ [
2
+ {
3
+ "query": "I need a comprehensive market analysis of the developer tools space — cover the top 20 players, their funding, market positioning, and where the industry is heading over the next 3-5 years. Include data from Gartner, Forrester, and any recent VC investment reports",
4
+ "should_trigger": true
5
+ },
6
+ {
7
+ "query": "Write a deep research report on the state of WebAssembly adoption in production — synthesize case studies from companies using WASM, performance benchmarks vs native code, ecosystem maturity analysis, and strategic recommendations for when to adopt it",
8
+ "should_trigger": true
9
+ },
10
+ {
11
+ "query": "Our CTO wants a competitive landscape analysis comparing Convex, Supabase, Firebase, and PlanetScale for our backend rewrite. Need comprehensive feature comparison, pricing analysis at our scale (50K DAU), community health metrics, and risk assessment for each",
12
+ "should_trigger": true
13
+ },
14
+ {
15
+ "query": "Create a strategic report on the AI code generation market — analyze GitHub Copilot, Cursor, Claude Code, and emerging competitors. Cover adoption rates, developer satisfaction surveys, enterprise pricing trends, and predictions for 2027",
16
+ "should_trigger": true
17
+ },
18
+ {
19
+ "query": "I'm doing /plan Phase 2 research — produce a comprehensive analysis of event-driven architecture patterns in microservices, covering Kafka vs RabbitMQ vs NATS, with production case studies, failure mode analysis, and recommendations for our 100K events/sec throughput requirement",
20
+ "should_trigger": true
21
+ },
22
+ {
23
+ "query": "Write an industry deep-dive into the observability market — compare Datadog, Grafana Cloud, and New Relic across features, pricing, scalability, and OpenTelemetry support. Include customer migration stories and TCO analysis for a 500-node cluster",
24
+ "should_trigger": true
25
+ },
26
+ {
27
+ "query": "Research and synthesize the current state of edge computing for real-time AI inference — cover hardware options, cloud provider edge offerings, latency benchmarks, and case studies from autonomous vehicles and IoT deployments",
28
+ "should_trigger": true
29
+ },
30
+ {
31
+ "query": "Quick search: what's the current price of Bitcoin and what were Anthropic's latest announcements?",
32
+ "should_trigger": false
33
+ },
34
+ {
35
+ "query": "Find the official migration guide URL for ESLint's new flat config format",
36
+ "should_trigger": false
37
+ },
38
+ {
39
+ "query": "Search for recent blog posts about Bun 2.0 features and release date",
40
+ "should_trigger": false
41
+ },
42
+ {
43
+ "query": "Go to https://stripe.com/docs/api and extract the authentication section with all the code examples",
44
+ "should_trigger": false
45
+ },
46
+ {
47
+ "query": "Scrape https://openai.com/pricing and extract the per-token costs for each model tier",
48
+ "should_trigger": false
49
+ },
50
+ {
51
+ "query": "Build a structured company profile for Databricks — return JSON with founding year, funding rounds, valuation, employee count, and tech stack",
52
+ "should_trigger": false
53
+ },
54
+ {
55
+ "query": "Fix the TypeScript compilation error in src/services/auth.ts — it's complaining about missing type for the session token",
56
+ "should_trigger": false
57
+ },
58
+ {
59
+ "query": "Add unit tests for the rate limiter middleware in src/middleware/rateLimit.ts",
60
+ "should_trigger": false
61
+ }
62
+ ]