forge-workflow 0.0.4 → 0.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/.claude/commands/dev.md +340 -340
  2. package/.claude/commands/plan.md +521 -521
  3. package/.claude/commands/premerge.md +176 -176
  4. package/.claude/commands/research.md +42 -42
  5. package/.claude/commands/review.md +442 -442
  6. package/.claude/commands/rollback.md +721 -721
  7. package/.claude/commands/ship.md +164 -164
  8. package/.claude/commands/sonarcloud.md +152 -152
  9. package/.claude/commands/status.md +48 -48
  10. package/.claude/commands/validate.md +282 -282
  11. package/.claude/commands/verify.md +221 -221
  12. package/.claude/rules/greptile-review-process.md +285 -285
  13. package/.claude/rules/workflow.md +105 -105
  14. package/.claude/scripts/greptile-resolve.sh +526 -526
  15. package/.claude/scripts/load-env.sh +32 -32
  16. package/.cline/workflows/dev.md +337 -337
  17. package/.cline/workflows/plan.md +518 -518
  18. package/.cline/workflows/premerge.md +173 -173
  19. package/.cline/workflows/research.md +39 -39
  20. package/.cline/workflows/review.md +439 -439
  21. package/.cline/workflows/rollback.md +718 -718
  22. package/.cline/workflows/ship.md +161 -161
  23. package/.cline/workflows/sonarcloud.md +146 -146
  24. package/.cline/workflows/status.md +45 -45
  25. package/.cline/workflows/validate.md +279 -279
  26. package/.cline/workflows/verify.md +218 -218
  27. package/.codex/config.toml +11 -11
  28. package/.codex/skills/dev/SKILL.md +340 -340
  29. package/.codex/skills/plan/SKILL.md +521 -521
  30. package/.codex/skills/premerge/SKILL.md +176 -176
  31. package/.codex/skills/research/SKILL.md +42 -42
  32. package/.codex/skills/review/SKILL.md +442 -442
  33. package/.codex/skills/rollback/SKILL.md +721 -721
  34. package/.codex/skills/ship/SKILL.md +164 -164
  35. package/.codex/skills/sonarcloud/SKILL.md +149 -149
  36. package/.codex/skills/status/SKILL.md +48 -48
  37. package/.codex/skills/validate/SKILL.md +282 -282
  38. package/.codex/skills/verify/SKILL.md +221 -221
  39. package/.cursor/commands/dev.md +337 -337
  40. package/.cursor/commands/plan.md +518 -518
  41. package/.cursor/commands/premerge.md +173 -173
  42. package/.cursor/commands/research.md +39 -39
  43. package/.cursor/commands/review.md +439 -439
  44. package/.cursor/commands/rollback.md +718 -718
  45. package/.cursor/commands/ship.md +161 -161
  46. package/.cursor/commands/sonarcloud.md +146 -146
  47. package/.cursor/commands/status.md +45 -45
  48. package/.cursor/commands/validate.md +279 -279
  49. package/.cursor/commands/verify.md +218 -218
  50. package/.cursor/rules/permissions-guidance.mdc +37 -37
  51. package/.forge/hooks/check-tdd.js +240 -240
  52. package/.github/PLUGIN_TEMPLATE.json +32 -32
  53. package/.github/prompts/dev.prompt.md +342 -342
  54. package/.github/prompts/plan.prompt.md +523 -523
  55. package/.github/prompts/premerge.prompt.md +178 -178
  56. package/.github/prompts/research.prompt.md +44 -44
  57. package/.github/prompts/review.prompt.md +444 -444
  58. package/.github/prompts/rollback.prompt.md +723 -723
  59. package/.github/prompts/ship.prompt.md +166 -166
  60. package/.github/prompts/sonarcloud.prompt.md +151 -151
  61. package/.github/prompts/status.prompt.md +50 -50
  62. package/.github/prompts/validate.prompt.md +284 -284
  63. package/.github/prompts/verify.prompt.md +223 -223
  64. package/.github/workflows/beads-to-github.yml +56 -0
  65. package/.github/workflows/github-to-beads.yml +97 -0
  66. package/.kilocode/workflows/dev.md +341 -341
  67. package/.kilocode/workflows/plan.md +522 -522
  68. package/.kilocode/workflows/premerge.md +177 -177
  69. package/.kilocode/workflows/research.md +43 -43
  70. package/.kilocode/workflows/review.md +443 -443
  71. package/.kilocode/workflows/rollback.md +722 -722
  72. package/.kilocode/workflows/ship.md +165 -165
  73. package/.kilocode/workflows/sonarcloud.md +150 -150
  74. package/.kilocode/workflows/status.md +49 -49
  75. package/.kilocode/workflows/validate.md +283 -283
  76. package/.kilocode/workflows/verify.md +222 -222
  77. package/.mcp.json.example +12 -12
  78. package/.opencode/commands/dev.md +340 -340
  79. package/.opencode/commands/plan.md +521 -521
  80. package/.opencode/commands/premerge.md +176 -176
  81. package/.opencode/commands/research.md +42 -42
  82. package/.opencode/commands/review.md +442 -442
  83. package/.opencode/commands/rollback.md +721 -721
  84. package/.opencode/commands/ship.md +164 -164
  85. package/.opencode/commands/sonarcloud.md +149 -149
  86. package/.opencode/commands/status.md +48 -48
  87. package/.opencode/commands/validate.md +282 -282
  88. package/.opencode/commands/verify.md +221 -221
  89. package/.roo/commands/dev.md +341 -341
  90. package/.roo/commands/plan.md +522 -522
  91. package/.roo/commands/premerge.md +177 -177
  92. package/.roo/commands/research.md +43 -43
  93. package/.roo/commands/review.md +443 -443
  94. package/.roo/commands/rollback.md +722 -722
  95. package/.roo/commands/ship.md +165 -165
  96. package/.roo/commands/sonarcloud.md +150 -150
  97. package/.roo/commands/status.md +49 -49
  98. package/.roo/commands/validate.md +283 -283
  99. package/.roo/commands/verify.md +222 -222
  100. package/AGENTS.md +175 -175
  101. package/CLAUDE.md +100 -100
  102. package/README.md +429 -416
  103. package/bin/forge-cmd.js +313 -313
  104. package/bin/forge-preflight.js +309 -309
  105. package/bin/forge.js +4596 -4303
  106. package/docs/AGENT_INSTALL_PROMPT.md +342 -342
  107. package/docs/BEADS_GITHUB_SYNC.md +251 -251
  108. package/docs/ENHANCED_ONBOARDING.md +602 -602
  109. package/docs/EXAMPLES.md +482 -482
  110. package/docs/GREPTILE_SETUP.md +400 -400
  111. package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
  112. package/docs/ROADMAP.md +359 -359
  113. package/docs/SETUP.md +663 -631
  114. package/docs/TOOLCHAIN.md +630 -630
  115. package/docs/VALIDATION.md +363 -363
  116. package/install.sh +40 -1056
  117. package/lefthook.yml +39 -39
  118. package/lib/agents/README.md +198 -198
  119. package/lib/agents/claude.plugin.json +28 -28
  120. package/lib/agents/cline.plugin.json +22 -22
  121. package/lib/agents/codex.plugin.json +19 -19
  122. package/lib/agents/copilot.plugin.json +24 -24
  123. package/lib/agents/cursor.plugin.json +25 -25
  124. package/lib/agents/kilocode.plugin.json +22 -22
  125. package/lib/agents/opencode.plugin.json +20 -20
  126. package/lib/agents/roo.plugin.json +23 -23
  127. package/lib/agents-config.js +2112 -2112
  128. package/lib/beads-health-check.js +143 -0
  129. package/lib/beads-setup.js +341 -0
  130. package/lib/beads-sync-scaffold.js +260 -0
  131. package/lib/commands/dev.js +513 -513
  132. package/lib/commands/plan.js +692 -692
  133. package/lib/commands/recommend.js +119 -119
  134. package/lib/commands/ship.js +377 -377
  135. package/lib/commands/status.js +378 -378
  136. package/lib/commands/validate.js +602 -602
  137. package/lib/context-merge.js +359 -359
  138. package/lib/dep-guard/analyzer.js +294 -294
  139. package/lib/dep-guard/behavior-detector.js +98 -98
  140. package/lib/dep-guard/contract-detector.js +162 -162
  141. package/lib/dep-guard/import-detector.js +498 -498
  142. package/lib/dep-guard/path-utils.js +13 -13
  143. package/lib/dep-guard/rubric.js +120 -120
  144. package/lib/dep-guard/task-parser.js +318 -318
  145. package/lib/detect-agent.js +191 -191
  146. package/lib/detect-worktree.js +47 -47
  147. package/lib/file-hash.js +26 -26
  148. package/lib/husky-migration.js +450 -0
  149. package/lib/lefthook-check.js +65 -0
  150. package/lib/pat-setup.js +207 -0
  151. package/lib/plugin-catalog.js +350 -350
  152. package/lib/plugin-manager.js +166 -166
  153. package/lib/plugin-recommender.js +141 -141
  154. package/lib/project-discovery.js +491 -491
  155. package/lib/setup-action-log.js +139 -139
  156. package/lib/setup-summary-renderer.js +106 -106
  157. package/lib/setup-utils.js +96 -0
  158. package/lib/setup.js +192 -192
  159. package/lib/smart-merge.js +64 -0
  160. package/lib/symlink-utils.js +81 -0
  161. package/lib/workflow-profiles.js +197 -197
  162. package/package.json +131 -128
  163. package/scripts/beads-context.sh +291 -0
  164. package/scripts/beads-context.test.js +563 -0
  165. package/scripts/behavioral-judge.sh +378 -0
  166. package/scripts/benchmark.js +85 -0
  167. package/scripts/branch-protection.js +183 -0
  168. package/scripts/check-agents.js +172 -0
  169. package/scripts/commitlint.js +42 -0
  170. package/scripts/conflict-detect.sh +323 -0
  171. package/scripts/dep-guard-analyze.js +71 -0
  172. package/scripts/dep-guard.sh +811 -0
  173. package/scripts/eval_win.py +249 -0
  174. package/scripts/file-index.sh +399 -0
  175. package/scripts/github-beads-sync/comment.mjs +64 -0
  176. package/scripts/github-beads-sync/config.mjs +148 -0
  177. package/scripts/github-beads-sync/github-api.mjs +131 -0
  178. package/scripts/github-beads-sync/index.mjs +332 -0
  179. package/scripts/github-beads-sync/label-mapper.mjs +54 -0
  180. package/scripts/github-beads-sync/mapping.mjs +78 -0
  181. package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
  182. package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
  183. package/scripts/github-beads-sync/run-bd.mjs +159 -0
  184. package/scripts/github-beads-sync/sanitize.mjs +121 -0
  185. package/scripts/github-beads-sync.config.json +26 -0
  186. package/scripts/improve-command.js +375 -0
  187. package/scripts/lib/eval-runner.js +229 -0
  188. package/scripts/lib/eval-schema.js +135 -0
  189. package/scripts/lib/eval-storage.js +78 -0
  190. package/scripts/lib/grading.js +203 -0
  191. package/scripts/lib/transcript-parser.js +63 -0
  192. package/scripts/lint.js +47 -0
  193. package/scripts/migrate-to-bun-test.js +412 -0
  194. package/scripts/run-command-eval.js +236 -0
  195. package/scripts/smart-status.sh +782 -0
  196. package/scripts/sync-commands.js +571 -0
  197. package/scripts/sync-utils.sh +460 -0
  198. package/scripts/test-dashboard.js +123 -0
  199. package/scripts/test.js +44 -0
  200. package/scripts/validate.sh +94 -0
  201. package/skills/parallel-deep-research/SKILL.md +108 -108
  202. package/skills/parallel-deep-research/evals/README.md +27 -27
  203. package/skills/parallel-deep-research/evals/evals.json +62 -62
  204. package/skills/sonarcloud-analysis/SKILL.md +171 -171
  205. package/skills/sonarcloud-analysis/evals/README.md +27 -27
  206. package/skills/sonarcloud-analysis/evals/evals.json +50 -50
  207. package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
  208. package/.cursor/hooks/state/continual-learning-index.json +0 -19
  209. package/.cursor/hooks/state/continual-learning.json +0 -8
@@ -0,0 +1,94 @@
1
+ #!/usr/bin/env bash
2
+ # Unified validation script for Forge project
3
+ # Runs all quality checks in sequence
4
+ # Exits on first failure
5
+
6
+ set -e # Exit on first error
7
+ set -o pipefail # Catch errors in pipes
8
+
9
+ # Colors for output
10
+ if [ -t 1 ]; then
11
+ RED='\033[0;31m'
12
+ GREEN='\033[0;32m'
13
+ YELLOW='\033[1;33m'
14
+ BLUE='\033[0;34m'
15
+ NC='\033[0m' # No Color
16
+ else
17
+ RED=''
18
+ GREEN=''
19
+ YELLOW=''
20
+ BLUE=''
21
+ NC=''
22
+ fi
23
+
24
+ # Print section header
25
+ print_header() {
26
+ echo ""
27
+ echo -e "${BLUE}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
28
+ echo -e "${BLUE}▶ $1${NC}"
29
+ echo -e "${BLUE}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
30
+ }
31
+
32
+ # Print success message
33
+ print_success() {
34
+ echo -e "${GREEN}✓ $1${NC}"
35
+ }
36
+
37
+ # Print error message
38
+ print_error() {
39
+ echo -e "${RED}✗ $1${NC}"
40
+ }
41
+
42
+ # Print warning message
43
+ print_warning() {
44
+ echo -e "${YELLOW}⚠ $1${NC}"
45
+ }
46
+
47
+ echo ""
48
+ echo -e "${BLUE}╔═══════════════════════════════════════════╗${NC}"
49
+ echo -e "${BLUE}║ Forge Quality Gate - Running Checks ║${NC}"
50
+ echo -e "${BLUE}╚═══════════════════════════════════════════╝${NC}"
51
+
52
+ # Step 1: Type Check
53
+ print_header "1/4: Type Check"
54
+ print_warning "SKIPPED (no TypeScript in project)"
55
+
56
+ # Step 2: Lint
57
+ print_header "2/4: Lint"
58
+ if bun run lint; then
59
+ print_success "Lint passed"
60
+ else
61
+ print_error "Lint failed"
62
+ exit 1
63
+ fi
64
+
65
+ # Step 3: Security Audit
66
+ print_header "3/4: Security Audit"
67
+ AUDIT_OUTPUT=$(bun audit 2>&1 || true)
68
+ if echo "$AUDIT_OUTPUT" | grep -qiE 'critical|high'; then
69
+ print_error "Security audit found critical/high vulnerabilities"
70
+ echo "$AUDIT_OUTPUT"
71
+ exit 1
72
+ fi
73
+ if bun audit; then
74
+ print_success "Security audit passed"
75
+ else
76
+ print_warning "Security audit found issues (moderate/low — non-blocking)"
77
+ fi
78
+
79
+ # Step 4: Tests
80
+ print_header "4/4: Tests"
81
+ if bun test; then
82
+ print_success "All tests passed"
83
+ else
84
+ print_error "Tests failed"
85
+ exit 1
86
+ fi
87
+
88
+ echo ""
89
+ echo -e "${GREEN}╔═══════════════════════════════════════════╗${NC}"
90
+ echo -e "${GREEN}║ ✓ All Checks Passed Successfully ║${NC}"
91
+ echo -e "${GREEN}╚═══════════════════════════════════════════╝${NC}"
92
+ echo ""
93
+
94
+ exit 0
@@ -1,108 +1,108 @@
1
- ---
2
- name: parallel-deep-research
3
- description: >
4
- Produces comprehensive research reports that go far beyond what built-in web
5
- search can achieve. Sends research tasks to Parallel AI's pro/ultra processors
6
- which spend 3-25 minutes autonomously crawling, reading, and synthesizing dozens
7
- of sources — returning structured reports with citations. Built-in WebSearch
8
- can only run a few queries; this skill runs an entire research pipeline externally.
9
- No binary install — requires PARALLEL_API_KEY in .env.local. ALWAYS use this
10
- skill instead of doing multiple WebSearch calls when the user needs a comprehensive
11
- report, market analysis, competitive landscape, industry deep-dive, strategic
12
- recommendations, or multi-source synthesis. This is the RIGHT tool for any
13
- research task that would require more than 3-4 web searches to answer properly.
14
- Also trigger during /plan Phase 2 research and /research workflows.
15
- compatibility: Requires PARALLEL_API_KEY in .env.local. Uses curl. Takes 3-25 minutes.
16
- metadata:
17
- author: harshanandak
18
- version: "1.0.0"
19
- ---
20
-
21
- # Parallel Deep Research
22
-
23
- Comprehensive research reports with multi-source synthesis. Use `pro` (3-9 min, $0.10) or `ultra` (5-25 min, $0.30) for deep analysis.
24
-
25
- > **CLI alternative (recommended)**: Install `parallel-cli` for official skill:
26
- > `npx skills add parallel-web/parallel-agent-skills --skill parallel-deep-research`
27
-
28
- ## Setup
29
-
30
- ```bash
31
- API_KEY=$(grep "^PARALLEL_API_KEY=" .env.local | cut -d= -f2)
32
- ```
33
-
34
- ## Create Research Task
35
-
36
- ```bash
37
- curl -s -X POST "https://api.parallel.ai/v1/tasks/runs" \
38
- -H "x-api-key: $API_KEY" \
39
- -H "Content-Type: application/json" \
40
- -d '{
41
- "input": "Analyze the AI chatbot market. Include: size, growth, key players, trends, competitive threats",
42
- "processor": "pro"
43
- }'
44
- ```
45
-
46
- Response: `{"run_id": "trun_abc123...", "status": "queued"}`
47
-
48
- ## Get Result (polling)
49
-
50
- The result endpoint returns both status and output in one call. Poll until `status` is `completed`.
51
-
52
- ```bash
53
- RUN_ID="trun_abc123..."
54
- MAX_POLLS=180 # 30 min max (180 × 10s)
55
-
56
- for i in $(seq 1 $MAX_POLLS); do
57
- RESULT=$(curl -s "https://api.parallel.ai/v1/tasks/runs/$RUN_ID/result" \
58
- -H "x-api-key: $API_KEY")
59
-
60
- STATUS=$(echo "$RESULT" | grep -o '"status":"[^"]*"' | head -1 | cut -d'"' -f4)
61
-
62
- if [ "$STATUS" = "completed" ]; then
63
- echo "$RESULT"
64
- break
65
- elif [ "$STATUS" = "failed" ]; then
66
- echo "Task failed: $RESULT"
67
- break
68
- fi
69
-
70
- echo "Poll $i/$MAX_POLLS — Status: $STATUS"
71
- sleep 10
72
- done
73
-
74
- if [ "$i" = "$MAX_POLLS" ] && [ "$STATUS" != "completed" ] && [ "$STATUS" != "failed" ]; then
75
- echo "Timeout: task did not complete within 30 minutes"
76
- fi
77
- ```
78
-
79
- ## Processors
80
-
81
- | Processor | Speed | Cost | Use For |
82
- |-----------|-------|------|---------|
83
- | pro | 3-9 min | $0.10/task | Market analysis, strategic reports |
84
- | ultra | 5-25 min | $0.30/task | Comprehensive deep research |
85
-
86
- ## Example: Market Analysis
87
-
88
- ```json
89
- {
90
- "input": "Analyze the AI chip market in 2024. Include market size, growth rate, key players (NVIDIA, AMD, Intel), emerging competitors, and 2025 outlook.",
91
- "processor": "pro"
92
- }
93
- ```
94
-
95
- Result: Markdown report with citations.
96
-
97
- ## When to Use
98
-
99
- - Market research and competitive analysis
100
- - Strategic reports requiring multiple sources
101
- - Research that needs synthesis across many documents
102
- - Any task that would need more than 3-4 web searches to answer properly
103
-
104
- For quick facts or single-source lookups, use built-in WebSearch instead.
105
-
106
- ## Timeout
107
-
108
- Set polling timeout to 1800s (30 min) for ultra tasks. Pro tasks typically complete in 3-9 min.
1
+ ---
2
+ name: parallel-deep-research
3
+ description: >
4
+ Produces comprehensive research reports that go far beyond what built-in web
5
+ search can achieve. Sends research tasks to Parallel AI's pro/ultra processors
6
+ which spend 3-25 minutes autonomously crawling, reading, and synthesizing dozens
7
+ of sources — returning structured reports with citations. Built-in WebSearch
8
+ can only run a few queries; this skill runs an entire research pipeline externally.
9
+ No binary install — requires PARALLEL_API_KEY in .env.local. ALWAYS use this
10
+ skill instead of doing multiple WebSearch calls when the user needs a comprehensive
11
+ report, market analysis, competitive landscape, industry deep-dive, strategic
12
+ recommendations, or multi-source synthesis. This is the RIGHT tool for any
13
+ research task that would require more than 3-4 web searches to answer properly.
14
+ Also trigger during /plan Phase 2 research and /research workflows.
15
+ compatibility: Requires PARALLEL_API_KEY in .env.local. Uses curl. Takes 3-25 minutes.
16
+ metadata:
17
+ author: harshanandak
18
+ version: "1.0.0"
19
+ ---
20
+
21
+ # Parallel Deep Research
22
+
23
+ Comprehensive research reports with multi-source synthesis. Use `pro` (3-9 min, $0.10) or `ultra` (5-25 min, $0.30) for deep analysis.
24
+
25
+ > **CLI alternative (recommended)**: Install `parallel-cli` for official skill:
26
+ > `npx skills add parallel-web/parallel-agent-skills --skill parallel-deep-research`
27
+
28
+ ## Setup
29
+
30
+ ```bash
31
+ API_KEY=$(grep "^PARALLEL_API_KEY=" .env.local | cut -d= -f2)
32
+ ```
33
+
34
+ ## Create Research Task
35
+
36
+ ```bash
37
+ curl -s -X POST "https://api.parallel.ai/v1/tasks/runs" \
38
+ -H "x-api-key: $API_KEY" \
39
+ -H "Content-Type: application/json" \
40
+ -d '{
41
+ "input": "Analyze the AI chatbot market. Include: size, growth, key players, trends, competitive threats",
42
+ "processor": "pro"
43
+ }'
44
+ ```
45
+
46
+ Response: `{"run_id": "trun_abc123...", "status": "queued"}`
47
+
48
+ ## Get Result (polling)
49
+
50
+ The result endpoint returns both status and output in one call. Poll until `status` is `completed`.
51
+
52
+ ```bash
53
+ RUN_ID="trun_abc123..."
54
+ MAX_POLLS=180 # 30 min max (180 × 10s)
55
+
56
+ for i in $(seq 1 $MAX_POLLS); do
57
+ RESULT=$(curl -s "https://api.parallel.ai/v1/tasks/runs/$RUN_ID/result" \
58
+ -H "x-api-key: $API_KEY")
59
+
60
+ STATUS=$(echo "$RESULT" | grep -o '"status":"[^"]*"' | head -1 | cut -d'"' -f4)
61
+
62
+ if [ "$STATUS" = "completed" ]; then
63
+ echo "$RESULT"
64
+ break
65
+ elif [ "$STATUS" = "failed" ]; then
66
+ echo "Task failed: $RESULT"
67
+ break
68
+ fi
69
+
70
+ echo "Poll $i/$MAX_POLLS — Status: $STATUS"
71
+ sleep 10
72
+ done
73
+
74
+ if [ "$i" = "$MAX_POLLS" ] && [ "$STATUS" != "completed" ] && [ "$STATUS" != "failed" ]; then
75
+ echo "Timeout: task did not complete within 30 minutes"
76
+ fi
77
+ ```
78
+
79
+ ## Processors
80
+
81
+ | Processor | Speed | Cost | Use For |
82
+ |-----------|-------|------|---------|
83
+ | pro | 3-9 min | $0.10/task | Market analysis, strategic reports |
84
+ | ultra | 5-25 min | $0.30/task | Comprehensive deep research |
85
+
86
+ ## Example: Market Analysis
87
+
88
+ ```json
89
+ {
90
+ "input": "Analyze the AI chip market in 2024. Include market size, growth rate, key players (NVIDIA, AMD, Intel), emerging competitors, and 2025 outlook.",
91
+ "processor": "pro"
92
+ }
93
+ ```
94
+
95
+ Result: Markdown report with citations.
96
+
97
+ ## When to Use
98
+
99
+ - Market research and competitive analysis
100
+ - Strategic reports requiring multiple sources
101
+ - Research that needs synthesis across many documents
102
+ - Any task that would need more than 3-4 web searches to answer properly
103
+
104
+ For quick facts or single-source lookups, use built-in WebSearch instead.
105
+
106
+ ## Timeout
107
+
108
+ Set polling timeout to 1800s (30 min) for ultra tasks. Pro tasks typically complete in 3-9 min.
@@ -1,27 +1,27 @@
1
- # Eval Sets
2
-
3
- These eval files use a flat JSON array format targeting `scripts/eval_win.py`:
4
-
5
- ```json
6
- [{"query": "...", "should_trigger": true}]
7
- ```
8
-
9
- This is **not** compatible with the skill-creator plugin's `run_loop.py` which expects:
10
-
11
- ```json
12
- {"skill_name": "...", "evals": [{"id": "...", "prompt": "...", "should_trigger": true}]}
13
- ```
14
-
15
- To convert for `run_loop.py`:
16
-
17
- ```bash
18
- python3 -c "
19
- import json, pathlib, uuid
20
- data = json.loads(pathlib.Path('evals.json').read_text())
21
- out = {'skill_name': 'SKILL-NAME-HERE', 'evals': [
22
- {'id': str(uuid.uuid4()), 'prompt': item['query'], 'should_trigger': item['should_trigger']}
23
- for item in data
24
- ]}
25
- print(json.dumps(out, indent=2))
26
- " > evals_skill_creator.json
27
- ```
1
+ # Eval Sets
2
+
3
+ These eval files use a flat JSON array format targeting `scripts/eval_win.py`:
4
+
5
+ ```json
6
+ [{"query": "...", "should_trigger": true}]
7
+ ```
8
+
9
+ This is **not** compatible with the skill-creator plugin's `run_loop.py` which expects:
10
+
11
+ ```json
12
+ {"skill_name": "...", "evals": [{"id": "...", "prompt": "...", "should_trigger": true}]}
13
+ ```
14
+
15
+ To convert for `run_loop.py`:
16
+
17
+ ```bash
18
+ python3 -c "
19
+ import json, pathlib, uuid
20
+ data = json.loads(pathlib.Path('evals.json').read_text())
21
+ out = {'skill_name': 'SKILL-NAME-HERE', 'evals': [
22
+ {'id': str(uuid.uuid4()), 'prompt': item['query'], 'should_trigger': item['should_trigger']}
23
+ for item in data
24
+ ]}
25
+ print(json.dumps(out, indent=2))
26
+ " > evals_skill_creator.json
27
+ ```
@@ -1,62 +1,62 @@
1
- [
2
- {
3
- "query": "I need a comprehensive market analysis of the developer tools space — cover the top 20 players, their funding, market positioning, and where the industry is heading over the next 3-5 years. Include data from Gartner, Forrester, and any recent VC investment reports",
4
- "should_trigger": true
5
- },
6
- {
7
- "query": "Write a deep research report on the state of WebAssembly adoption in production — synthesize case studies from companies using WASM, performance benchmarks vs native code, ecosystem maturity analysis, and strategic recommendations for when to adopt it",
8
- "should_trigger": true
9
- },
10
- {
11
- "query": "Our CTO wants a competitive landscape analysis comparing Convex, Supabase, Firebase, and PlanetScale for our backend rewrite. Need comprehensive feature comparison, pricing analysis at our scale (50K DAU), community health metrics, and risk assessment for each",
12
- "should_trigger": true
13
- },
14
- {
15
- "query": "Create a strategic report on the AI code generation market — analyze GitHub Copilot, Cursor, Claude Code, and emerging competitors. Cover adoption rates, developer satisfaction surveys, enterprise pricing trends, and predictions for 2027",
16
- "should_trigger": true
17
- },
18
- {
19
- "query": "I'm doing /plan Phase 2 research — produce a comprehensive analysis of event-driven architecture patterns in microservices, covering Kafka vs RabbitMQ vs NATS, with production case studies, failure mode analysis, and recommendations for our 100K events/sec throughput requirement",
20
- "should_trigger": true
21
- },
22
- {
23
- "query": "Write an industry deep-dive into the observability market — compare Datadog, Grafana Cloud, and New Relic across features, pricing, scalability, and OpenTelemetry support. Include customer migration stories and TCO analysis for a 500-node cluster",
24
- "should_trigger": true
25
- },
26
- {
27
- "query": "Research and synthesize the current state of edge computing for real-time AI inference — cover hardware options, cloud provider edge offerings, latency benchmarks, and case studies from autonomous vehicles and IoT deployments",
28
- "should_trigger": true
29
- },
30
- {
31
- "query": "Quick search: what's the current price of Bitcoin and what were Anthropic's latest announcements?",
32
- "should_trigger": false
33
- },
34
- {
35
- "query": "Find the official migration guide URL for ESLint's new flat config format",
36
- "should_trigger": false
37
- },
38
- {
39
- "query": "Search for recent blog posts about Bun 2.0 features and release date",
40
- "should_trigger": false
41
- },
42
- {
43
- "query": "Go to https://stripe.com/docs/api and extract the authentication section with all the code examples",
44
- "should_trigger": false
45
- },
46
- {
47
- "query": "Scrape https://openai.com/pricing and extract the per-token costs for each model tier",
48
- "should_trigger": false
49
- },
50
- {
51
- "query": "Build a structured company profile for Databricks — return JSON with founding year, funding rounds, valuation, employee count, and tech stack",
52
- "should_trigger": false
53
- },
54
- {
55
- "query": "Fix the TypeScript compilation error in src/services/auth.ts — it's complaining about missing type for the session token",
56
- "should_trigger": false
57
- },
58
- {
59
- "query": "Add unit tests for the rate limiter middleware in src/middleware/rateLimit.ts",
60
- "should_trigger": false
61
- }
62
- ]
1
+ [
2
+ {
3
+ "query": "I need a comprehensive market analysis of the developer tools space — cover the top 20 players, their funding, market positioning, and where the industry is heading over the next 3-5 years. Include data from Gartner, Forrester, and any recent VC investment reports",
4
+ "should_trigger": true
5
+ },
6
+ {
7
+ "query": "Write a deep research report on the state of WebAssembly adoption in production — synthesize case studies from companies using WASM, performance benchmarks vs native code, ecosystem maturity analysis, and strategic recommendations for when to adopt it",
8
+ "should_trigger": true
9
+ },
10
+ {
11
+ "query": "Our CTO wants a competitive landscape analysis comparing Convex, Supabase, Firebase, and PlanetScale for our backend rewrite. Need comprehensive feature comparison, pricing analysis at our scale (50K DAU), community health metrics, and risk assessment for each",
12
+ "should_trigger": true
13
+ },
14
+ {
15
+ "query": "Create a strategic report on the AI code generation market — analyze GitHub Copilot, Cursor, Claude Code, and emerging competitors. Cover adoption rates, developer satisfaction surveys, enterprise pricing trends, and predictions for 2027",
16
+ "should_trigger": true
17
+ },
18
+ {
19
+ "query": "I'm doing /plan Phase 2 research — produce a comprehensive analysis of event-driven architecture patterns in microservices, covering Kafka vs RabbitMQ vs NATS, with production case studies, failure mode analysis, and recommendations for our 100K events/sec throughput requirement",
20
+ "should_trigger": true
21
+ },
22
+ {
23
+ "query": "Write an industry deep-dive into the observability market — compare Datadog, Grafana Cloud, and New Relic across features, pricing, scalability, and OpenTelemetry support. Include customer migration stories and TCO analysis for a 500-node cluster",
24
+ "should_trigger": true
25
+ },
26
+ {
27
+ "query": "Research and synthesize the current state of edge computing for real-time AI inference — cover hardware options, cloud provider edge offerings, latency benchmarks, and case studies from autonomous vehicles and IoT deployments",
28
+ "should_trigger": true
29
+ },
30
+ {
31
+ "query": "Quick search: what's the current price of Bitcoin and what were Anthropic's latest announcements?",
32
+ "should_trigger": false
33
+ },
34
+ {
35
+ "query": "Find the official migration guide URL for ESLint's new flat config format",
36
+ "should_trigger": false
37
+ },
38
+ {
39
+ "query": "Search for recent blog posts about Bun 2.0 features and release date",
40
+ "should_trigger": false
41
+ },
42
+ {
43
+ "query": "Go to https://stripe.com/docs/api and extract the authentication section with all the code examples",
44
+ "should_trigger": false
45
+ },
46
+ {
47
+ "query": "Scrape https://openai.com/pricing and extract the per-token costs for each model tier",
48
+ "should_trigger": false
49
+ },
50
+ {
51
+ "query": "Build a structured company profile for Databricks — return JSON with founding year, funding rounds, valuation, employee count, and tech stack",
52
+ "should_trigger": false
53
+ },
54
+ {
55
+ "query": "Fix the TypeScript compilation error in src/services/auth.ts — it's complaining about missing type for the session token",
56
+ "should_trigger": false
57
+ },
58
+ {
59
+ "query": "Add unit tests for the rate limiter middleware in src/middleware/rateLimit.ts",
60
+ "should_trigger": false
61
+ }
62
+ ]