@rune-kit/rune 2.10.0 → 2.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +65 -6
  3. package/commands/rune.md +168 -168
  4. package/compiler/__tests__/detect-invariants.test.js +136 -0
  5. package/compiler/__tests__/doctor-mesh.test.js +229 -0
  6. package/compiler/__tests__/hook-dispatch.test.js +91 -0
  7. package/compiler/__tests__/hooks-antigravity.test.js +118 -0
  8. package/compiler/__tests__/hooks-cursor.test.js +139 -0
  9. package/compiler/__tests__/hooks-install.test.js +305 -0
  10. package/compiler/__tests__/hooks-merge.test.js +204 -0
  11. package/compiler/__tests__/hooks-tiers.test.js +519 -0
  12. package/compiler/__tests__/hooks-windsurf.test.js +115 -0
  13. package/compiler/__tests__/inject-claude-md.test.js +152 -0
  14. package/compiler/__tests__/load-invariants.test.js +408 -0
  15. package/compiler/__tests__/onboard-invariants.test.js +240 -0
  16. package/compiler/adapters/hooks/antigravity.js +140 -0
  17. package/compiler/adapters/hooks/claude.js +166 -0
  18. package/compiler/adapters/hooks/cursor.js +191 -0
  19. package/compiler/adapters/hooks/index.js +82 -0
  20. package/compiler/adapters/hooks/tier-emitter.js +182 -0
  21. package/compiler/adapters/hooks/windsurf.js +202 -0
  22. package/compiler/bin/rune.js +196 -6
  23. package/compiler/commands/hook-dispatch.js +87 -0
  24. package/compiler/commands/hooks/install.js +120 -0
  25. package/compiler/commands/hooks/merge.js +211 -0
  26. package/compiler/commands/hooks/presets.js +116 -0
  27. package/compiler/commands/hooks/status.js +112 -0
  28. package/compiler/commands/hooks/tiers.js +221 -0
  29. package/compiler/commands/hooks/uninstall.js +94 -0
  30. package/compiler/doctor.js +236 -0
  31. package/contexts/dev.md +34 -34
  32. package/contexts/research.md +43 -43
  33. package/contexts/review.md +55 -55
  34. package/extensions/ai-ml/PACK.md +88 -88
  35. package/extensions/ai-ml/skills/ai-agents.md +172 -172
  36. package/extensions/ai-ml/skills/code-sandbox.md +187 -187
  37. package/extensions/ai-ml/skills/deep-research.md +146 -146
  38. package/extensions/ai-ml/skills/embedding-search.md +66 -66
  39. package/extensions/ai-ml/skills/fine-tuning-guide.md +74 -74
  40. package/extensions/ai-ml/skills/llm-architect.md +125 -125
  41. package/extensions/ai-ml/skills/llm-integration.md +64 -64
  42. package/extensions/ai-ml/skills/prompt-patterns.md +72 -72
  43. package/extensions/ai-ml/skills/rag-patterns.md +66 -66
  44. package/extensions/ai-ml/skills/web-extraction.md +114 -114
  45. package/extensions/analytics/PACK.md +92 -92
  46. package/extensions/analytics/skills/ab-testing.md +72 -72
  47. package/extensions/analytics/skills/dashboard-patterns.md +83 -83
  48. package/extensions/analytics/skills/data-validation.md +68 -68
  49. package/extensions/analytics/skills/funnel-analysis.md +81 -81
  50. package/extensions/analytics/skills/sql-patterns.md +57 -57
  51. package/extensions/analytics/skills/statistical-analysis.md +79 -79
  52. package/extensions/analytics/skills/tracking-setup.md +71 -71
  53. package/extensions/backend/PACK.md +104 -104
  54. package/extensions/backend/skills/api-patterns.md +84 -84
  55. package/extensions/backend/skills/async-pipeline.md +193 -193
  56. package/extensions/backend/skills/auth-patterns.md +97 -97
  57. package/extensions/backend/skills/background-jobs.md +133 -133
  58. package/extensions/backend/skills/caching-patterns.md +108 -108
  59. package/extensions/backend/skills/cli-generation.md +133 -133
  60. package/extensions/backend/skills/database-patterns.md +87 -87
  61. package/extensions/backend/skills/middleware-patterns.md +104 -104
  62. package/extensions/chrome-ext/PACK.md +93 -93
  63. package/extensions/chrome-ext/skills/cws-preflight.md +143 -143
  64. package/extensions/chrome-ext/skills/cws-publish.md +104 -104
  65. package/extensions/chrome-ext/skills/ext-ai-integration.md +251 -251
  66. package/extensions/chrome-ext/skills/ext-messaging.md +139 -139
  67. package/extensions/chrome-ext/skills/ext-storage.md +133 -133
  68. package/extensions/chrome-ext/skills/mv3-scaffold.md +164 -164
  69. package/extensions/content/PACK.md +96 -96
  70. package/extensions/content/skills/blog-patterns.md +88 -88
  71. package/extensions/content/skills/cms-integration.md +131 -131
  72. package/extensions/content/skills/content-scoring.md +107 -107
  73. package/extensions/content/skills/i18n.md +83 -83
  74. package/extensions/content/skills/mdx-authoring.md +137 -137
  75. package/extensions/content/skills/reference.md +1014 -1014
  76. package/extensions/content/skills/seo-patterns.md +67 -67
  77. package/extensions/content/skills/video-repurpose.md +153 -153
  78. package/extensions/devops/PACK.md +101 -101
  79. package/extensions/devops/skills/chaos-testing.md +67 -67
  80. package/extensions/devops/skills/ci-cd.md +75 -75
  81. package/extensions/devops/skills/docker.md +58 -58
  82. package/extensions/devops/skills/edge-serverless.md +163 -163
  83. package/extensions/devops/skills/infra-as-code.md +158 -158
  84. package/extensions/devops/skills/kubernetes.md +110 -110
  85. package/extensions/devops/skills/monitoring.md +57 -57
  86. package/extensions/devops/skills/server-setup.md +64 -64
  87. package/extensions/devops/skills/ssl-domain.md +42 -42
  88. package/extensions/ecommerce/PACK.md +116 -116
  89. package/extensions/ecommerce/skills/cart-system.md +79 -79
  90. package/extensions/ecommerce/skills/inventory-mgmt.md +102 -102
  91. package/extensions/ecommerce/skills/order-management.md +126 -126
  92. package/extensions/ecommerce/skills/payment-integration.md +472 -472
  93. package/extensions/ecommerce/skills/shopify-dev.md +69 -69
  94. package/extensions/ecommerce/skills/subscription-billing.md +93 -93
  95. package/extensions/ecommerce/skills/tax-compliance.md +117 -117
  96. package/extensions/gamedev/PACK.md +142 -142
  97. package/extensions/gamedev/skills/asset-pipeline.md +74 -74
  98. package/extensions/gamedev/skills/audio-system.md +129 -129
  99. package/extensions/gamedev/skills/camera-system.md +87 -87
  100. package/extensions/gamedev/skills/ecs.md +98 -98
  101. package/extensions/gamedev/skills/game-loops.md +72 -72
  102. package/extensions/gamedev/skills/input-system.md +199 -199
  103. package/extensions/gamedev/skills/multiplayer.md +180 -180
  104. package/extensions/gamedev/skills/particles.md +105 -105
  105. package/extensions/gamedev/skills/physics-engine.md +89 -89
  106. package/extensions/gamedev/skills/scene-management.md +146 -146
  107. package/extensions/gamedev/skills/threejs-patterns.md +90 -90
  108. package/extensions/gamedev/skills/webgl.md +71 -71
  109. package/extensions/mobile/PACK.md +106 -106
  110. package/extensions/mobile/skills/app-store-connect.md +152 -152
  111. package/extensions/mobile/skills/app-store-prep.md +66 -66
  112. package/extensions/mobile/skills/deep-linking.md +109 -109
  113. package/extensions/mobile/skills/flutter.md +60 -60
  114. package/extensions/mobile/skills/ios-build-pipeline.md +142 -142
  115. package/extensions/mobile/skills/native-bridge.md +66 -66
  116. package/extensions/mobile/skills/ota-updates.md +97 -97
  117. package/extensions/mobile/skills/push-notifications.md +111 -111
  118. package/extensions/mobile/skills/react-native.md +82 -82
  119. package/extensions/saas/PACK.md +116 -116
  120. package/extensions/saas/skills/billing-integration.md +200 -200
  121. package/extensions/saas/skills/feature-flags.md +130 -130
  122. package/extensions/saas/skills/multi-tenant.md +103 -103
  123. package/extensions/saas/skills/onboarding-flow.md +139 -139
  124. package/extensions/saas/skills/subscription-flow.md +95 -95
  125. package/extensions/saas/skills/team-management.md +144 -144
  126. package/extensions/security/PACK.md +99 -99
  127. package/extensions/security/skills/api-security.md +140 -140
  128. package/extensions/security/skills/compliance.md +68 -68
  129. package/extensions/security/skills/owasp-audit.md +64 -64
  130. package/extensions/security/skills/pentest-patterns.md +77 -77
  131. package/extensions/security/skills/secret-mgmt.md +65 -65
  132. package/extensions/security/skills/supply-chain.md +65 -65
  133. package/extensions/trading/PACK.md +80 -80
  134. package/extensions/trading/skills/chart-components.md +55 -55
  135. package/extensions/trading/skills/experiment-loop.md +125 -125
  136. package/extensions/trading/skills/fintech-patterns.md +47 -47
  137. package/extensions/trading/skills/indicator-library.md +58 -58
  138. package/extensions/trading/skills/quant-analysis.md +111 -111
  139. package/extensions/trading/skills/realtime-data.md +58 -58
  140. package/extensions/trading/skills/trade-logic.md +104 -104
  141. package/extensions/ui/PACK.md +130 -130
  142. package/extensions/ui/skills/a11y-audit.md +91 -91
  143. package/extensions/ui/skills/animation-patterns.md +127 -127
  144. package/extensions/ui/skills/component-patterns.md +100 -100
  145. package/extensions/ui/skills/design-decision.md +108 -108
  146. package/extensions/ui/skills/design-system.md +68 -68
  147. package/extensions/ui/skills/landing-patterns.md +155 -155
  148. package/extensions/ui/skills/palette-picker.md +173 -173
  149. package/extensions/ui/skills/react-health.md +90 -90
  150. package/extensions/ui/skills/type-system.md +125 -125
  151. package/extensions/ui/skills/web-vitals.md +153 -153
  152. package/extensions/zalo/PACK.md +145 -145
  153. package/extensions/zalo/skills/zalo-oa-mcp.md +317 -317
  154. package/extensions/zalo/skills/zalo-oa-messaging.md +429 -429
  155. package/extensions/zalo/skills/zalo-oa-setup.md +236 -236
  156. package/extensions/zalo/skills/zalo-oa-webhook.md +189 -189
  157. package/extensions/zalo/skills/zalo-personal-messaging.md +194 -194
  158. package/extensions/zalo/skills/zalo-personal-setup.md +153 -153
  159. package/extensions/zalo/skills/zalo-rate-guard.md +219 -219
  160. package/hooks/auto-format/index.cjs +48 -48
  161. package/hooks/hooks.json +111 -111
  162. package/hooks/post-session-reflect/index.cjs +189 -189
  163. package/hooks/pre-compact/index.cjs +95 -95
  164. package/hooks/run-hook.cmd +1 -1
  165. package/hooks/secrets-scan/index.cjs +100 -100
  166. package/hooks/session-start/index.cjs +71 -71
  167. package/hooks/typecheck/index.cjs +65 -65
  168. package/package.json +63 -63
  169. package/references/ui-pro-max-data/LICENSE-UI-PRO-MAX +21 -21
  170. package/references/ui-pro-max-data/charts.csv +26 -26
  171. package/references/ui-pro-max-data/colors.csv +161 -161
  172. package/references/ui-pro-max-data/styles.csv +68 -68
  173. package/references/ui-pro-max-data/typography.csv +74 -74
  174. package/references/ui-pro-max-data/ui-reasoning.csv +162 -162
  175. package/references/ui-pro-max-data/ux-guidelines.csv +99 -99
  176. package/skills/adversary/SKILL.md +283 -283
  177. package/skills/asset-creator/SKILL.md +157 -157
  178. package/skills/audit/SKILL.md +147 -2
  179. package/skills/autopsy/SKILL.md +335 -335
  180. package/skills/ba/SKILL.md +85 -1
  181. package/skills/brainstorm/SKILL.md +380 -342
  182. package/skills/browser-pilot/SKILL.md +169 -168
  183. package/skills/constraint-check/SKILL.md +165 -165
  184. package/skills/context-engine/SKILL.md +408 -404
  185. package/skills/cook/SKILL.md +917 -863
  186. package/skills/db/SKILL.md +273 -273
  187. package/skills/debug/SKILL.md +465 -465
  188. package/skills/dependency-doctor/SKILL.md +265 -235
  189. package/skills/deploy/SKILL.md +274 -231
  190. package/skills/design/DESIGN-REFERENCE.md +365 -365
  191. package/skills/design/SKILL.md +590 -589
  192. package/skills/doc-processor/SKILL.md +254 -254
  193. package/skills/docs/SKILL.md +374 -374
  194. package/skills/docs-seeker/SKILL.md +178 -177
  195. package/skills/fix/SKILL.md +332 -330
  196. package/skills/git/SKILL.md +339 -339
  197. package/skills/hallucination-guard/SKILL.md +220 -219
  198. package/skills/incident/SKILL.md +254 -253
  199. package/skills/integrity-check/SKILL.md +169 -169
  200. package/skills/journal/SKILL.md +241 -240
  201. package/skills/launch/SKILL.md +344 -344
  202. package/skills/logic-guardian/SKILL.md +269 -251
  203. package/skills/marketing/SKILL.md +351 -289
  204. package/skills/mcp-builder/SKILL.md +425 -425
  205. package/skills/neural-memory/SKILL.md +359 -362
  206. package/skills/onboard/SKILL.md +432 -403
  207. package/skills/onboard/references/invariants-template.md +76 -0
  208. package/skills/onboard/scripts/detect-invariants.js +439 -0
  209. package/skills/onboard/scripts/inject-claude-md.js +150 -0
  210. package/skills/onboard/scripts/onboard-invariants.js +194 -0
  211. package/skills/perf/SKILL.md +347 -346
  212. package/skills/plan/SKILL.md +435 -428
  213. package/skills/preflight/SKILL.md +415 -415
  214. package/skills/problem-solver/SKILL.md +380 -284
  215. package/skills/rescue/SKILL.md +474 -474
  216. package/skills/research/SKILL.md +4 -0
  217. package/skills/retro/SKILL.md +3 -1
  218. package/skills/review/SKILL.md +614 -588
  219. package/skills/review-intake/SKILL.md +249 -249
  220. package/skills/safeguard/SKILL.md +200 -200
  221. package/skills/sast/SKILL.md +190 -190
  222. package/skills/scaffold/SKILL.md +328 -287
  223. package/skills/scope-guard/SKILL.md +183 -180
  224. package/skills/scout/SKILL.md +269 -263
  225. package/skills/sentinel/SKILL.md +384 -381
  226. package/skills/sentinel-env/SKILL.md +254 -254
  227. package/skills/sequential-thinking/SKILL.md +234 -234
  228. package/skills/session-bridge/SKILL.md +595 -543
  229. package/skills/session-bridge/scripts/load-invariants.js +397 -0
  230. package/skills/skill-forge/SKILL.md +581 -581
  231. package/skills/skill-router/SKILL.md +3 -0
  232. package/skills/slides/SKILL.md +19 -0
  233. package/skills/surgeon/SKILL.md +215 -215
  234. package/skills/team/SKILL.md +557 -537
  235. package/skills/test/SKILL.md +620 -614
  236. package/skills/trend-scout/SKILL.md +145 -145
  237. package/skills/verification/SKILL.md +334 -326
  238. package/skills/video-creator/SKILL.md +201 -201
  239. package/skills/watchdog/SKILL.md +168 -168
  240. package/skills/worktree/SKILL.md +140 -140
@@ -1,187 +1,187 @@
1
- ---
2
- name: "code-sandbox"
3
- pack: "@rune/ai-ml"
4
- description: "Secure code execution for AI agents — sandboxed environments for running LLM-generated code safely with container isolation, resource limits, and timeout enforcement."
5
- model: sonnet
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # code-sandbox
10
-
11
- Secure code execution for AI agents — sandboxed environments for running LLM-generated code safely. Covers container isolation, resource limits, timeout enforcement, file system boundaries, and output capture for code interpreter, CI/CD, and interactive development use cases.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Assess execution requirements**
16
- Determine what kind of code the agent needs to run:
17
-
18
- | Use Case | Isolation Level | Runtime |
19
- |---|---|---|
20
- | Code interpreter (data analysis, math) | High — untrusted code | Python + pandas/numpy |
21
- | Build/test pipeline | Medium — project code | Node.js / Python with project deps |
22
- | Interactive preview (web app) | Medium — expose HTTP port | Node.js + browser preview |
23
- | Shell commands (file ops, git) | Low — trusted context | System shell with path restrictions |
24
-
25
- **Step 2 — Configure sandbox environment**
26
- Emit sandbox configuration based on use case:
27
-
28
- ```typescript
29
- // Sandbox factory — select isolation level by use case
30
- interface SandboxConfig {
31
- language: 'python' | 'javascript' | 'typescript';
32
- timeout: number; // max execution time in ms
33
- memoryLimit: number; // max memory in MB
34
- networkAccess: boolean;
35
- fileSystemRoot: string; // restricted working directory
36
- allowedModules: string[];
37
- }
38
-
39
- const SANDBOX_PRESETS: Record<string, SandboxConfig> = {
40
- 'code-interpreter': {
41
- language: 'python',
42
- timeout: 30_000,
43
- memoryLimit: 256,
44
- networkAccess: false,
45
- fileSystemRoot: '/workspace',
46
- allowedModules: ['pandas', 'numpy', 'matplotlib', 'scipy', 'json', 'csv', 'math'],
47
- },
48
- 'build-test': {
49
- language: 'typescript',
50
- timeout: 120_000,
51
- memoryLimit: 512,
52
- networkAccess: true, // needs npm registry
53
- fileSystemRoot: '/project',
54
- allowedModules: ['*'], // project dependencies
55
- },
56
- 'preview': {
57
- language: 'javascript',
58
- timeout: 300_000,
59
- memoryLimit: 256,
60
- networkAccess: true,
61
- fileSystemRoot: '/app',
62
- allowedModules: ['*'],
63
- },
64
- };
65
- ```
66
-
67
- **Step 3 — Implement execution with resource limits**
68
- Emit code execution wrapper with safety boundaries:
69
-
70
- ```typescript
71
- // Docker-based sandbox execution
72
- import { spawn } from 'child_process';
73
-
74
- interface ExecutionResult {
75
- stdout: string;
76
- stderr: string;
77
- exitCode: number;
78
- durationMs: number;
79
- timedOut: boolean;
80
- }
81
-
82
- async function executeInSandbox(
83
- code: string,
84
- config: SandboxConfig
85
- ): Promise<ExecutionResult> {
86
- const start = Date.now();
87
-
88
- // Write code to temp file in sandbox root
89
- const codePath = `${config.fileSystemRoot}/run.${config.language === 'python' ? 'py' : 'ts'}`;
90
- await writeFile(codePath, code);
91
-
92
- const proc = spawn('docker', [
93
- 'run', '--rm',
94
- '--memory', `${config.memoryLimit}m`,
95
- '--cpus', '1',
96
- '--network', config.networkAccess ? 'bridge' : 'none',
97
- '--read-only',
98
- '--tmpfs', '/tmp:size=64m',
99
- '-v', `${config.fileSystemRoot}:/workspace:ro`,
100
- '-w', '/workspace',
101
- `sandbox-${config.language}:latest`,
102
- config.language === 'python' ? 'python' : 'npx tsx',
103
- `/workspace/run.${config.language === 'python' ? 'py' : 'ts'}`,
104
- ]);
105
-
106
- let stdout = '';
107
- let stderr = '';
108
- let timedOut = false;
109
-
110
- proc.stdout.on('data', (d) => { stdout += d.toString(); });
111
- proc.stderr.on('data', (d) => { stderr += d.toString(); });
112
-
113
- const timeout = setTimeout(() => {
114
- timedOut = true;
115
- proc.kill('SIGKILL');
116
- }, config.timeout);
117
-
118
- const exitCode = await new Promise<number>((resolve) => {
119
- proc.on('close', (code) => {
120
- clearTimeout(timeout);
121
- resolve(code ?? 1);
122
- });
123
- });
124
-
125
- return { stdout, stderr, exitCode, durationMs: Date.now() - start, timedOut };
126
- }
127
- ```
128
-
129
- **Step 4 — Code interpreter mode (stateful sessions)**
130
- For multi-turn code execution where variables persist between runs:
131
-
132
- ```typescript
133
- // Stateful code interpreter — variables persist across executions
134
- interface CodeSession {
135
- id: string;
136
- language: 'python' | 'javascript';
137
- history: { code: string; result: ExecutionResult }[];
138
- }
139
-
140
- async function runInSession(
141
- session: CodeSession,
142
- code: string
143
- ): Promise<ExecutionResult> {
144
- // Python: use exec() with persistent globals dict
145
- // JavaScript: use Node.js vm module with persistent context
146
- const wrappedCode = session.language === 'python'
147
- ? `exec(${JSON.stringify(code)}, _globals)`
148
- : code;
149
-
150
- const result = await executeInSandbox(wrappedCode, SANDBOX_PRESETS['code-interpreter']);
151
-
152
- // Append to history (immutable update)
153
- session.history = [...session.history, { code, result }];
154
-
155
- return result;
156
- }
157
-
158
- // Rich output capture — not just stdout
159
- interface RichOutput {
160
- text?: string;
161
- images?: { data: string; mimeType: string }[]; // base64 encoded
162
- tables?: { headers: string[]; rows: string[][] }[];
163
- error?: string;
164
- }
165
- ```
166
-
167
- **Step 5 — Security boundaries**
168
- Enforce isolation guarantees:
169
-
170
- | Boundary | Enforcement |
171
- |---|---|
172
- | File system | Read-only mount + tmpfs for temp files. No access to host filesystem. |
173
- | Network | `--network none` for code interpreter. Whitelist for build/test. |
174
- | Memory | Docker `--memory` limit. OOM killed if exceeded. |
175
- | CPU | Docker `--cpus` limit. Prevents crypto mining / infinite loops. |
176
- | Time | Kill process after timeout. Return partial output. |
177
- | Secrets | Never mount env vars or secrets into sandbox container. |
178
- | Output size | Cap stdout/stderr at 1MB. Truncate with `[output truncated]` marker. |
179
-
180
- #### Sharp Edges
181
-
182
- | Failure Mode | Mitigation |
183
- |---|---|
184
- | Sandbox escape via Docker vulnerability | Pin Docker version; use rootless Docker; consider gVisor/Firecracker for high-security |
185
- | Code writes to /tmp exhausting disk | Use `--tmpfs` with size limit (64MB default) |
186
- | Infinite loop inside sandbox hangs API | Hard timeout with SIGKILL — never rely on SIGTERM alone |
187
- | Stateful session grows unbounded memory | Limit session history to last 50 executions; reset context on overflow |
1
+ ---
2
+ name: "code-sandbox"
3
+ pack: "@rune/ai-ml"
4
+ description: "Secure code execution for AI agents — sandboxed environments for running LLM-generated code safely with container isolation, resource limits, and timeout enforcement."
5
+ model: sonnet
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # code-sandbox
10
+
11
+ Secure code execution for AI agents — sandboxed environments for running LLM-generated code safely. Covers container isolation, resource limits, timeout enforcement, file system boundaries, and output capture for code interpreter, CI/CD, and interactive development use cases.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Assess execution requirements**
16
+ Determine what kind of code the agent needs to run:
17
+
18
+ | Use Case | Isolation Level | Runtime |
19
+ |---|---|---|
20
+ | Code interpreter (data analysis, math) | High — untrusted code | Python + pandas/numpy |
21
+ | Build/test pipeline | Medium — project code | Node.js / Python with project deps |
22
+ | Interactive preview (web app) | Medium — expose HTTP port | Node.js + browser preview |
23
+ | Shell commands (file ops, git) | Low — trusted context | System shell with path restrictions |
24
+
25
+ **Step 2 — Configure sandbox environment**
26
+ Emit sandbox configuration based on use case:
27
+
28
+ ```typescript
29
+ // Sandbox factory — select isolation level by use case
30
+ interface SandboxConfig {
31
+ language: 'python' | 'javascript' | 'typescript';
32
+ timeout: number; // max execution time in ms
33
+ memoryLimit: number; // max memory in MB
34
+ networkAccess: boolean;
35
+ fileSystemRoot: string; // restricted working directory
36
+ allowedModules: string[];
37
+ }
38
+
39
+ const SANDBOX_PRESETS: Record<string, SandboxConfig> = {
40
+ 'code-interpreter': {
41
+ language: 'python',
42
+ timeout: 30_000,
43
+ memoryLimit: 256,
44
+ networkAccess: false,
45
+ fileSystemRoot: '/workspace',
46
+ allowedModules: ['pandas', 'numpy', 'matplotlib', 'scipy', 'json', 'csv', 'math'],
47
+ },
48
+ 'build-test': {
49
+ language: 'typescript',
50
+ timeout: 120_000,
51
+ memoryLimit: 512,
52
+ networkAccess: true, // needs npm registry
53
+ fileSystemRoot: '/project',
54
+ allowedModules: ['*'], // project dependencies
55
+ },
56
+ 'preview': {
57
+ language: 'javascript',
58
+ timeout: 300_000,
59
+ memoryLimit: 256,
60
+ networkAccess: true,
61
+ fileSystemRoot: '/app',
62
+ allowedModules: ['*'],
63
+ },
64
+ };
65
+ ```
66
+
67
+ **Step 3 — Implement execution with resource limits**
68
+ Emit code execution wrapper with safety boundaries:
69
+
70
+ ```typescript
71
+ // Docker-based sandbox execution
72
+ import { spawn } from 'child_process';
73
+
74
+ interface ExecutionResult {
75
+ stdout: string;
76
+ stderr: string;
77
+ exitCode: number;
78
+ durationMs: number;
79
+ timedOut: boolean;
80
+ }
81
+
82
+ async function executeInSandbox(
83
+ code: string,
84
+ config: SandboxConfig
85
+ ): Promise<ExecutionResult> {
86
+ const start = Date.now();
87
+
88
+ // Write code to temp file in sandbox root
89
+ const codePath = `${config.fileSystemRoot}/run.${config.language === 'python' ? 'py' : 'ts'}`;
90
+ await writeFile(codePath, code);
91
+
92
+ const proc = spawn('docker', [
93
+ 'run', '--rm',
94
+ '--memory', `${config.memoryLimit}m`,
95
+ '--cpus', '1',
96
+ '--network', config.networkAccess ? 'bridge' : 'none',
97
+ '--read-only',
98
+ '--tmpfs', '/tmp:size=64m',
99
+ '-v', `${config.fileSystemRoot}:/workspace:ro`,
100
+ '-w', '/workspace',
101
+ `sandbox-${config.language}:latest`,
102
+ config.language === 'python' ? 'python' : 'npx tsx',
103
+ `/workspace/run.${config.language === 'python' ? 'py' : 'ts'}`,
104
+ ]);
105
+
106
+ let stdout = '';
107
+ let stderr = '';
108
+ let timedOut = false;
109
+
110
+ proc.stdout.on('data', (d) => { stdout += d.toString(); });
111
+ proc.stderr.on('data', (d) => { stderr += d.toString(); });
112
+
113
+ const timeout = setTimeout(() => {
114
+ timedOut = true;
115
+ proc.kill('SIGKILL');
116
+ }, config.timeout);
117
+
118
+ const exitCode = await new Promise<number>((resolve) => {
119
+ proc.on('close', (code) => {
120
+ clearTimeout(timeout);
121
+ resolve(code ?? 1);
122
+ });
123
+ });
124
+
125
+ return { stdout, stderr, exitCode, durationMs: Date.now() - start, timedOut };
126
+ }
127
+ ```
128
+
129
+ **Step 4 — Code interpreter mode (stateful sessions)**
130
+ For multi-turn code execution where variables persist between runs:
131
+
132
+ ```typescript
133
+ // Stateful code interpreter — variables persist across executions
134
+ interface CodeSession {
135
+ id: string;
136
+ language: 'python' | 'javascript';
137
+ history: { code: string; result: ExecutionResult }[];
138
+ }
139
+
140
+ async function runInSession(
141
+ session: CodeSession,
142
+ code: string
143
+ ): Promise<ExecutionResult> {
144
+ // Python: use exec() with persistent globals dict
145
+ // JavaScript: use Node.js vm module with persistent context
146
+ const wrappedCode = session.language === 'python'
147
+ ? `exec(${JSON.stringify(code)}, _globals)`
148
+ : code;
149
+
150
+ const result = await executeInSandbox(wrappedCode, SANDBOX_PRESETS['code-interpreter']);
151
+
152
+ // Append to history (immutable update)
153
+ session.history = [...session.history, { code, result }];
154
+
155
+ return result;
156
+ }
157
+
158
+ // Rich output capture — not just stdout
159
+ interface RichOutput {
160
+ text?: string;
161
+ images?: { data: string; mimeType: string }[]; // base64 encoded
162
+ tables?: { headers: string[]; rows: string[][] }[];
163
+ error?: string;
164
+ }
165
+ ```
166
+
167
+ **Step 5 — Security boundaries**
168
+ Enforce isolation guarantees:
169
+
170
+ | Boundary | Enforcement |
171
+ |---|---|
172
+ | File system | Read-only mount + tmpfs for temp files. No access to host filesystem. |
173
+ | Network | `--network none` for code interpreter. Whitelist for build/test. |
174
+ | Memory | Docker `--memory` limit. OOM killed if exceeded. |
175
+ | CPU | Docker `--cpus` limit. Prevents crypto mining / infinite loops. |
176
+ | Time | Kill process after timeout. Return partial output. |
177
+ | Secrets | Never mount env vars or secrets into sandbox container. |
178
+ | Output size | Cap stdout/stderr at 1MB. Truncate with `[output truncated]` marker. |
179
+
180
+ #### Sharp Edges
181
+
182
+ | Failure Mode | Mitigation |
183
+ |---|---|
184
+ | Sandbox escape via Docker vulnerability | Pin Docker version; use rootless Docker; consider gVisor/Firecracker for high-security |
185
+ | Code writes to /tmp exhausting disk | Use `--tmpfs` with size limit (64MB default) |
186
+ | Infinite loop inside sandbox hangs API | Hard timeout with SIGKILL — never rely on SIGTERM alone |
187
+ | Stateful session grows unbounded memory | Limit session history to last 50 executions; reset context on overflow |
@@ -1,146 +1,146 @@
1
- ---
2
- name: "deep-research"
3
- pack: "@rune/ai-ml"
4
- description: "Iterative AI research loop that converges on comprehensive answers — search, analyze, identify gaps, repeat. Outputs synthesized report with source attribution."
5
- model: sonnet
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # deep-research
10
-
11
- Iterative AI research loop that converges on comprehensive answers. Search → analyze → identify gaps → search again. Bounded by depth, time, and URL limits. Outputs synthesized report with source attribution.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Initialize research state**
16
- ```typescript
17
- interface ResearchState {
18
- query: string;
19
- findings: Finding[]; // max 50 most recent (memory bound)
20
- gaps: string[]; // what we still don't know
21
- seenUrls: Set<string>; // dedup
22
- failedQueries: number; // convergence signal
23
- depth: number; // current iteration
24
- maxDepth: number; // hard limit (default: 10)
25
- maxUrls: number; // hard limit (default: 100)
26
- maxTimeMs: number; // hard limit (default: 300_000 = 5 min)
27
- startedAt: number;
28
- activityLog: ActivityEntry[]; // for progress streaming
29
- }
30
-
31
- interface Finding {
32
- content: string;
33
- sourceUrl: string;
34
- relevance: number; // 0-1
35
- extractedAt: number;
36
- }
37
- ```
38
-
39
- **Step 2 — Generate search queries from current state**
40
- Each iteration, LLM generates 3 search queries based on:
41
- - Original research question
42
- - Current findings (what we know)
43
- - Current gaps (what we don't know)
44
-
45
- ```typescript
46
- const queryPrompt = `Given the research question: "${state.query}"
47
- Current findings: ${summarizeFindings(state.findings)}
48
- Knowledge gaps: ${state.gaps.join(', ')}
49
-
50
- Generate 3 specific search queries that would fill the most important gaps.
51
- Avoid queries similar to: ${state.seenQueries.join(', ')}`;
52
- ```
53
-
54
- **Step 3 — Search and deduplicate**
55
- Execute queries in parallel → collect URLs → filter against `seenUrls` → scrape new URLs → extract relevant content.
56
-
57
- ```typescript
58
- async function searchAndExtract(queries: string[], state: ResearchState): Promise<Finding[]> {
59
- // Parallel search
60
- const allResults = await Promise.all(queries.map(q => webSearch(q, { limit: 10 })));
61
- const urls = deduplicateUrls(allResults.flat(), state.seenUrls);
62
-
63
- // Mark as seen immediately (even before scraping)
64
- for (const url of urls) state.seenUrls.add(url);
65
-
66
- // Scrape and extract in parallel (with concurrency limit)
67
- const findings = await pMap(urls, async (url) => {
68
- const content = await scrapeAndClean(url);
69
- const relevance = await scoreRelevance(content, state.query);
70
- return { content: summarize(content, 500), sourceUrl: url, relevance, extractedAt: Date.now() };
71
- }, { concurrency: 5 });
72
-
73
- return findings.filter(f => f.relevance > 0.3); // threshold
74
- }
75
- ```
76
-
77
- **Step 4 — Analyze findings and detect gaps**
78
- LLM analyzes new findings against existing knowledge:
79
- ```typescript
80
- interface AnalysisResult {
81
- newInsights: string[];
82
- updatedGaps: string[];
83
- shouldContinue: boolean;
84
- nextSearchTopic: string | null;
85
- confidence: number; // 0-1: how complete is our understanding?
86
- }
87
- ```
88
-
89
- **Step 5 — Check convergence criteria**
90
- Stop when ANY of:
91
- - `depth >= maxDepth`
92
- - `seenUrls.size >= maxUrls`
93
- - `Date.now() - startedAt >= maxTimeMs`
94
- - `gaps.length === 0` (all gaps filled)
95
- - `failedQueries >= 3` consecutive (no new information available)
96
- - `confidence >= 0.9` (LLM believes research is comprehensive)
97
-
98
- **Step 6 — Synthesize final report**
99
- ```typescript
100
- interface ResearchReport {
101
- question: string;
102
- answer: string; // comprehensive markdown synthesis
103
- confidence: number;
104
- sources: Array<{
105
- url: string;
106
- title: string;
107
- relevance: number;
108
- citedIn: string[]; // which sections cite this source
109
- }>;
110
- methodology: {
111
- totalIterations: number;
112
- urlsExamined: number;
113
- findingsCount: number;
114
- timeElapsed: number;
115
- remainingGaps: string[];
116
- };
117
- }
118
- ```
119
-
120
- Memory management: keep only 50 most recent findings to avoid context explosion. Summarize older findings into a "background knowledge" string before dropping them.
121
-
122
- #### Example
123
-
124
- ```typescript
125
- // Usage
126
- const report = await deepResearch({
127
- query: 'What are the best practices for implementing RAG in production in 2026?',
128
- maxDepth: 8,
129
- maxUrls: 50,
130
- maxTimeMs: 180_000, // 3 minutes
131
- onProgress: (entry) => console.log(`[${entry.depth}] ${entry.action}: ${entry.detail}`),
132
- });
133
-
134
- // Output: comprehensive report with 15-30 sources, gap analysis, confidence score
135
- ```
136
-
137
- #### Sharp Edges
138
-
139
- | Failure Mode | Mitigation |
140
- |---|---|
141
- | Research loop runs forever (no convergence) | Hard limits on depth, URLs, and time; monitor `failedQueries` counter |
142
- | LLM generates duplicate search queries | Track seen queries; include exclusion list in prompt |
143
- | Memory explosion from accumulating findings | Cap at 50 findings; summarize oldest into background knowledge string |
144
- | Low-quality sources pollute findings | Relevance threshold (0.3); domain blocklist for known low-quality sites |
145
- | Rate limiting on search API | Per-provider rate limiter; fallback to alternative search provider |
146
- | Circular research (keeps finding same information) | Track `confidence` — if stable for 3 iterations, force stop |
1
+ ---
2
+ name: "deep-research"
3
+ pack: "@rune/ai-ml"
4
+ description: "Iterative AI research loop that converges on comprehensive answers — search, analyze, identify gaps, repeat. Outputs synthesized report with source attribution."
5
+ model: sonnet
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # deep-research
10
+
11
+ Iterative AI research loop that converges on comprehensive answers. Search → analyze → identify gaps → search again. Bounded by depth, time, and URL limits. Outputs synthesized report with source attribution.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Initialize research state**
16
+ ```typescript
17
+ interface ResearchState {
18
+ query: string;
19
+ findings: Finding[]; // max 50 most recent (memory bound)
20
+ gaps: string[]; // what we still don't know
21
+ seenUrls: Set<string>; // dedup
22
+ failedQueries: number; // convergence signal
23
+ depth: number; // current iteration
24
+ maxDepth: number; // hard limit (default: 10)
25
+ maxUrls: number; // hard limit (default: 100)
26
+ maxTimeMs: number; // hard limit (default: 300_000 = 5 min)
27
+ startedAt: number;
28
+ activityLog: ActivityEntry[]; // for progress streaming
29
+ }
30
+
31
+ interface Finding {
32
+ content: string;
33
+ sourceUrl: string;
34
+ relevance: number; // 0-1
35
+ extractedAt: number;
36
+ }
37
+ ```
38
+
39
+ **Step 2 — Generate search queries from current state**
40
+ Each iteration, LLM generates 3 search queries based on:
41
+ - Original research question
42
+ - Current findings (what we know)
43
+ - Current gaps (what we don't know)
44
+
45
+ ```typescript
46
+ const queryPrompt = `Given the research question: "${state.query}"
47
+ Current findings: ${summarizeFindings(state.findings)}
48
+ Knowledge gaps: ${state.gaps.join(', ')}
49
+
50
+ Generate 3 specific search queries that would fill the most important gaps.
51
+ Avoid queries similar to: ${state.seenQueries.join(', ')}`;
52
+ ```
53
+
54
+ **Step 3 — Search and deduplicate**
55
+ Execute queries in parallel → collect URLs → filter against `seenUrls` → scrape new URLs → extract relevant content.
56
+
57
+ ```typescript
58
+ async function searchAndExtract(queries: string[], state: ResearchState): Promise<Finding[]> {
59
+ // Parallel search
60
+ const allResults = await Promise.all(queries.map(q => webSearch(q, { limit: 10 })));
61
+ const urls = deduplicateUrls(allResults.flat(), state.seenUrls);
62
+
63
+ // Mark as seen immediately (even before scraping)
64
+ for (const url of urls) state.seenUrls.add(url);
65
+
66
+ // Scrape and extract in parallel (with concurrency limit)
67
+ const findings = await pMap(urls, async (url) => {
68
+ const content = await scrapeAndClean(url);
69
+ const relevance = await scoreRelevance(content, state.query);
70
+ return { content: summarize(content, 500), sourceUrl: url, relevance, extractedAt: Date.now() };
71
+ }, { concurrency: 5 });
72
+
73
+ return findings.filter(f => f.relevance > 0.3); // threshold
74
+ }
75
+ ```
76
+
77
+ **Step 4 — Analyze findings and detect gaps**
78
+ LLM analyzes new findings against existing knowledge:
79
+ ```typescript
80
+ interface AnalysisResult {
81
+ newInsights: string[];
82
+ updatedGaps: string[];
83
+ shouldContinue: boolean;
84
+ nextSearchTopic: string | null;
85
+ confidence: number; // 0-1: how complete is our understanding?
86
+ }
87
+ ```
88
+
89
+ **Step 5 — Check convergence criteria**
90
+ Stop when ANY of:
91
+ - `depth >= maxDepth`
92
+ - `seenUrls.size >= maxUrls`
93
+ - `Date.now() - startedAt >= maxTimeMs`
94
+ - `gaps.length === 0` (all gaps filled)
95
+ - `failedQueries >= 3` consecutive (no new information available)
96
+ - `confidence >= 0.9` (LLM believes research is comprehensive)
97
+
98
+ **Step 6 — Synthesize final report**
99
+ ```typescript
100
+ interface ResearchReport {
101
+ question: string;
102
+ answer: string; // comprehensive markdown synthesis
103
+ confidence: number;
104
+ sources: Array<{
105
+ url: string;
106
+ title: string;
107
+ relevance: number;
108
+ citedIn: string[]; // which sections cite this source
109
+ }>;
110
+ methodology: {
111
+ totalIterations: number;
112
+ urlsExamined: number;
113
+ findingsCount: number;
114
+ timeElapsed: number;
115
+ remainingGaps: string[];
116
+ };
117
+ }
118
+ ```
119
+
120
+ Memory management: keep only 50 most recent findings to avoid context explosion. Summarize older findings into a "background knowledge" string before dropping them.
121
+
122
+ #### Example
123
+
124
+ ```typescript
125
+ // Usage
126
+ const report = await deepResearch({
127
+ query: 'What are the best practices for implementing RAG in production in 2026?',
128
+ maxDepth: 8,
129
+ maxUrls: 50,
130
+ maxTimeMs: 180_000, // 3 minutes
131
+ onProgress: (entry) => console.log(`[${entry.depth}] ${entry.action}: ${entry.detail}`),
132
+ });
133
+
134
+ // Output: comprehensive report with 15-30 sources, gap analysis, confidence score
135
+ ```
136
+
137
+ #### Sharp Edges
138
+
139
+ | Failure Mode | Mitigation |
140
+ |---|---|
141
+ | Research loop runs forever (no convergence) | Hard limits on depth, URLs, and time; monitor `failedQueries` counter |
142
+ | LLM generates duplicate search queries | Track seen queries; include exclusion list in prompt |
143
+ | Memory explosion from accumulating findings | Cap at 50 findings; summarize oldest into background knowledge string |
144
+ | Low-quality sources pollute findings | Relevance threshold (0.3); domain blocklist for known low-quality sites |
145
+ | Rate limiting on search API | Per-provider rate limiter; fallback to alternative search provider |
146
+ | Circular research (keeps finding same information) | Track `confidence` — if stable for 3 iterations, force stop |