@rune-kit/rune 2.10.0 → 2.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +65 -6
  3. package/commands/rune.md +168 -168
  4. package/compiler/__tests__/detect-invariants.test.js +136 -0
  5. package/compiler/__tests__/doctor-mesh.test.js +229 -0
  6. package/compiler/__tests__/hook-dispatch.test.js +91 -0
  7. package/compiler/__tests__/hooks-antigravity.test.js +118 -0
  8. package/compiler/__tests__/hooks-cursor.test.js +139 -0
  9. package/compiler/__tests__/hooks-install.test.js +305 -0
  10. package/compiler/__tests__/hooks-merge.test.js +204 -0
  11. package/compiler/__tests__/hooks-tiers.test.js +519 -0
  12. package/compiler/__tests__/hooks-windsurf.test.js +115 -0
  13. package/compiler/__tests__/inject-claude-md.test.js +152 -0
  14. package/compiler/__tests__/load-invariants.test.js +408 -0
  15. package/compiler/__tests__/onboard-invariants.test.js +240 -0
  16. package/compiler/adapters/hooks/antigravity.js +140 -0
  17. package/compiler/adapters/hooks/claude.js +166 -0
  18. package/compiler/adapters/hooks/cursor.js +191 -0
  19. package/compiler/adapters/hooks/index.js +82 -0
  20. package/compiler/adapters/hooks/tier-emitter.js +182 -0
  21. package/compiler/adapters/hooks/windsurf.js +202 -0
  22. package/compiler/bin/rune.js +196 -6
  23. package/compiler/commands/hook-dispatch.js +87 -0
  24. package/compiler/commands/hooks/install.js +120 -0
  25. package/compiler/commands/hooks/merge.js +211 -0
  26. package/compiler/commands/hooks/presets.js +116 -0
  27. package/compiler/commands/hooks/status.js +112 -0
  28. package/compiler/commands/hooks/tiers.js +221 -0
  29. package/compiler/commands/hooks/uninstall.js +94 -0
  30. package/compiler/doctor.js +236 -0
  31. package/contexts/dev.md +34 -34
  32. package/contexts/research.md +43 -43
  33. package/contexts/review.md +55 -55
  34. package/extensions/ai-ml/PACK.md +88 -88
  35. package/extensions/ai-ml/skills/ai-agents.md +172 -172
  36. package/extensions/ai-ml/skills/code-sandbox.md +187 -187
  37. package/extensions/ai-ml/skills/deep-research.md +146 -146
  38. package/extensions/ai-ml/skills/embedding-search.md +66 -66
  39. package/extensions/ai-ml/skills/fine-tuning-guide.md +74 -74
  40. package/extensions/ai-ml/skills/llm-architect.md +125 -125
  41. package/extensions/ai-ml/skills/llm-integration.md +64 -64
  42. package/extensions/ai-ml/skills/prompt-patterns.md +72 -72
  43. package/extensions/ai-ml/skills/rag-patterns.md +66 -66
  44. package/extensions/ai-ml/skills/web-extraction.md +114 -114
  45. package/extensions/analytics/PACK.md +92 -92
  46. package/extensions/analytics/skills/ab-testing.md +72 -72
  47. package/extensions/analytics/skills/dashboard-patterns.md +83 -83
  48. package/extensions/analytics/skills/data-validation.md +68 -68
  49. package/extensions/analytics/skills/funnel-analysis.md +81 -81
  50. package/extensions/analytics/skills/sql-patterns.md +57 -57
  51. package/extensions/analytics/skills/statistical-analysis.md +79 -79
  52. package/extensions/analytics/skills/tracking-setup.md +71 -71
  53. package/extensions/backend/PACK.md +104 -104
  54. package/extensions/backend/skills/api-patterns.md +84 -84
  55. package/extensions/backend/skills/async-pipeline.md +193 -193
  56. package/extensions/backend/skills/auth-patterns.md +97 -97
  57. package/extensions/backend/skills/background-jobs.md +133 -133
  58. package/extensions/backend/skills/caching-patterns.md +108 -108
  59. package/extensions/backend/skills/cli-generation.md +133 -133
  60. package/extensions/backend/skills/database-patterns.md +87 -87
  61. package/extensions/backend/skills/middleware-patterns.md +104 -104
  62. package/extensions/chrome-ext/PACK.md +93 -93
  63. package/extensions/chrome-ext/skills/cws-preflight.md +143 -143
  64. package/extensions/chrome-ext/skills/cws-publish.md +104 -104
  65. package/extensions/chrome-ext/skills/ext-ai-integration.md +251 -251
  66. package/extensions/chrome-ext/skills/ext-messaging.md +139 -139
  67. package/extensions/chrome-ext/skills/ext-storage.md +133 -133
  68. package/extensions/chrome-ext/skills/mv3-scaffold.md +164 -164
  69. package/extensions/content/PACK.md +96 -96
  70. package/extensions/content/skills/blog-patterns.md +88 -88
  71. package/extensions/content/skills/cms-integration.md +131 -131
  72. package/extensions/content/skills/content-scoring.md +107 -107
  73. package/extensions/content/skills/i18n.md +83 -83
  74. package/extensions/content/skills/mdx-authoring.md +137 -137
  75. package/extensions/content/skills/reference.md +1014 -1014
  76. package/extensions/content/skills/seo-patterns.md +67 -67
  77. package/extensions/content/skills/video-repurpose.md +153 -153
  78. package/extensions/devops/PACK.md +101 -101
  79. package/extensions/devops/skills/chaos-testing.md +67 -67
  80. package/extensions/devops/skills/ci-cd.md +75 -75
  81. package/extensions/devops/skills/docker.md +58 -58
  82. package/extensions/devops/skills/edge-serverless.md +163 -163
  83. package/extensions/devops/skills/infra-as-code.md +158 -158
  84. package/extensions/devops/skills/kubernetes.md +110 -110
  85. package/extensions/devops/skills/monitoring.md +57 -57
  86. package/extensions/devops/skills/server-setup.md +64 -64
  87. package/extensions/devops/skills/ssl-domain.md +42 -42
  88. package/extensions/ecommerce/PACK.md +116 -116
  89. package/extensions/ecommerce/skills/cart-system.md +79 -79
  90. package/extensions/ecommerce/skills/inventory-mgmt.md +102 -102
  91. package/extensions/ecommerce/skills/order-management.md +126 -126
  92. package/extensions/ecommerce/skills/payment-integration.md +472 -472
  93. package/extensions/ecommerce/skills/shopify-dev.md +69 -69
  94. package/extensions/ecommerce/skills/subscription-billing.md +93 -93
  95. package/extensions/ecommerce/skills/tax-compliance.md +117 -117
  96. package/extensions/gamedev/PACK.md +142 -142
  97. package/extensions/gamedev/skills/asset-pipeline.md +74 -74
  98. package/extensions/gamedev/skills/audio-system.md +129 -129
  99. package/extensions/gamedev/skills/camera-system.md +87 -87
  100. package/extensions/gamedev/skills/ecs.md +98 -98
  101. package/extensions/gamedev/skills/game-loops.md +72 -72
  102. package/extensions/gamedev/skills/input-system.md +199 -199
  103. package/extensions/gamedev/skills/multiplayer.md +180 -180
  104. package/extensions/gamedev/skills/particles.md +105 -105
  105. package/extensions/gamedev/skills/physics-engine.md +89 -89
  106. package/extensions/gamedev/skills/scene-management.md +146 -146
  107. package/extensions/gamedev/skills/threejs-patterns.md +90 -90
  108. package/extensions/gamedev/skills/webgl.md +71 -71
  109. package/extensions/mobile/PACK.md +106 -106
  110. package/extensions/mobile/skills/app-store-connect.md +152 -152
  111. package/extensions/mobile/skills/app-store-prep.md +66 -66
  112. package/extensions/mobile/skills/deep-linking.md +109 -109
  113. package/extensions/mobile/skills/flutter.md +60 -60
  114. package/extensions/mobile/skills/ios-build-pipeline.md +142 -142
  115. package/extensions/mobile/skills/native-bridge.md +66 -66
  116. package/extensions/mobile/skills/ota-updates.md +97 -97
  117. package/extensions/mobile/skills/push-notifications.md +111 -111
  118. package/extensions/mobile/skills/react-native.md +82 -82
  119. package/extensions/saas/PACK.md +116 -116
  120. package/extensions/saas/skills/billing-integration.md +200 -200
  121. package/extensions/saas/skills/feature-flags.md +130 -130
  122. package/extensions/saas/skills/multi-tenant.md +103 -103
  123. package/extensions/saas/skills/onboarding-flow.md +139 -139
  124. package/extensions/saas/skills/subscription-flow.md +95 -95
  125. package/extensions/saas/skills/team-management.md +144 -144
  126. package/extensions/security/PACK.md +99 -99
  127. package/extensions/security/skills/api-security.md +140 -140
  128. package/extensions/security/skills/compliance.md +68 -68
  129. package/extensions/security/skills/owasp-audit.md +64 -64
  130. package/extensions/security/skills/pentest-patterns.md +77 -77
  131. package/extensions/security/skills/secret-mgmt.md +65 -65
  132. package/extensions/security/skills/supply-chain.md +65 -65
  133. package/extensions/trading/PACK.md +80 -80
  134. package/extensions/trading/skills/chart-components.md +55 -55
  135. package/extensions/trading/skills/experiment-loop.md +125 -125
  136. package/extensions/trading/skills/fintech-patterns.md +47 -47
  137. package/extensions/trading/skills/indicator-library.md +58 -58
  138. package/extensions/trading/skills/quant-analysis.md +111 -111
  139. package/extensions/trading/skills/realtime-data.md +58 -58
  140. package/extensions/trading/skills/trade-logic.md +104 -104
  141. package/extensions/ui/PACK.md +130 -130
  142. package/extensions/ui/skills/a11y-audit.md +91 -91
  143. package/extensions/ui/skills/animation-patterns.md +127 -127
  144. package/extensions/ui/skills/component-patterns.md +100 -100
  145. package/extensions/ui/skills/design-decision.md +108 -108
  146. package/extensions/ui/skills/design-system.md +68 -68
  147. package/extensions/ui/skills/landing-patterns.md +155 -155
  148. package/extensions/ui/skills/palette-picker.md +173 -173
  149. package/extensions/ui/skills/react-health.md +90 -90
  150. package/extensions/ui/skills/type-system.md +125 -125
  151. package/extensions/ui/skills/web-vitals.md +153 -153
  152. package/extensions/zalo/PACK.md +145 -145
  153. package/extensions/zalo/skills/zalo-oa-mcp.md +317 -317
  154. package/extensions/zalo/skills/zalo-oa-messaging.md +429 -429
  155. package/extensions/zalo/skills/zalo-oa-setup.md +236 -236
  156. package/extensions/zalo/skills/zalo-oa-webhook.md +189 -189
  157. package/extensions/zalo/skills/zalo-personal-messaging.md +194 -194
  158. package/extensions/zalo/skills/zalo-personal-setup.md +153 -153
  159. package/extensions/zalo/skills/zalo-rate-guard.md +219 -219
  160. package/hooks/auto-format/index.cjs +48 -48
  161. package/hooks/hooks.json +111 -111
  162. package/hooks/post-session-reflect/index.cjs +189 -189
  163. package/hooks/pre-compact/index.cjs +95 -95
  164. package/hooks/run-hook.cmd +1 -1
  165. package/hooks/secrets-scan/index.cjs +100 -100
  166. package/hooks/session-start/index.cjs +71 -71
  167. package/hooks/typecheck/index.cjs +65 -65
  168. package/package.json +63 -63
  169. package/references/ui-pro-max-data/LICENSE-UI-PRO-MAX +21 -21
  170. package/references/ui-pro-max-data/charts.csv +26 -26
  171. package/references/ui-pro-max-data/colors.csv +161 -161
  172. package/references/ui-pro-max-data/styles.csv +68 -68
  173. package/references/ui-pro-max-data/typography.csv +74 -74
  174. package/references/ui-pro-max-data/ui-reasoning.csv +162 -162
  175. package/references/ui-pro-max-data/ux-guidelines.csv +99 -99
  176. package/skills/adversary/SKILL.md +283 -283
  177. package/skills/asset-creator/SKILL.md +157 -157
  178. package/skills/audit/SKILL.md +147 -2
  179. package/skills/autopsy/SKILL.md +335 -335
  180. package/skills/ba/SKILL.md +85 -1
  181. package/skills/brainstorm/SKILL.md +380 -342
  182. package/skills/browser-pilot/SKILL.md +169 -168
  183. package/skills/constraint-check/SKILL.md +165 -165
  184. package/skills/context-engine/SKILL.md +408 -404
  185. package/skills/cook/SKILL.md +917 -863
  186. package/skills/db/SKILL.md +273 -273
  187. package/skills/debug/SKILL.md +465 -465
  188. package/skills/dependency-doctor/SKILL.md +265 -235
  189. package/skills/deploy/SKILL.md +274 -231
  190. package/skills/design/DESIGN-REFERENCE.md +365 -365
  191. package/skills/design/SKILL.md +590 -589
  192. package/skills/doc-processor/SKILL.md +254 -254
  193. package/skills/docs/SKILL.md +374 -374
  194. package/skills/docs-seeker/SKILL.md +178 -177
  195. package/skills/fix/SKILL.md +332 -330
  196. package/skills/git/SKILL.md +339 -339
  197. package/skills/hallucination-guard/SKILL.md +220 -219
  198. package/skills/incident/SKILL.md +254 -253
  199. package/skills/integrity-check/SKILL.md +169 -169
  200. package/skills/journal/SKILL.md +241 -240
  201. package/skills/launch/SKILL.md +344 -344
  202. package/skills/logic-guardian/SKILL.md +269 -251
  203. package/skills/marketing/SKILL.md +351 -289
  204. package/skills/mcp-builder/SKILL.md +425 -425
  205. package/skills/neural-memory/SKILL.md +359 -362
  206. package/skills/onboard/SKILL.md +432 -403
  207. package/skills/onboard/references/invariants-template.md +76 -0
  208. package/skills/onboard/scripts/detect-invariants.js +439 -0
  209. package/skills/onboard/scripts/inject-claude-md.js +150 -0
  210. package/skills/onboard/scripts/onboard-invariants.js +194 -0
  211. package/skills/perf/SKILL.md +347 -346
  212. package/skills/plan/SKILL.md +435 -428
  213. package/skills/preflight/SKILL.md +415 -415
  214. package/skills/problem-solver/SKILL.md +380 -284
  215. package/skills/rescue/SKILL.md +474 -474
  216. package/skills/research/SKILL.md +4 -0
  217. package/skills/retro/SKILL.md +3 -1
  218. package/skills/review/SKILL.md +614 -588
  219. package/skills/review-intake/SKILL.md +249 -249
  220. package/skills/safeguard/SKILL.md +200 -200
  221. package/skills/sast/SKILL.md +190 -190
  222. package/skills/scaffold/SKILL.md +328 -287
  223. package/skills/scope-guard/SKILL.md +183 -180
  224. package/skills/scout/SKILL.md +269 -263
  225. package/skills/sentinel/SKILL.md +384 -381
  226. package/skills/sentinel-env/SKILL.md +254 -254
  227. package/skills/sequential-thinking/SKILL.md +234 -234
  228. package/skills/session-bridge/SKILL.md +595 -543
  229. package/skills/session-bridge/scripts/load-invariants.js +397 -0
  230. package/skills/skill-forge/SKILL.md +581 -581
  231. package/skills/skill-router/SKILL.md +3 -0
  232. package/skills/slides/SKILL.md +19 -0
  233. package/skills/surgeon/SKILL.md +215 -215
  234. package/skills/team/SKILL.md +557 -537
  235. package/skills/test/SKILL.md +620 -614
  236. package/skills/trend-scout/SKILL.md +145 -145
  237. package/skills/verification/SKILL.md +334 -326
  238. package/skills/video-creator/SKILL.md +201 -201
  239. package/skills/watchdog/SKILL.md +168 -168
  240. package/skills/worktree/SKILL.md +140 -140
@@ -1,88 +1,88 @@
1
- ---
2
- name: "@rune/ai-ml"
3
- description: AI/ML integration patterns — LLM integration, RAG pipelines, embeddings, fine-tuning workflows, stateful AI agents, code execution sandboxes, web extraction, and deep research loops.
4
- metadata:
5
- author: runedev
6
- version: "0.4.0"
7
- layer: L4
8
- price: "$15"
9
- target: AI engineers
10
- format: split
11
- ---
12
-
13
- # @rune/ai-ml
14
-
15
- ## Purpose
16
-
17
- AI-powered features fail in predictable ways: LLM calls without retry logic that crash on rate limits, RAG pipelines that retrieve irrelevant chunks because the chunking strategy ignores document structure, embedding search that returns semantic matches with zero keyword overlap, fine-tuning runs that overfit because the eval set leaked into training data, AI agents that leak state across requests or lose progress on crashes, and code interpreters that execute untrusted LLM output without isolation. This pack codifies production patterns for each — from API client resilience to retrieval quality to model evaluation to agent state management to secure sandboxed execution — so AI features ship with the reliability of traditional software.
18
-
19
- ## Triggers
20
-
21
- - Auto-trigger: when `openai`, `anthropic`, `@langchain`, `pinecone`, `pgvector`, `embedding`, `llm` detected in dependencies or code
22
- - `/rune llm-integration` — audit or improve LLM API usage
23
- - `/rune rag-patterns` — build or audit RAG pipeline
24
- - `/rune embedding-search` — implement or optimize semantic search
25
- - `/rune fine-tuning-guide` — prepare and execute fine-tuning workflow
26
- - `/rune ai-agents` — design and build stateful AI agents
27
- - `/rune code-sandbox` — set up secure code execution for AI
28
- - `/rune web-extraction` — build structured data extraction from web pages
29
- - `/rune deep-research` — implement iterative AI research loops with convergence
30
- - Called by `cook` (L1) when AI/ML task detected
31
- - Called by `plan` (L2) when AI architecture decisions needed
32
-
33
- ## Skills Included
34
-
35
- | Skill | Model | Description |
36
- |-------|-------|-------------|
37
- | [llm-integration](skills/llm-integration.md) | sonnet | API client wrappers, streaming, structured output, retry + fallback chain, prompt versioning |
38
- | [rag-patterns](skills/rag-patterns.md) | sonnet | Document chunking, embedding generation, vector store setup, retrieval, reranking |
39
- | [embedding-search](skills/embedding-search.md) | sonnet | Semantic search, hybrid BM25 + vector, similarity thresholds, index optimization |
40
- | [fine-tuning-guide](skills/fine-tuning-guide.md) | sonnet | Dataset preparation, training config, evaluation metrics, deployment, A/B testing |
41
- | [llm-architect](skills/llm-architect.md) | opus | Model selection, prompt engineering, evaluation frameworks, cost optimization, guardrails |
42
- | [prompt-patterns](skills/prompt-patterns.md) | sonnet | Structured output, chain-of-thought, self-critique, ReAct, multi-turn memory management |
43
- | [ai-agents](skills/ai-agents.md) | sonnet | Stateful agents, RPC methods, scheduling, multi-agent coordination, MCP integration, HITL |
44
- | [code-sandbox](skills/code-sandbox.md) | sonnet | Container isolation, resource limits, timeout enforcement, stateful sessions, output capture |
45
- | [web-extraction](skills/web-extraction.md) | sonnet | Schema-driven extraction, anti-bot handling, prompt injection defense, multi-entity dedup |
46
- | [deep-research](skills/deep-research.md) | sonnet | Iterative research loop with convergence, source attribution, confidence scoring |
47
-
48
- ## Connections
49
-
50
- ```
51
- Calls → research (L3): lookup model documentation and best practices
52
- Calls → docs-seeker (L3): API reference for LLM providers
53
- Calls → verification (L3): validate pipeline correctness
54
- Calls → @rune/devops (L4): ai-agents → edge-serverless for agent deployment (Workers, Lambda)
55
- Calls → @rune/backend (L4): ai-agents → API patterns for agent endpoints and WebSocket handlers
56
- Calls → sentinel (L2): code-sandbox security audit on container isolation
57
- Called By ← cook (L1): when AI/ML task detected
58
- Called By ← plan (L2): when AI architecture decisions needed
59
- Called By ← review (L2): when AI code under review
60
- Called By ← mcp-builder (L2): ai-agents feeds MCP server patterns for agent-based MCP
61
- ai-agents → code-sandbox: agents use sandboxes for executing LLM-generated code safely
62
- code-sandbox → ai-agents: sandbox results feed back into agent state and conversation
63
- web-extraction → rag-patterns: extracted structured data feeds into RAG ingestion pipeline
64
- deep-research → web-extraction: research loop uses extraction for each discovered URL
65
- deep-research → embedding-search: relevance scoring uses embeddings for semantic similarity
66
- ```
67
-
68
- ## Sharp Edges
69
-
70
- - **Rate limits**: MUST implement exponential backoff retry on all LLM API calls — guaranteed at scale.
71
- - **Schema validation**: MUST validate LLM output with Zod/Pydantic — never trust raw text parsing.
72
- - **Eval leakage**: MUST separate training and evaluation datasets — leakage invalidates all metrics.
73
- - **Similarity thresholds**: MUST set thresholds on vector search — unrestricted results degrade quality.
74
- - **PII in embeddings**: MUST NOT embed sensitive data without consent — not easily deletable from vector stores.
75
- - **Embedding model pinning**: Pin model version in index metadata — dimension mismatch on upgrade is CRITICAL.
76
- - **Prompt injection**: Web pages may contain adversarial content targeting extraction LLMs — system prompt must block.
77
- - **Sandbox escape**: Use rootless Docker or gVisor for high-security code execution environments.
78
-
79
- ## Done When
80
-
81
- - LLM API client implemented with retry logic, exponential backoff, and structured output validation via Zod/Pydantic
82
- - RAG pipeline operational: chunking, embedding, vector store, retrieval, and reranking all configured and tested
83
- - Embedding index metadata includes pinned model version and dimension count to prevent upgrade mismatches
84
- - AI agent state persists across requests with no cross-session leakage and graceful crash recovery
85
-
86
- ## Cost Profile
87
-
88
- ~24,000–40,000 tokens per full pack run (all 10 skills). Individual skill: ~2,500–5,000 tokens. Sonnet default. Use haiku for code detection scans; escalate to sonnet for pipeline design, extraction strategy, and research loop orchestration.
1
+ ---
2
+ name: "@rune/ai-ml"
3
+ description: AI/ML integration patterns — LLM integration, RAG pipelines, embeddings, fine-tuning workflows, stateful AI agents, code execution sandboxes, web extraction, and deep research loops.
4
+ metadata:
5
+ author: runedev
6
+ version: "0.4.0"
7
+ layer: L4
8
+ price: "$15"
9
+ target: AI engineers
10
+ format: split
11
+ ---
12
+
13
+ # @rune/ai-ml
14
+
15
+ ## Purpose
16
+
17
+ AI-powered features fail in predictable ways: LLM calls without retry logic that crash on rate limits, RAG pipelines that retrieve irrelevant chunks because the chunking strategy ignores document structure, embedding search that returns semantic matches with zero keyword overlap, fine-tuning runs that overfit because the eval set leaked into training data, AI agents that leak state across requests or lose progress on crashes, and code interpreters that execute untrusted LLM output without isolation. This pack codifies production patterns for each — from API client resilience to retrieval quality to model evaluation to agent state management to secure sandboxed execution — so AI features ship with the reliability of traditional software.
18
+
19
+ ## Triggers
20
+
21
+ - Auto-trigger: when `openai`, `anthropic`, `@langchain`, `pinecone`, `pgvector`, `embedding`, `llm` detected in dependencies or code
22
+ - `/rune llm-integration` — audit or improve LLM API usage
23
+ - `/rune rag-patterns` — build or audit RAG pipeline
24
+ - `/rune embedding-search` — implement or optimize semantic search
25
+ - `/rune fine-tuning-guide` — prepare and execute fine-tuning workflow
26
+ - `/rune ai-agents` — design and build stateful AI agents
27
+ - `/rune code-sandbox` — set up secure code execution for AI
28
+ - `/rune web-extraction` — build structured data extraction from web pages
29
+ - `/rune deep-research` — implement iterative AI research loops with convergence
30
+ - Called by `cook` (L1) when AI/ML task detected
31
+ - Called by `plan` (L2) when AI architecture decisions needed
32
+
33
+ ## Skills Included
34
+
35
+ | Skill | Model | Description |
36
+ |-------|-------|-------------|
37
+ | [llm-integration](skills/llm-integration.md) | sonnet | API client wrappers, streaming, structured output, retry + fallback chain, prompt versioning |
38
+ | [rag-patterns](skills/rag-patterns.md) | sonnet | Document chunking, embedding generation, vector store setup, retrieval, reranking |
39
+ | [embedding-search](skills/embedding-search.md) | sonnet | Semantic search, hybrid BM25 + vector, similarity thresholds, index optimization |
40
+ | [fine-tuning-guide](skills/fine-tuning-guide.md) | sonnet | Dataset preparation, training config, evaluation metrics, deployment, A/B testing |
41
+ | [llm-architect](skills/llm-architect.md) | opus | Model selection, prompt engineering, evaluation frameworks, cost optimization, guardrails |
42
+ | [prompt-patterns](skills/prompt-patterns.md) | sonnet | Structured output, chain-of-thought, self-critique, ReAct, multi-turn memory management |
43
+ | [ai-agents](skills/ai-agents.md) | sonnet | Stateful agents, RPC methods, scheduling, multi-agent coordination, MCP integration, HITL |
44
+ | [code-sandbox](skills/code-sandbox.md) | sonnet | Container isolation, resource limits, timeout enforcement, stateful sessions, output capture |
45
+ | [web-extraction](skills/web-extraction.md) | sonnet | Schema-driven extraction, anti-bot handling, prompt injection defense, multi-entity dedup |
46
+ | [deep-research](skills/deep-research.md) | sonnet | Iterative research loop with convergence, source attribution, confidence scoring |
47
+
48
+ ## Connections
49
+
50
+ ```
51
+ Calls → research (L3): lookup model documentation and best practices
52
+ Calls → docs-seeker (L3): API reference for LLM providers
53
+ Calls → verification (L3): validate pipeline correctness
54
+ Calls → @rune/devops (L4): ai-agents → edge-serverless for agent deployment (Workers, Lambda)
55
+ Calls → @rune/backend (L4): ai-agents → API patterns for agent endpoints and WebSocket handlers
56
+ Calls → sentinel (L2): code-sandbox security audit on container isolation
57
+ Called By ← cook (L1): when AI/ML task detected
58
+ Called By ← plan (L2): when AI architecture decisions needed
59
+ Called By ← review (L2): when AI code under review
60
+ Called By ← mcp-builder (L2): ai-agents feeds MCP server patterns for agent-based MCP
61
+ ai-agents → code-sandbox: agents use sandboxes for executing LLM-generated code safely
62
+ code-sandbox → ai-agents: sandbox results feed back into agent state and conversation
63
+ web-extraction → rag-patterns: extracted structured data feeds into RAG ingestion pipeline
64
+ deep-research → web-extraction: research loop uses extraction for each discovered URL
65
+ deep-research → embedding-search: relevance scoring uses embeddings for semantic similarity
66
+ ```
67
+
68
+ ## Sharp Edges
69
+
70
+ - **Rate limits**: MUST implement exponential backoff retry on all LLM API calls — guaranteed at scale.
71
+ - **Schema validation**: MUST validate LLM output with Zod/Pydantic — never trust raw text parsing.
72
+ - **Eval leakage**: MUST separate training and evaluation datasets — leakage invalidates all metrics.
73
+ - **Similarity thresholds**: MUST set thresholds on vector search — unrestricted results degrade quality.
74
+ - **PII in embeddings**: MUST NOT embed sensitive data without consent — not easily deletable from vector stores.
75
+ - **Embedding model pinning**: Pin model version in index metadata — dimension mismatch on upgrade is CRITICAL.
76
+ - **Prompt injection**: Web pages may contain adversarial content targeting extraction LLMs — system prompt must block.
77
+ - **Sandbox escape**: Use rootless Docker or gVisor for high-security code execution environments.
78
+
79
+ ## Done When
80
+
81
+ - LLM API client implemented with retry logic, exponential backoff, and structured output validation via Zod/Pydantic
82
+ - RAG pipeline operational: chunking, embedding, vector store, retrieval, and reranking all configured and tested
83
+ - Embedding index metadata includes pinned model version and dimension count to prevent upgrade mismatches
84
+ - AI agent state persists across requests with no cross-session leakage and graceful crash recovery
85
+
86
+ ## Cost Profile
87
+
88
+ ~24,000–40,000 tokens per full pack run (all 10 skills). Individual skill: ~2,500–5,000 tokens. Sonnet default. Use haiku for code detection scans; escalate to sonnet for pipeline design, extraction strategy, and research loop orchestration.
@@ -1,172 +1,172 @@
1
- ---
2
- name: "ai-agents"
3
- pack: "@rune/ai-ml"
4
- description: "Stateful AI agent architecture — persistent state, callable RPC methods, scheduling, multi-agent coordination, MCP server integration, and real-time client communication via WebSocket."
5
- model: sonnet
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # ai-agents
10
-
11
- Stateful AI agent architecture — persistent state, callable RPC methods, scheduling, multi-agent coordination, MCP server integration, and real-time client communication via WebSocket. Covers agent lifecycle, state management patterns, tool registration, human-in-the-loop approval flows, and durable workflow orchestration for long-running agent tasks.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Classify agent type**
16
- Identify what the agent needs to do and map to an architecture:
17
-
18
- | Agent Type | Key Characteristics | Platform Options |
19
- |---|---|---|
20
- | Stateless tool-caller | Single request → tool calls → response. No memory between requests. | Any LLM API + function calling |
21
- | Conversational with memory | Multi-turn dialogue. Needs chat history persistence. | Session store (Redis, KV) + LLM |
22
- | Stateful autonomous | Persistent state, scheduled tasks, reacts to events. Long-lived. | Cloudflare Agents SDK, LangGraph, CrewAI |
23
- | Multi-agent coordinator | Multiple specialized agents collaborating on a task. | LangGraph, AutoGen, custom orchestrator |
24
- | MCP server | Exposes tools/resources to any MCP-compatible client. | Cloudflare McpAgent, custom MCP server |
25
-
26
- **Step 2 — Design state management**
27
- For stateful agents, define the state contract:
28
-
29
- ```typescript
30
- // State must be serializable (JSON-safe) — no functions, no circular refs
31
- interface AgentState {
32
- // Domain state
33
- conversations: ConversationEntry[];
34
- preferences: Record<string, string>;
35
- taskQueue: ScheduledTask[];
36
-
37
- // Metadata
38
- createdAt: string;
39
- lastActiveAt: string;
40
- version: number;
41
- }
42
-
43
- // State validation — reject invalid transitions
44
- function validateStateChange(current: AgentState, next: AgentState): void {
45
- if (next.version < current.version) {
46
- throw new Error('State version cannot decrease — concurrent modification detected');
47
- }
48
- if (next.conversations.length > 10_000) {
49
- throw new Error('Conversation limit exceeded — archive old entries first');
50
- }
51
- }
52
- ```
53
-
54
- **Step 3 — Implement tool registration**
55
- Define agent capabilities as typed, callable methods:
56
-
57
- ```typescript
58
- // Tools as typed RPC methods (Cloudflare Agents SDK pattern)
59
- import { Agent, callable } from 'agents';
60
-
61
- export class ResearchAgent extends Agent<Env, ResearchState> {
62
- initialState: ResearchState = { findings: [], status: 'idle' };
63
-
64
- @callable()
65
- async search(query: string): Promise<SearchResult[]> {
66
- this.setState({ ...this.state, status: 'searching' });
67
- const results = await this.env.AI.run('@cf/meta/llama-3-8b-instruct', {
68
- prompt: `Search for: ${query}`,
69
- });
70
- const findings = parseResults(results);
71
- this.setState({
72
- ...this.state,
73
- findings: [...this.state.findings, ...findings],
74
- status: 'idle',
75
- });
76
- return findings;
77
- }
78
-
79
- @callable()
80
- async summarize(): Promise<string> {
81
- if (this.state.findings.length === 0) {
82
- throw new Error('No findings to summarize — run search first');
83
- }
84
- return generateSummary(this.state.findings);
85
- }
86
- }
87
- ```
88
-
89
- **Step 4 — Add scheduling and durability**
90
- For agents that need to perform work on a schedule or survive restarts:
91
-
92
- ```typescript
93
- // Scheduled tasks — one-time, recurring, and cron
94
- @callable()
95
- async scheduleDigest(userId: string) {
96
- // Daily digest at 9 AM
97
- await this.schedule('0 9 * * *', 'sendDigest', { userId });
98
-
99
- // One-time reminder in 1 hour
100
- await this.schedule(3600, 'sendReminder', { userId, message: 'Check results' });
101
-
102
- // Recurring every 30 minutes
103
- await this.scheduleEvery(1800, 'pollDataSource');
104
- }
105
-
106
- // Handler runs when scheduled time arrives — even if agent was hibernated
107
- async onScheduledTask(task: ScheduledTask) {
108
- switch (task.type) {
109
- case 'sendDigest':
110
- await this.compileAndSendDigest(task.payload.userId);
111
- break;
112
- case 'pollDataSource':
113
- const newData = await fetchLatest();
114
- if (newData.length > 0) {
115
- this.setState({ ...this.state, lastPoll: Date.now(), data: newData });
116
- }
117
- break;
118
- }
119
- }
120
- ```
121
-
122
- **Step 5 — Human-in-the-loop patterns**
123
- For agents that need approval before taking high-impact actions:
124
-
125
- ```typescript
126
- // Approval flow — agent pauses, human approves, agent resumes
127
- interface PendingApproval {
128
- id: string;
129
- action: string;
130
- params: Record<string, unknown>;
131
- requestedAt: string;
132
- status: 'pending' | 'approved' | 'rejected';
133
- }
134
-
135
- @callable()
136
- async requestApproval(action: string, params: Record<string, unknown>): Promise<string> {
137
- const approval: PendingApproval = {
138
- id: crypto.randomUUID(),
139
- action,
140
- params,
141
- requestedAt: new Date().toISOString(),
142
- status: 'pending',
143
- };
144
- this.setState({
145
- ...this.state,
146
- pendingApprovals: [...this.state.pendingApprovals, approval],
147
- });
148
- // Client receives state update via WebSocket → shows approval UI
149
- return approval.id;
150
- }
151
-
152
- @callable()
153
- async resolveApproval(id: string, decision: 'approved' | 'rejected') {
154
- const updated = this.state.pendingApprovals.map(a =>
155
- a.id === id ? { ...a, status: decision } : a
156
- );
157
- this.setState({ ...this.state, pendingApprovals: updated });
158
- if (decision === 'approved') {
159
- const approval = updated.find(a => a.id === id)!;
160
- await this.executeAction(approval.action, approval.params);
161
- }
162
- }
163
- ```
164
-
165
- #### Sharp Edges
166
-
167
- | Failure Mode | Mitigation |
168
- |---|---|
169
- | State grows unbounded (conversation history, logs) | Implement max size limits with archival; prune old entries on state update |
170
- | Concurrent state mutations from multiple clients | Use version counter in state; reject updates with stale version |
171
- | Agent crashes mid-workflow, loses progress | Use durable workflows (Cloudflare Workflows, Temporal) for multi-step tasks — each step is persisted |
172
- | Scheduled tasks pile up during agent hibernation | Deduplicate on wake-up; use idempotency keys for task handlers |
1
+ ---
2
+ name: "ai-agents"
3
+ pack: "@rune/ai-ml"
4
+ description: "Stateful AI agent architecture — persistent state, callable RPC methods, scheduling, multi-agent coordination, MCP server integration, and real-time client communication via WebSocket."
5
+ model: sonnet
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # ai-agents
10
+
11
+ Stateful AI agent architecture — persistent state, callable RPC methods, scheduling, multi-agent coordination, MCP server integration, and real-time client communication via WebSocket. Covers agent lifecycle, state management patterns, tool registration, human-in-the-loop approval flows, and durable workflow orchestration for long-running agent tasks.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Classify agent type**
16
+ Identify what the agent needs to do and map to an architecture:
17
+
18
+ | Agent Type | Key Characteristics | Platform Options |
19
+ |---|---|---|
20
+ | Stateless tool-caller | Single request → tool calls → response. No memory between requests. | Any LLM API + function calling |
21
+ | Conversational with memory | Multi-turn dialogue. Needs chat history persistence. | Session store (Redis, KV) + LLM |
22
+ | Stateful autonomous | Persistent state, scheduled tasks, reacts to events. Long-lived. | Cloudflare Agents SDK, LangGraph, CrewAI |
23
+ | Multi-agent coordinator | Multiple specialized agents collaborating on a task. | LangGraph, AutoGen, custom orchestrator |
24
+ | MCP server | Exposes tools/resources to any MCP-compatible client. | Cloudflare McpAgent, custom MCP server |
25
+
26
+ **Step 2 — Design state management**
27
+ For stateful agents, define the state contract:
28
+
29
+ ```typescript
30
+ // State must be serializable (JSON-safe) — no functions, no circular refs
31
+ interface AgentState {
32
+ // Domain state
33
+ conversations: ConversationEntry[];
34
+ preferences: Record<string, string>;
35
+ taskQueue: ScheduledTask[];
36
+
37
+ // Metadata
38
+ createdAt: string;
39
+ lastActiveAt: string;
40
+ version: number;
41
+ }
42
+
43
+ // State validation — reject invalid transitions
44
+ function validateStateChange(current: AgentState, next: AgentState): void {
45
+ if (next.version < current.version) {
46
+ throw new Error('State version cannot decrease — concurrent modification detected');
47
+ }
48
+ if (next.conversations.length > 10_000) {
49
+ throw new Error('Conversation limit exceeded — archive old entries first');
50
+ }
51
+ }
52
+ ```
53
+
54
+ **Step 3 — Implement tool registration**
55
+ Define agent capabilities as typed, callable methods:
56
+
57
+ ```typescript
58
+ // Tools as typed RPC methods (Cloudflare Agents SDK pattern)
59
+ import { Agent, callable } from 'agents';
60
+
61
+ export class ResearchAgent extends Agent<Env, ResearchState> {
62
+ initialState: ResearchState = { findings: [], status: 'idle' };
63
+
64
+ @callable()
65
+ async search(query: string): Promise<SearchResult[]> {
66
+ this.setState({ ...this.state, status: 'searching' });
67
+ const results = await this.env.AI.run('@cf/meta/llama-3-8b-instruct', {
68
+ prompt: `Search for: ${query}`,
69
+ });
70
+ const findings = parseResults(results);
71
+ this.setState({
72
+ ...this.state,
73
+ findings: [...this.state.findings, ...findings],
74
+ status: 'idle',
75
+ });
76
+ return findings;
77
+ }
78
+
79
+ @callable()
80
+ async summarize(): Promise<string> {
81
+ if (this.state.findings.length === 0) {
82
+ throw new Error('No findings to summarize — run search first');
83
+ }
84
+ return generateSummary(this.state.findings);
85
+ }
86
+ }
87
+ ```
88
+
89
+ **Step 4 — Add scheduling and durability**
90
+ For agents that need to perform work on a schedule or survive restarts:
91
+
92
+ ```typescript
93
+ // Scheduled tasks — one-time, recurring, and cron
94
+ @callable()
95
+ async scheduleDigest(userId: string) {
96
+ // Daily digest at 9 AM
97
+ await this.schedule('0 9 * * *', 'sendDigest', { userId });
98
+
99
+ // One-time reminder in 1 hour
100
+ await this.schedule(3600, 'sendReminder', { userId, message: 'Check results' });
101
+
102
+ // Recurring every 30 minutes
103
+ await this.scheduleEvery(1800, 'pollDataSource');
104
+ }
105
+
106
+ // Handler runs when scheduled time arrives — even if agent was hibernated
107
+ async onScheduledTask(task: ScheduledTask) {
108
+ switch (task.type) {
109
+ case 'sendDigest':
110
+ await this.compileAndSendDigest(task.payload.userId);
111
+ break;
112
+ case 'pollDataSource':
113
+ const newData = await fetchLatest();
114
+ if (newData.length > 0) {
115
+ this.setState({ ...this.state, lastPoll: Date.now(), data: newData });
116
+ }
117
+ break;
118
+ }
119
+ }
120
+ ```
121
+
122
+ **Step 5 — Human-in-the-loop patterns**
123
+ For agents that need approval before taking high-impact actions:
124
+
125
+ ```typescript
126
+ // Approval flow — agent pauses, human approves, agent resumes
127
+ interface PendingApproval {
128
+ id: string;
129
+ action: string;
130
+ params: Record<string, unknown>;
131
+ requestedAt: string;
132
+ status: 'pending' | 'approved' | 'rejected';
133
+ }
134
+
135
+ @callable()
136
+ async requestApproval(action: string, params: Record<string, unknown>): Promise<string> {
137
+ const approval: PendingApproval = {
138
+ id: crypto.randomUUID(),
139
+ action,
140
+ params,
141
+ requestedAt: new Date().toISOString(),
142
+ status: 'pending',
143
+ };
144
+ this.setState({
145
+ ...this.state,
146
+ pendingApprovals: [...this.state.pendingApprovals, approval],
147
+ });
148
+ // Client receives state update via WebSocket → shows approval UI
149
+ return approval.id;
150
+ }
151
+
152
+ @callable()
153
+ async resolveApproval(id: string, decision: 'approved' | 'rejected') {
154
+ const updated = this.state.pendingApprovals.map(a =>
155
+ a.id === id ? { ...a, status: decision } : a
156
+ );
157
+ this.setState({ ...this.state, pendingApprovals: updated });
158
+ if (decision === 'approved') {
159
+ const approval = updated.find(a => a.id === id)!;
160
+ await this.executeAction(approval.action, approval.params);
161
+ }
162
+ }
163
+ ```
164
+
165
+ #### Sharp Edges
166
+
167
+ | Failure Mode | Mitigation |
168
+ |---|---|
169
+ | State grows unbounded (conversation history, logs) | Implement max size limits with archival; prune old entries on state update |
170
+ | Concurrent state mutations from multiple clients | Use version counter in state; reject updates with stale version |
171
+ | Agent crashes mid-workflow, loses progress | Use durable workflows (Cloudflare Workflows, Temporal) for multi-step tasks — each step is persisted |
172
+ | Scheduled tasks pile up during agent hibernation | Deduplicate on wake-up; use idempotency keys for task handlers |