@rune-kit/rune 2.10.0 → 2.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +65 -6
  3. package/commands/rune.md +168 -168
  4. package/compiler/__tests__/detect-invariants.test.js +136 -0
  5. package/compiler/__tests__/doctor-mesh.test.js +229 -0
  6. package/compiler/__tests__/hook-dispatch.test.js +91 -0
  7. package/compiler/__tests__/hooks-antigravity.test.js +118 -0
  8. package/compiler/__tests__/hooks-cursor.test.js +139 -0
  9. package/compiler/__tests__/hooks-install.test.js +305 -0
  10. package/compiler/__tests__/hooks-merge.test.js +204 -0
  11. package/compiler/__tests__/hooks-tiers.test.js +519 -0
  12. package/compiler/__tests__/hooks-windsurf.test.js +115 -0
  13. package/compiler/__tests__/inject-claude-md.test.js +152 -0
  14. package/compiler/__tests__/load-invariants.test.js +408 -0
  15. package/compiler/__tests__/onboard-invariants.test.js +240 -0
  16. package/compiler/adapters/hooks/antigravity.js +140 -0
  17. package/compiler/adapters/hooks/claude.js +166 -0
  18. package/compiler/adapters/hooks/cursor.js +191 -0
  19. package/compiler/adapters/hooks/index.js +82 -0
  20. package/compiler/adapters/hooks/tier-emitter.js +182 -0
  21. package/compiler/adapters/hooks/windsurf.js +202 -0
  22. package/compiler/bin/rune.js +196 -6
  23. package/compiler/commands/hook-dispatch.js +87 -0
  24. package/compiler/commands/hooks/install.js +120 -0
  25. package/compiler/commands/hooks/merge.js +211 -0
  26. package/compiler/commands/hooks/presets.js +116 -0
  27. package/compiler/commands/hooks/status.js +112 -0
  28. package/compiler/commands/hooks/tiers.js +221 -0
  29. package/compiler/commands/hooks/uninstall.js +94 -0
  30. package/compiler/doctor.js +236 -0
  31. package/contexts/dev.md +34 -34
  32. package/contexts/research.md +43 -43
  33. package/contexts/review.md +55 -55
  34. package/extensions/ai-ml/PACK.md +88 -88
  35. package/extensions/ai-ml/skills/ai-agents.md +172 -172
  36. package/extensions/ai-ml/skills/code-sandbox.md +187 -187
  37. package/extensions/ai-ml/skills/deep-research.md +146 -146
  38. package/extensions/ai-ml/skills/embedding-search.md +66 -66
  39. package/extensions/ai-ml/skills/fine-tuning-guide.md +74 -74
  40. package/extensions/ai-ml/skills/llm-architect.md +125 -125
  41. package/extensions/ai-ml/skills/llm-integration.md +64 -64
  42. package/extensions/ai-ml/skills/prompt-patterns.md +72 -72
  43. package/extensions/ai-ml/skills/rag-patterns.md +66 -66
  44. package/extensions/ai-ml/skills/web-extraction.md +114 -114
  45. package/extensions/analytics/PACK.md +92 -92
  46. package/extensions/analytics/skills/ab-testing.md +72 -72
  47. package/extensions/analytics/skills/dashboard-patterns.md +83 -83
  48. package/extensions/analytics/skills/data-validation.md +68 -68
  49. package/extensions/analytics/skills/funnel-analysis.md +81 -81
  50. package/extensions/analytics/skills/sql-patterns.md +57 -57
  51. package/extensions/analytics/skills/statistical-analysis.md +79 -79
  52. package/extensions/analytics/skills/tracking-setup.md +71 -71
  53. package/extensions/backend/PACK.md +104 -104
  54. package/extensions/backend/skills/api-patterns.md +84 -84
  55. package/extensions/backend/skills/async-pipeline.md +193 -193
  56. package/extensions/backend/skills/auth-patterns.md +97 -97
  57. package/extensions/backend/skills/background-jobs.md +133 -133
  58. package/extensions/backend/skills/caching-patterns.md +108 -108
  59. package/extensions/backend/skills/cli-generation.md +133 -133
  60. package/extensions/backend/skills/database-patterns.md +87 -87
  61. package/extensions/backend/skills/middleware-patterns.md +104 -104
  62. package/extensions/chrome-ext/PACK.md +93 -93
  63. package/extensions/chrome-ext/skills/cws-preflight.md +143 -143
  64. package/extensions/chrome-ext/skills/cws-publish.md +104 -104
  65. package/extensions/chrome-ext/skills/ext-ai-integration.md +251 -251
  66. package/extensions/chrome-ext/skills/ext-messaging.md +139 -139
  67. package/extensions/chrome-ext/skills/ext-storage.md +133 -133
  68. package/extensions/chrome-ext/skills/mv3-scaffold.md +164 -164
  69. package/extensions/content/PACK.md +96 -96
  70. package/extensions/content/skills/blog-patterns.md +88 -88
  71. package/extensions/content/skills/cms-integration.md +131 -131
  72. package/extensions/content/skills/content-scoring.md +107 -107
  73. package/extensions/content/skills/i18n.md +83 -83
  74. package/extensions/content/skills/mdx-authoring.md +137 -137
  75. package/extensions/content/skills/reference.md +1014 -1014
  76. package/extensions/content/skills/seo-patterns.md +67 -67
  77. package/extensions/content/skills/video-repurpose.md +153 -153
  78. package/extensions/devops/PACK.md +101 -101
  79. package/extensions/devops/skills/chaos-testing.md +67 -67
  80. package/extensions/devops/skills/ci-cd.md +75 -75
  81. package/extensions/devops/skills/docker.md +58 -58
  82. package/extensions/devops/skills/edge-serverless.md +163 -163
  83. package/extensions/devops/skills/infra-as-code.md +158 -158
  84. package/extensions/devops/skills/kubernetes.md +110 -110
  85. package/extensions/devops/skills/monitoring.md +57 -57
  86. package/extensions/devops/skills/server-setup.md +64 -64
  87. package/extensions/devops/skills/ssl-domain.md +42 -42
  88. package/extensions/ecommerce/PACK.md +116 -116
  89. package/extensions/ecommerce/skills/cart-system.md +79 -79
  90. package/extensions/ecommerce/skills/inventory-mgmt.md +102 -102
  91. package/extensions/ecommerce/skills/order-management.md +126 -126
  92. package/extensions/ecommerce/skills/payment-integration.md +472 -472
  93. package/extensions/ecommerce/skills/shopify-dev.md +69 -69
  94. package/extensions/ecommerce/skills/subscription-billing.md +93 -93
  95. package/extensions/ecommerce/skills/tax-compliance.md +117 -117
  96. package/extensions/gamedev/PACK.md +142 -142
  97. package/extensions/gamedev/skills/asset-pipeline.md +74 -74
  98. package/extensions/gamedev/skills/audio-system.md +129 -129
  99. package/extensions/gamedev/skills/camera-system.md +87 -87
  100. package/extensions/gamedev/skills/ecs.md +98 -98
  101. package/extensions/gamedev/skills/game-loops.md +72 -72
  102. package/extensions/gamedev/skills/input-system.md +199 -199
  103. package/extensions/gamedev/skills/multiplayer.md +180 -180
  104. package/extensions/gamedev/skills/particles.md +105 -105
  105. package/extensions/gamedev/skills/physics-engine.md +89 -89
  106. package/extensions/gamedev/skills/scene-management.md +146 -146
  107. package/extensions/gamedev/skills/threejs-patterns.md +90 -90
  108. package/extensions/gamedev/skills/webgl.md +71 -71
  109. package/extensions/mobile/PACK.md +106 -106
  110. package/extensions/mobile/skills/app-store-connect.md +152 -152
  111. package/extensions/mobile/skills/app-store-prep.md +66 -66
  112. package/extensions/mobile/skills/deep-linking.md +109 -109
  113. package/extensions/mobile/skills/flutter.md +60 -60
  114. package/extensions/mobile/skills/ios-build-pipeline.md +142 -142
  115. package/extensions/mobile/skills/native-bridge.md +66 -66
  116. package/extensions/mobile/skills/ota-updates.md +97 -97
  117. package/extensions/mobile/skills/push-notifications.md +111 -111
  118. package/extensions/mobile/skills/react-native.md +82 -82
  119. package/extensions/saas/PACK.md +116 -116
  120. package/extensions/saas/skills/billing-integration.md +200 -200
  121. package/extensions/saas/skills/feature-flags.md +130 -130
  122. package/extensions/saas/skills/multi-tenant.md +103 -103
  123. package/extensions/saas/skills/onboarding-flow.md +139 -139
  124. package/extensions/saas/skills/subscription-flow.md +95 -95
  125. package/extensions/saas/skills/team-management.md +144 -144
  126. package/extensions/security/PACK.md +99 -99
  127. package/extensions/security/skills/api-security.md +140 -140
  128. package/extensions/security/skills/compliance.md +68 -68
  129. package/extensions/security/skills/owasp-audit.md +64 -64
  130. package/extensions/security/skills/pentest-patterns.md +77 -77
  131. package/extensions/security/skills/secret-mgmt.md +65 -65
  132. package/extensions/security/skills/supply-chain.md +65 -65
  133. package/extensions/trading/PACK.md +80 -80
  134. package/extensions/trading/skills/chart-components.md +55 -55
  135. package/extensions/trading/skills/experiment-loop.md +125 -125
  136. package/extensions/trading/skills/fintech-patterns.md +47 -47
  137. package/extensions/trading/skills/indicator-library.md +58 -58
  138. package/extensions/trading/skills/quant-analysis.md +111 -111
  139. package/extensions/trading/skills/realtime-data.md +58 -58
  140. package/extensions/trading/skills/trade-logic.md +104 -104
  141. package/extensions/ui/PACK.md +130 -130
  142. package/extensions/ui/skills/a11y-audit.md +91 -91
  143. package/extensions/ui/skills/animation-patterns.md +127 -127
  144. package/extensions/ui/skills/component-patterns.md +100 -100
  145. package/extensions/ui/skills/design-decision.md +108 -108
  146. package/extensions/ui/skills/design-system.md +68 -68
  147. package/extensions/ui/skills/landing-patterns.md +155 -155
  148. package/extensions/ui/skills/palette-picker.md +173 -173
  149. package/extensions/ui/skills/react-health.md +90 -90
  150. package/extensions/ui/skills/type-system.md +125 -125
  151. package/extensions/ui/skills/web-vitals.md +153 -153
  152. package/extensions/zalo/PACK.md +145 -145
  153. package/extensions/zalo/skills/zalo-oa-mcp.md +317 -317
  154. package/extensions/zalo/skills/zalo-oa-messaging.md +429 -429
  155. package/extensions/zalo/skills/zalo-oa-setup.md +236 -236
  156. package/extensions/zalo/skills/zalo-oa-webhook.md +189 -189
  157. package/extensions/zalo/skills/zalo-personal-messaging.md +194 -194
  158. package/extensions/zalo/skills/zalo-personal-setup.md +153 -153
  159. package/extensions/zalo/skills/zalo-rate-guard.md +219 -219
  160. package/hooks/auto-format/index.cjs +48 -48
  161. package/hooks/hooks.json +111 -111
  162. package/hooks/post-session-reflect/index.cjs +189 -189
  163. package/hooks/pre-compact/index.cjs +95 -95
  164. package/hooks/run-hook.cmd +1 -1
  165. package/hooks/secrets-scan/index.cjs +100 -100
  166. package/hooks/session-start/index.cjs +71 -71
  167. package/hooks/typecheck/index.cjs +65 -65
  168. package/package.json +63 -63
  169. package/references/ui-pro-max-data/LICENSE-UI-PRO-MAX +21 -21
  170. package/references/ui-pro-max-data/charts.csv +26 -26
  171. package/references/ui-pro-max-data/colors.csv +161 -161
  172. package/references/ui-pro-max-data/styles.csv +68 -68
  173. package/references/ui-pro-max-data/typography.csv +74 -74
  174. package/references/ui-pro-max-data/ui-reasoning.csv +162 -162
  175. package/references/ui-pro-max-data/ux-guidelines.csv +99 -99
  176. package/skills/adversary/SKILL.md +283 -283
  177. package/skills/asset-creator/SKILL.md +157 -157
  178. package/skills/audit/SKILL.md +147 -2
  179. package/skills/autopsy/SKILL.md +335 -335
  180. package/skills/ba/SKILL.md +85 -1
  181. package/skills/brainstorm/SKILL.md +380 -342
  182. package/skills/browser-pilot/SKILL.md +169 -168
  183. package/skills/constraint-check/SKILL.md +165 -165
  184. package/skills/context-engine/SKILL.md +408 -404
  185. package/skills/cook/SKILL.md +917 -863
  186. package/skills/db/SKILL.md +273 -273
  187. package/skills/debug/SKILL.md +465 -465
  188. package/skills/dependency-doctor/SKILL.md +265 -235
  189. package/skills/deploy/SKILL.md +274 -231
  190. package/skills/design/DESIGN-REFERENCE.md +365 -365
  191. package/skills/design/SKILL.md +590 -589
  192. package/skills/doc-processor/SKILL.md +254 -254
  193. package/skills/docs/SKILL.md +374 -374
  194. package/skills/docs-seeker/SKILL.md +178 -177
  195. package/skills/fix/SKILL.md +332 -330
  196. package/skills/git/SKILL.md +339 -339
  197. package/skills/hallucination-guard/SKILL.md +220 -219
  198. package/skills/incident/SKILL.md +254 -253
  199. package/skills/integrity-check/SKILL.md +169 -169
  200. package/skills/journal/SKILL.md +241 -240
  201. package/skills/launch/SKILL.md +344 -344
  202. package/skills/logic-guardian/SKILL.md +269 -251
  203. package/skills/marketing/SKILL.md +351 -289
  204. package/skills/mcp-builder/SKILL.md +425 -425
  205. package/skills/neural-memory/SKILL.md +359 -362
  206. package/skills/onboard/SKILL.md +432 -403
  207. package/skills/onboard/references/invariants-template.md +76 -0
  208. package/skills/onboard/scripts/detect-invariants.js +439 -0
  209. package/skills/onboard/scripts/inject-claude-md.js +150 -0
  210. package/skills/onboard/scripts/onboard-invariants.js +194 -0
  211. package/skills/perf/SKILL.md +347 -346
  212. package/skills/plan/SKILL.md +435 -428
  213. package/skills/preflight/SKILL.md +415 -415
  214. package/skills/problem-solver/SKILL.md +380 -284
  215. package/skills/rescue/SKILL.md +474 -474
  216. package/skills/research/SKILL.md +4 -0
  217. package/skills/retro/SKILL.md +3 -1
  218. package/skills/review/SKILL.md +614 -588
  219. package/skills/review-intake/SKILL.md +249 -249
  220. package/skills/safeguard/SKILL.md +200 -200
  221. package/skills/sast/SKILL.md +190 -190
  222. package/skills/scaffold/SKILL.md +328 -287
  223. package/skills/scope-guard/SKILL.md +183 -180
  224. package/skills/scout/SKILL.md +269 -263
  225. package/skills/sentinel/SKILL.md +384 -381
  226. package/skills/sentinel-env/SKILL.md +254 -254
  227. package/skills/sequential-thinking/SKILL.md +234 -234
  228. package/skills/session-bridge/SKILL.md +595 -543
  229. package/skills/session-bridge/scripts/load-invariants.js +397 -0
  230. package/skills/skill-forge/SKILL.md +581 -581
  231. package/skills/skill-router/SKILL.md +3 -0
  232. package/skills/slides/SKILL.md +19 -0
  233. package/skills/surgeon/SKILL.md +215 -215
  234. package/skills/team/SKILL.md +557 -537
  235. package/skills/test/SKILL.md +620 -614
  236. package/skills/trend-scout/SKILL.md +145 -145
  237. package/skills/verification/SKILL.md +334 -326
  238. package/skills/video-creator/SKILL.md +201 -201
  239. package/skills/watchdog/SKILL.md +168 -168
  240. package/skills/worktree/SKILL.md +140 -140
@@ -1,66 +1,66 @@
1
- ---
2
- name: "embedding-search"
3
- pack: "@rune/ai-ml"
4
- description: "Embedding-based search — semantic search, hybrid search (BM25 + vector), similarity thresholds, index optimization."
5
- model: sonnet
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # embedding-search
10
-
11
- Embedding-based search — semantic search, hybrid search (BM25 + vector), similarity thresholds, index optimization.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Detect search implementation**
16
- Use Grep to find search code: `similarity_search`, `vector_search`, `fts`, `tsvector`, `BM25`. Read search handlers to understand: query flow, ranking strategy, and result formatting.
17
-
18
- **Step 2 — Audit search quality**
19
- Check for: pure vector search without keyword fallback (misses exact matches), no similarity threshold (returns irrelevant results at low scores), missing query embedding cache (repeated queries re-embed), no hybrid scoring (BM25 for exact + vector for semantic), and unoptimized vector index (HNSW parameters not tuned).
20
-
21
- **Step 3 — Emit hybrid search**
22
- Emit: combined BM25 + vector search with reciprocal rank fusion, similarity threshold filtering, query embedding cache, and HNSW index tuning.
23
-
24
- #### Example
25
-
26
- ```typescript
27
- // Hybrid search — BM25 + vector with reciprocal rank fusion
28
- async function hybridSearch(query: string, limit = 10) {
29
- // Parallel: keyword (BM25) + semantic (vector)
30
- const [keywordResults, vectorResults] = await Promise.all([
31
- db.execute(sql`
32
- SELECT id, content, ts_rank(search_vector, plainto_tsquery(${query})) AS bm25_score
33
- FROM documents
34
- WHERE search_vector @@ plainto_tsquery(${query})
35
- ORDER BY bm25_score DESC LIMIT ${limit * 2}
36
- `),
37
- db.execute(sql`
38
- SELECT id, content, 1 - (embedding <=> ${await getEmbedding(query)}) AS vector_score
39
- FROM documents
40
- ORDER BY embedding <=> ${await getEmbedding(query)}
41
- LIMIT ${limit * 2}
42
- `),
43
- ]);
44
-
45
- // Reciprocal rank fusion (k=60)
46
- const scores = new Map<string, number>();
47
- const K = 60;
48
- keywordResults.forEach((r, i) => scores.set(r.id, (scores.get(r.id) || 0) + 1 / (K + i + 1)));
49
- vectorResults.forEach((r, i) => scores.set(r.id, (scores.get(r.id) || 0) + 1 / (K + i + 1)));
50
-
51
- return [...scores.entries()]
52
- .sort((a, b) => b[1] - a[1])
53
- .slice(0, limit)
54
- .filter(([_, score]) => score > 0.01); // threshold
55
- }
56
-
57
- // Embedding cache (avoid re-embedding repeated queries)
58
- const embeddingCache = new Map<string, number[]>();
59
- async function getEmbedding(text: string): Promise<number[]> {
60
- const cached = embeddingCache.get(text);
61
- if (cached) return cached;
62
- const { data } = await openai.embeddings.create({ model: 'text-embedding-3-small', input: text });
63
- embeddingCache.set(text, data[0].embedding);
64
- return data[0].embedding;
65
- }
66
- ```
1
+ ---
2
+ name: "embedding-search"
3
+ pack: "@rune/ai-ml"
4
+ description: "Embedding-based search — semantic search, hybrid search (BM25 + vector), similarity thresholds, index optimization."
5
+ model: sonnet
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # embedding-search
10
+
11
+ Embedding-based search — semantic search, hybrid search (BM25 + vector), similarity thresholds, index optimization.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Detect search implementation**
16
+ Use Grep to find search code: `similarity_search`, `vector_search`, `fts`, `tsvector`, `BM25`. Read search handlers to understand: query flow, ranking strategy, and result formatting.
17
+
18
+ **Step 2 — Audit search quality**
19
+ Check for: pure vector search without keyword fallback (misses exact matches), no similarity threshold (returns irrelevant results at low scores), missing query embedding cache (repeated queries re-embed), no hybrid scoring (BM25 for exact + vector for semantic), and unoptimized vector index (HNSW parameters not tuned).
20
+
21
+ **Step 3 — Emit hybrid search**
22
+ Emit: combined BM25 + vector search with reciprocal rank fusion, similarity threshold filtering, query embedding cache, and HNSW index tuning.
23
+
24
+ #### Example
25
+
26
+ ```typescript
27
+ // Hybrid search — BM25 + vector with reciprocal rank fusion
28
+ async function hybridSearch(query: string, limit = 10) {
29
+ // Parallel: keyword (BM25) + semantic (vector)
30
+ const [keywordResults, vectorResults] = await Promise.all([
31
+ db.execute(sql`
32
+ SELECT id, content, ts_rank(search_vector, plainto_tsquery(${query})) AS bm25_score
33
+ FROM documents
34
+ WHERE search_vector @@ plainto_tsquery(${query})
35
+ ORDER BY bm25_score DESC LIMIT ${limit * 2}
36
+ `),
37
+ db.execute(sql`
38
+ SELECT id, content, 1 - (embedding <=> ${await getEmbedding(query)}) AS vector_score
39
+ FROM documents
40
+ ORDER BY embedding <=> ${await getEmbedding(query)}
41
+ LIMIT ${limit * 2}
42
+ `),
43
+ ]);
44
+
45
+ // Reciprocal rank fusion (k=60)
46
+ const scores = new Map<string, number>();
47
+ const K = 60;
48
+ keywordResults.forEach((r, i) => scores.set(r.id, (scores.get(r.id) || 0) + 1 / (K + i + 1)));
49
+ vectorResults.forEach((r, i) => scores.set(r.id, (scores.get(r.id) || 0) + 1 / (K + i + 1)));
50
+
51
+ return [...scores.entries()]
52
+ .sort((a, b) => b[1] - a[1])
53
+ .slice(0, limit)
54
+ .filter(([_, score]) => score > 0.01); // threshold
55
+ }
56
+
57
+ // Embedding cache (avoid re-embedding repeated queries)
58
+ const embeddingCache = new Map<string, number[]>();
59
+ async function getEmbedding(text: string): Promise<number[]> {
60
+ const cached = embeddingCache.get(text);
61
+ if (cached) return cached;
62
+ const { data } = await openai.embeddings.create({ model: 'text-embedding-3-small', input: text });
63
+ embeddingCache.set(text, data[0].embedding);
64
+ return data[0].embedding;
65
+ }
66
+ ```
@@ -1,74 +1,74 @@
1
- ---
2
- name: "fine-tuning-guide"
3
- pack: "@rune/ai-ml"
4
- description: "Fine-tuning workflows — dataset preparation, training configuration, evaluation metrics, deployment, A/B testing."
5
- model: sonnet
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # fine-tuning-guide
10
-
11
- Fine-tuning workflows — dataset preparation, training configuration, evaluation metrics, deployment, A/B testing.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Audit training data**
16
- Use Read to examine the dataset files. Check for: data format (JSONL with `messages` array), train/eval split (eval must not overlap with train), sufficient examples (minimum 50, recommended 200+), balanced class distribution, and PII in training data.
17
-
18
- **Step 2 — Prepare and validate dataset**
19
- Emit: JSONL formatter that validates each example, train/eval splitter with stratification, token count estimator (cost preview), and data quality checks (duplicate detection, format validation).
20
-
21
- **Step 3 — Execute fine-tuning and evaluate**
22
- Emit: fine-tune API call with hyperparameters, evaluation script that compares base vs fine-tuned on held-out set, and A/B deployment configuration.
23
-
24
- #### Example
25
-
26
- ```python
27
- # Fine-tuning workflow — prepare, train, evaluate
28
- import json
29
- import openai
30
- from sklearn.model_selection import train_test_split
31
-
32
- # Step 1: Prepare JSONL dataset
33
- def prepare_dataset(examples: list[dict], output_prefix: str):
34
- train, eval_set = train_test_split(examples, test_size=0.2, random_state=42)
35
-
36
- for split_name, split_data in [("train", train), ("eval", eval_set)]:
37
- path = f"{output_prefix}_{split_name}.jsonl"
38
- with open(path, "w") as f:
39
- for ex in split_data:
40
- f.write(json.dumps({"messages": [
41
- {"role": "system", "content": ex["system"]},
42
- {"role": "user", "content": ex["input"]},
43
- {"role": "assistant", "content": ex["output"]},
44
- ]}) + "\n")
45
- print(f"Wrote {len(split_data)} examples to {path}")
46
-
47
- # Step 2: Launch fine-tuning
48
- def start_fine_tune(train_file: str, eval_file: str):
49
- train_id = openai.files.create(file=open(train_file, "rb"), purpose="fine-tune").id
50
- eval_id = openai.files.create(file=open(eval_file, "rb"), purpose="fine-tune").id
51
-
52
- job = openai.fine_tuning.jobs.create(
53
- training_file=train_id,
54
- validation_file=eval_id,
55
- model="gpt-4o-mini-2024-07-18",
56
- hyperparameters={"n_epochs": 3, "batch_size": "auto", "learning_rate_multiplier": "auto"},
57
- )
58
- print(f"Fine-tuning job: {job.id} — status: {job.status}")
59
- return job
60
-
61
- # Step 3: Evaluate base vs fine-tuned
62
- def evaluate(base_model: str, ft_model: str, eval_set: list[dict]) -> dict:
63
- results = {"base": {"correct": 0}, "finetuned": {"correct": 0}}
64
- for ex in eval_set:
65
- for label, model in [("base", base_model), ("finetuned", ft_model)]:
66
- response = openai.chat.completions.create(
67
- model=model, messages=ex["messages"][:2], max_tokens=500,
68
- )
69
- if response.choices[0].message.content.strip() == ex["messages"][2]["content"].strip():
70
- results[label]["correct"] += 1
71
- for label in results:
72
- results[label]["accuracy"] = results[label]["correct"] / len(eval_set)
73
- return results
74
- ```
1
+ ---
2
+ name: "fine-tuning-guide"
3
+ pack: "@rune/ai-ml"
4
+ description: "Fine-tuning workflows — dataset preparation, training configuration, evaluation metrics, deployment, A/B testing."
5
+ model: sonnet
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # fine-tuning-guide
10
+
11
+ Fine-tuning workflows — dataset preparation, training configuration, evaluation metrics, deployment, A/B testing.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Audit training data**
16
+ Use Read to examine the dataset files. Check for: data format (JSONL with `messages` array), train/eval split (eval must not overlap with train), sufficient examples (minimum 50, recommended 200+), balanced class distribution, and PII in training data.
17
+
18
+ **Step 2 — Prepare and validate dataset**
19
+ Emit: JSONL formatter that validates each example, train/eval splitter with stratification, token count estimator (cost preview), and data quality checks (duplicate detection, format validation).
20
+
21
+ **Step 3 — Execute fine-tuning and evaluate**
22
+ Emit: fine-tune API call with hyperparameters, evaluation script that compares base vs fine-tuned on held-out set, and A/B deployment configuration.
23
+
24
+ #### Example
25
+
26
+ ```python
27
+ # Fine-tuning workflow — prepare, train, evaluate
28
+ import json
29
+ import openai
30
+ from sklearn.model_selection import train_test_split
31
+
32
+ # Step 1: Prepare JSONL dataset
33
+ def prepare_dataset(examples: list[dict], output_prefix: str):
34
+ train, eval_set = train_test_split(examples, test_size=0.2, random_state=42)
35
+
36
+ for split_name, split_data in [("train", train), ("eval", eval_set)]:
37
+ path = f"{output_prefix}_{split_name}.jsonl"
38
+ with open(path, "w") as f:
39
+ for ex in split_data:
40
+ f.write(json.dumps({"messages": [
41
+ {"role": "system", "content": ex["system"]},
42
+ {"role": "user", "content": ex["input"]},
43
+ {"role": "assistant", "content": ex["output"]},
44
+ ]}) + "\n")
45
+ print(f"Wrote {len(split_data)} examples to {path}")
46
+
47
+ # Step 2: Launch fine-tuning
48
+ def start_fine_tune(train_file: str, eval_file: str):
49
+ train_id = openai.files.create(file=open(train_file, "rb"), purpose="fine-tune").id
50
+ eval_id = openai.files.create(file=open(eval_file, "rb"), purpose="fine-tune").id
51
+
52
+ job = openai.fine_tuning.jobs.create(
53
+ training_file=train_id,
54
+ validation_file=eval_id,
55
+ model="gpt-4o-mini-2024-07-18",
56
+ hyperparameters={"n_epochs": 3, "batch_size": "auto", "learning_rate_multiplier": "auto"},
57
+ )
58
+ print(f"Fine-tuning job: {job.id} — status: {job.status}")
59
+ return job
60
+
61
+ # Step 3: Evaluate base vs fine-tuned
62
+ def evaluate(base_model: str, ft_model: str, eval_set: list[dict]) -> dict:
63
+ results = {"base": {"correct": 0}, "finetuned": {"correct": 0}}
64
+ for ex in eval_set:
65
+ for label, model in [("base", base_model), ("finetuned", ft_model)]:
66
+ response = openai.chat.completions.create(
67
+ model=model, messages=ex["messages"][:2], max_tokens=500,
68
+ )
69
+ if response.choices[0].message.content.strip() == ex["messages"][2]["content"].strip():
70
+ results[label]["correct"] += 1
71
+ for label in results:
72
+ results[label]["accuracy"] = results[label]["correct"] / len(eval_set)
73
+ return results
74
+ ```
@@ -1,125 +1,125 @@
1
- ---
2
- name: "llm-architect"
3
- pack: "@rune/ai-ml"
4
- description: "LLM system architecture — model selection, prompt engineering patterns, evaluation frameworks, cost optimization, multi-model routing, and guardrail design."
5
- model: opus
6
- tools: [Read, Edit, Write, Grep, Glob, Bash]
7
- ---
8
-
9
- # llm-architect
10
-
11
- LLM system architecture — model selection, prompt engineering patterns, evaluation frameworks, cost optimization, multi-model routing, and guardrail design.
12
-
13
- #### Workflow
14
-
15
- **Step 1 — Assess LLM requirements**
16
- Understand the use case: what does the LLM need to do? Classify into:
17
- - **Generation**: open-ended text (blog, email, creative writing)
18
- - **Extraction**: structured data from unstructured input (JSON from text, entities, classification)
19
- - **Reasoning**: multi-step logic (math, code generation, planning)
20
- - **Conversation**: multi-turn dialogue with memory
21
- - **Agentic**: tool use, function calling, autonomous task execution
22
-
23
- For each class, identify: latency requirements (real-time < 2s, async < 30s, batch), accuracy requirements (critical = needs eval suite, casual = spot check), cost sensitivity (per-call budget), and data sensitivity (PII, HIPAA, can data leave the network?).
24
-
25
- **Step 2 — Model selection matrix**
26
- Based on requirements, recommend model tier:
27
-
28
- | Requirement | Recommended | Fallback |
29
- |------------|-------------|----------|
30
- | Fast + cheap (classification, routing) | Haiku / GPT-4o-mini | Local (Llama 3) |
31
- | Balanced (code, summaries, RAG) | Sonnet / GPT-4o | Haiku with retry |
32
- | Deep reasoning (architecture, math) | Opus / o1 | Sonnet with chain-of-thought |
33
- | On-premise required | Llama 3 / Mistral | Ollama local deployment |
34
- | Multimodal (vision + text) | Sonnet / GPT-4o | Local LLaVA |
35
-
36
- Emit: primary model, fallback model, estimated cost per 1K calls, and latency p50/p99.
37
-
38
- **Step 3 — Prompt architecture**
39
- Design the prompt structure:
40
- - **System prompt**: Role definition, constraints, output format. Keep under 500 tokens for cost efficiency.
41
- - **Few-shot examples**: 2-3 examples for extraction/classification tasks. Format matches expected output exactly.
42
- - **Chain-of-thought**: For reasoning tasks, explicitly request step-by-step thinking before final answer.
43
- - **Structured output**: JSON mode or tool use for extraction. Define schema with Zod/Pydantic for validation.
44
-
45
- **Step 4 — Guardrails and evaluation**
46
- Design safety and quality layers:
47
- - **Input guardrails**: PII detection, prompt injection detection, topic filtering
48
- - **Output guardrails**: Schema validation, hallucination checks, toxicity filtering
49
- - **Evaluation framework**: Define eval dataset (50+ examples), metrics (accuracy, latency, cost), and regression threshold (new prompt must not drop > 2% on any metric)
50
-
51
- Save architecture doc to `.rune/ai/llm-architecture.md`.
52
-
53
- #### Example
54
-
55
- ```typescript
56
- // Multi-model router with fallback
57
- interface ModelConfig {
58
- id: string;
59
- provider: 'anthropic' | 'openai' | 'local';
60
- costPer1kTokens: number;
61
- maxTokens: number;
62
- latencyP50Ms: number;
63
- }
64
-
65
- const MODELS: Record<string, ModelConfig> = {
66
- fast: {
67
- id: 'claude-haiku-4-5-20251001',
68
- provider: 'anthropic',
69
- costPer1kTokens: 0.001,
70
- maxTokens: 4096,
71
- latencyP50Ms: 200,
72
- },
73
- balanced: {
74
- id: 'claude-sonnet-4-6',
75
- provider: 'anthropic',
76
- costPer1kTokens: 0.01,
77
- maxTokens: 8192,
78
- latencyP50Ms: 800,
79
- },
80
- deep: {
81
- id: 'claude-opus-4-6',
82
- provider: 'anthropic',
83
- costPer1kTokens: 0.05,
84
- maxTokens: 16384,
85
- latencyP50Ms: 2000,
86
- },
87
- };
88
-
89
- type TaskComplexity = 'trivial' | 'standard' | 'complex';
90
-
91
- function selectModel(complexity: TaskComplexity): ModelConfig {
92
- const map: Record<TaskComplexity, string> = {
93
- trivial: 'fast',
94
- standard: 'balanced',
95
- complex: 'deep',
96
- };
97
- return MODELS[map[complexity]];
98
- }
99
-
100
- // Prompt architecture template
101
- const systemPrompt = `You are a ${role} assistant.
102
-
103
- CONSTRAINTS:
104
- - ${constraints.join('\n- ')}
105
-
106
- OUTPUT FORMAT:
107
- Return valid JSON matching this schema:
108
- ${JSON.stringify(outputSchema, null, 2)}
109
-
110
- Do not include explanations outside the JSON.`;
111
-
112
- // Guardrail: validate structured output
113
- import { z } from 'zod';
114
-
115
- const OutputSchema = z.object({
116
- classification: z.enum(['positive', 'negative', 'neutral']),
117
- confidence: z.number().min(0).max(1),
118
- reasoning: z.string().max(200),
119
- });
120
-
121
- function validateOutput(raw: string): z.infer<typeof OutputSchema> {
122
- const parsed = JSON.parse(raw);
123
- return OutputSchema.parse(parsed); // throws if invalid
124
- }
125
- ```
1
+ ---
2
+ name: "llm-architect"
3
+ pack: "@rune/ai-ml"
4
+ description: "LLM system architecture — model selection, prompt engineering patterns, evaluation frameworks, cost optimization, multi-model routing, and guardrail design."
5
+ model: opus
6
+ tools: [Read, Edit, Write, Grep, Glob, Bash]
7
+ ---
8
+
9
+ # llm-architect
10
+
11
+ LLM system architecture — model selection, prompt engineering patterns, evaluation frameworks, cost optimization, multi-model routing, and guardrail design.
12
+
13
+ #### Workflow
14
+
15
+ **Step 1 — Assess LLM requirements**
16
+ Understand the use case: what does the LLM need to do? Classify into:
17
+ - **Generation**: open-ended text (blog, email, creative writing)
18
+ - **Extraction**: structured data from unstructured input (JSON from text, entities, classification)
19
+ - **Reasoning**: multi-step logic (math, code generation, planning)
20
+ - **Conversation**: multi-turn dialogue with memory
21
+ - **Agentic**: tool use, function calling, autonomous task execution
22
+
23
+ For each class, identify: latency requirements (real-time < 2s, async < 30s, batch), accuracy requirements (critical = needs eval suite, casual = spot check), cost sensitivity (per-call budget), and data sensitivity (PII, HIPAA, can data leave the network?).
24
+
25
+ **Step 2 — Model selection matrix**
26
+ Based on requirements, recommend model tier:
27
+
28
+ | Requirement | Recommended | Fallback |
29
+ |------------|-------------|----------|
30
+ | Fast + cheap (classification, routing) | Haiku / GPT-4o-mini | Local (Llama 3) |
31
+ | Balanced (code, summaries, RAG) | Sonnet / GPT-4o | Haiku with retry |
32
+ | Deep reasoning (architecture, math) | Opus / o1 | Sonnet with chain-of-thought |
33
+ | On-premise required | Llama 3 / Mistral | Ollama local deployment |
34
+ | Multimodal (vision + text) | Sonnet / GPT-4o | Local LLaVA |
35
+
36
+ Emit: primary model, fallback model, estimated cost per 1K calls, and latency p50/p99.
37
+
38
+ **Step 3 — Prompt architecture**
39
+ Design the prompt structure:
40
+ - **System prompt**: Role definition, constraints, output format. Keep under 500 tokens for cost efficiency.
41
+ - **Few-shot examples**: 2-3 examples for extraction/classification tasks. Format matches expected output exactly.
42
+ - **Chain-of-thought**: For reasoning tasks, explicitly request step-by-step thinking before final answer.
43
+ - **Structured output**: JSON mode or tool use for extraction. Define schema with Zod/Pydantic for validation.
44
+
45
+ **Step 4 — Guardrails and evaluation**
46
+ Design safety and quality layers:
47
+ - **Input guardrails**: PII detection, prompt injection detection, topic filtering
48
+ - **Output guardrails**: Schema validation, hallucination checks, toxicity filtering
49
+ - **Evaluation framework**: Define eval dataset (50+ examples), metrics (accuracy, latency, cost), and regression threshold (new prompt must not drop > 2% on any metric)
50
+
51
+ Save architecture doc to `.rune/ai/llm-architecture.md`.
52
+
53
+ #### Example
54
+
55
+ ```typescript
56
+ // Multi-model router with fallback
57
+ interface ModelConfig {
58
+ id: string;
59
+ provider: 'anthropic' | 'openai' | 'local';
60
+ costPer1kTokens: number;
61
+ maxTokens: number;
62
+ latencyP50Ms: number;
63
+ }
64
+
65
+ const MODELS: Record<string, ModelConfig> = {
66
+ fast: {
67
+ id: 'claude-haiku-4-5-20251001',
68
+ provider: 'anthropic',
69
+ costPer1kTokens: 0.001,
70
+ maxTokens: 4096,
71
+ latencyP50Ms: 200,
72
+ },
73
+ balanced: {
74
+ id: 'claude-sonnet-4-6',
75
+ provider: 'anthropic',
76
+ costPer1kTokens: 0.01,
77
+ maxTokens: 8192,
78
+ latencyP50Ms: 800,
79
+ },
80
+ deep: {
81
+ id: 'claude-opus-4-6',
82
+ provider: 'anthropic',
83
+ costPer1kTokens: 0.05,
84
+ maxTokens: 16384,
85
+ latencyP50Ms: 2000,
86
+ },
87
+ };
88
+
89
+ type TaskComplexity = 'trivial' | 'standard' | 'complex';
90
+
91
+ function selectModel(complexity: TaskComplexity): ModelConfig {
92
+ const map: Record<TaskComplexity, string> = {
93
+ trivial: 'fast',
94
+ standard: 'balanced',
95
+ complex: 'deep',
96
+ };
97
+ return MODELS[map[complexity]];
98
+ }
99
+
100
+ // Prompt architecture template
101
+ const systemPrompt = `You are a ${role} assistant.
102
+
103
+ CONSTRAINTS:
104
+ - ${constraints.join('\n- ')}
105
+
106
+ OUTPUT FORMAT:
107
+ Return valid JSON matching this schema:
108
+ ${JSON.stringify(outputSchema, null, 2)}
109
+
110
+ Do not include explanations outside the JSON.`;
111
+
112
+ // Guardrail: validate structured output
113
+ import { z } from 'zod';
114
+
115
+ const OutputSchema = z.object({
116
+ classification: z.enum(['positive', 'negative', 'neutral']),
117
+ confidence: z.number().min(0).max(1),
118
+ reasoning: z.string().max(200),
119
+ });
120
+
121
+ function validateOutput(raw: string): z.infer<typeof OutputSchema> {
122
+ const parsed = JSON.parse(raw);
123
+ return OutputSchema.parse(parsed); // throws if invalid
124
+ }
125
+ ```