novahiz 0.2.3 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (580) hide show
  1. package/NOTICE.md +36 -15
  2. package/README.md +33 -30
  3. package/adapters/README.md +1 -1
  4. package/adapters/opencode/agent/novahiz.md +2 -3
  5. package/adapters/opencode/instructions.md +3 -2
  6. package/adapters/opencode/novahiz.ts +336 -40
  7. package/bin/novahiz.cmd +2 -2
  8. package/catalog/categories.json +1358 -1093
  9. package/catalog/overrides.json +739 -13
  10. package/catalog/providers.json +57 -31
  11. package/catalog/rules.json +55 -29
  12. package/docs/ARCHITECTURE.md +4 -4
  13. package/docs/CATALOG.md +10 -9
  14. package/docs/CLASSIFICATION.md +3 -1
  15. package/docs/CONFIGURATION.md +5 -5
  16. package/docs/CONSTITUTION.md +4 -4
  17. package/docs/CONVERGENCE-2026-09-25.md +5 -0
  18. package/docs/GATE.md +32 -25
  19. package/docs/HARNESSES.md +15 -0
  20. package/docs/INSTALL.md +8 -5
  21. package/docs/PROVIDERS.md +26 -10
  22. package/docs/ROADMAPS.md +33 -3
  23. package/docs/RULES.md +15 -11
  24. package/docs/audit-2026-09-25.md +275 -0
  25. package/install/bootstrap.mjs +32 -37
  26. package/install/install.mjs +70 -53
  27. package/install/lib.mjs +20 -4
  28. package/mcp/novahiz-tools/README.md +5 -0
  29. package/mcp/novahiz-tools/index.mjs +78 -7
  30. package/novahiz.config.example.json +2 -5
  31. package/package.json +2 -3
  32. package/skills/apple-ui-audit/SKILL.md +95 -0
  33. package/skills/apple-ui-audit/references/screen-walkthrough.md +29 -0
  34. package/skills/browser-session/SKILL.md +61 -0
  35. package/skills/browser-session/references/session-run-sheet.md +30 -0
  36. package/skills/browser-session/scripts/browser_session_preflight.mjs +74 -0
  37. package/skills/browser-session/scripts/sample-run.json +5 -0
  38. package/skills/code-standards/SKILL.md +85 -0
  39. package/skills/code-standards/references/pr-review-worksheet.md +35 -0
  40. package/skills/design-token-pipeline/SKILL.md +85 -0
  41. package/skills/design-token-pipeline/references/dtcg-sample.json +37 -0
  42. package/skills/gate/SKILL.md +16 -1
  43. package/skills/llm-threat-review/SKILL.md +99 -0
  44. package/skills/llm-threat-review/references/review-worksheet.md +34 -0
  45. package/skills/llm-threat-review/scripts/llm_risk_lint.mjs +105 -0
  46. package/skills/novahiz-analyse/SKILL.md +2 -2
  47. package/skills/novahiz-audit/SKILL.md +2 -2
  48. package/skills/novahiz-code-review/SKILL.md +1 -1
  49. package/skills/novahiz-converge/SKILL.md +1 -1
  50. package/skills/novahiz-gate/SKILL.md +7 -5
  51. package/skills/novahiz-humanizer/SKILL.md +1 -1
  52. package/skills/novahiz-implement/SKILL.md +2 -2
  53. package/skills/novahiz-init/SKILL.md +1 -1
  54. package/skills/novahiz-plan/SKILL.md +2 -2
  55. package/skills/novahiz-planner/SKILL.md +6 -5
  56. package/skills/novahiz-task/SKILL.md +1 -1
  57. package/skills/openapi-mcp-server/SKILL.md +76 -0
  58. package/skills/openapi-mcp-server/references/mcp-ship-checklist.md +30 -0
  59. package/skills/openapi-mcp-server/references/tool-worksheet.md +30 -0
  60. package/skills/openapi-mcp-server/scripts/openapi_mcp_lint.mjs +133 -0
  61. package/skills/package-risk-audit/SKILL.md +83 -0
  62. package/skills/package-risk-audit/references/intake-worksheet.md +19 -0
  63. package/skills/package-risk-audit/scripts/dep_risk_lint.mjs +131 -0
  64. package/skills/prompt-rewriter/SKILL.md +18 -13
  65. package/skills/secrets-hygiene/SKILL.md +62 -0
  66. package/skills/secrets-hygiene/references/rotation-playbook.md +27 -0
  67. package/skills/secrets-hygiene/scripts/secret_scan.mjs +120 -0
  68. package/skills/skill-authoring/SKILL.md +93 -0
  69. package/skills/skill-authoring/references/skill-template.md +59 -0
  70. package/skills/skill-authoring/scripts/skill_frontmatter_lint.mjs +65 -0
  71. package/skills/skill-eval-loop/SKILL.md +103 -0
  72. package/skills/skill-eval-loop/references/negative-patterns.md +12 -0
  73. package/skills/skill-eval-loop/references/run-sheet.md +40 -0
  74. package/skills/skill-eval-loop/scripts/skill_structure_check.mjs +100 -0
  75. package/skills/ui-craft-rules/SKILL.md +96 -0
  76. package/skills/ui-slop-remover/SKILL.md +110 -0
  77. package/skills/ui-slop-remover/references/tell-catalogue.md +58 -0
  78. package/skills/ui-slop-remover/scripts/slop_lint.mjs +119 -0
  79. package/src/cli.ts +24 -4
  80. package/src/commands/context.ts +43 -4
  81. package/src/commands/doctor.ts +26 -8
  82. package/src/commands/gate.ts +68 -76
  83. package/src/commands/graft.ts +14 -9
  84. package/src/commands/inspect.ts +28 -2
  85. package/src/commands/task.ts +30 -1
  86. package/src/exec.ts +38 -10
  87. package/src/gate-repair.ts +87 -0
  88. package/src/gate.ts +543 -401
  89. package/src/graft.ts +56 -23
  90. package/src/ledger.ts +28 -2
  91. package/src/memory.ts +14 -1
  92. package/src/prompt-rewriter.ts +357 -24
  93. package/src/spec.ts +18 -5
  94. package/bundled-skills/ab-testing/SKILL.md +0 -353
  95. package/bundled-skills/ab-testing/evals/evals.json +0 -105
  96. package/bundled-skills/ab-testing/references/sample-size-guide.md +0 -263
  97. package/bundled-skills/ab-testing/references/test-templates.md +0 -277
  98. package/bundled-skills/ad-creative/SKILL.md +0 -425
  99. package/bundled-skills/ad-creative/assets/creative-review-template.html +0 -400
  100. package/bundled-skills/ad-creative/evals/evals.json +0 -217
  101. package/bundled-skills/ad-creative/references/creative-review-page.md +0 -106
  102. package/bundled-skills/ad-creative/references/creative-roadmap.md +0 -118
  103. package/bundled-skills/ad-creative/references/generative-tools.md +0 -637
  104. package/bundled-skills/ad-creative/references/hook-system.md +0 -115
  105. package/bundled-skills/ad-creative/references/imessage-video-ads.md +0 -201
  106. package/bundled-skills/ad-creative/references/meta-creative-formats.md +0 -114
  107. package/bundled-skills/ad-creative/references/motion-video-ads.md +0 -126
  108. package/bundled-skills/ad-creative/references/platform-specs.md +0 -213
  109. package/bundled-skills/ad-creative/references/short-form-video-specs.md +0 -213
  110. package/bundled-skills/ad-creative/references/static-ad-templates.md +0 -308
  111. package/bundled-skills/ads/SKILL.md +0 -499
  112. package/bundled-skills/ads/evals/evals.json +0 -157
  113. package/bundled-skills/ads/references/abm-playbook.md +0 -93
  114. package/bundled-skills/ads/references/ad-copy-templates.md +0 -207
  115. package/bundled-skills/ads/references/audience-targeting.md +0 -243
  116. package/bundled-skills/ads/references/audit-guardrails.md +0 -83
  117. package/bundled-skills/ads/references/b2b-paid-playbook.md +0 -115
  118. package/bundled-skills/ads/references/conversion-tracking.md +0 -361
  119. package/bundled-skills/ads/references/creative-research-automation.md +0 -103
  120. package/bundled-skills/ads/references/google-ads-audit-checklist.md +0 -84
  121. package/bundled-skills/ads/references/google-search-playbook.md +0 -120
  122. package/bundled-skills/ads/references/linkedin-b2b-playbook.md +0 -107
  123. package/bundled-skills/ads/references/meta-decision-system.md +0 -176
  124. package/bundled-skills/ads/references/payback-period.md +0 -69
  125. package/bundled-skills/ads/references/platform-setup-checklists.md +0 -277
  126. package/bundled-skills/ads/references/rsa-output-spec.md +0 -88
  127. package/bundled-skills/agent-device/SKILL.md +0 -20
  128. package/bundled-skills/ai-seo/SKILL.md +0 -488
  129. package/bundled-skills/ai-seo/evals/evals.json +0 -135
  130. package/bundled-skills/ai-seo/references/agent-readiness.md +0 -57
  131. package/bundled-skills/ai-seo/references/citations-vs-recommendations.md +0 -85
  132. package/bundled-skills/ai-seo/references/content-patterns.md +0 -287
  133. package/bundled-skills/ai-seo/references/content-types.md +0 -71
  134. package/bundled-skills/ai-seo/references/format-volatility.md +0 -98
  135. package/bundled-skills/ai-seo/references/okf.md +0 -104
  136. package/bundled-skills/ai-seo/references/platform-ranking-factors.md +0 -154
  137. package/bundled-skills/ai-seo/references/youtube-ai-citations.md +0 -60
  138. package/bundled-skills/analytics/SKILL.md +0 -310
  139. package/bundled-skills/analytics/evals/evals.json +0 -90
  140. package/bundled-skills/analytics/references/event-library.md +0 -260
  141. package/bundled-skills/analytics/references/ga4-implementation.md +0 -300
  142. package/bundled-skills/analytics/references/gtm-implementation.md +0 -390
  143. package/bundled-skills/android-emulator/SKILL.md +0 -32
  144. package/bundled-skills/android-reverse-engineering/LICENSE +0 -190
  145. package/bundled-skills/android-reverse-engineering/SKILL.md +0 -313
  146. package/bundled-skills/android-reverse-engineering/references/api-extraction-patterns.md +0 -197
  147. package/bundled-skills/android-reverse-engineering/references/call-flow-analysis.md +0 -210
  148. package/bundled-skills/android-reverse-engineering/references/fernflower-usage.md +0 -115
  149. package/bundled-skills/android-reverse-engineering/references/jadx-usage.md +0 -116
  150. package/bundled-skills/android-reverse-engineering/references/kotlin-name-recovery.md +0 -108
  151. package/bundled-skills/android-reverse-engineering/references/setup-guide.md +0 -221
  152. package/bundled-skills/android-reverse-engineering/references/third_party_hosts.txt +0 -122
  153. package/bundled-skills/android-reverse-engineering/scripts/check-deps.ps1 +0 -165
  154. package/bundled-skills/android-reverse-engineering/scripts/check-deps.sh +0 -129
  155. package/bundled-skills/android-reverse-engineering/scripts/decompile.ps1 +0 -400
  156. package/bundled-skills/android-reverse-engineering/scripts/decompile.sh +0 -578
  157. package/bundled-skills/android-reverse-engineering/scripts/find-api-calls.ps1 +0 -121
  158. package/bundled-skills/android-reverse-engineering/scripts/find-api-calls.sh +0 -339
  159. package/bundled-skills/android-reverse-engineering/scripts/fingerprint.sh +0 -241
  160. package/bundled-skills/android-reverse-engineering/scripts/install-dep.ps1 +0 -345
  161. package/bundled-skills/android-reverse-engineering/scripts/install-dep.sh +0 -448
  162. package/bundled-skills/android-reverse-engineering/scripts/lookup-name.sh +0 -85
  163. package/bundled-skills/android-reverse-engineering/scripts/recover-kotlin-names.sh +0 -140
  164. package/bundled-skills/build-graph/SKILL.md +0 -38
  165. package/bundled-skills/claude-history-ingest/SKILL.md +0 -463
  166. package/bundled-skills/claude-history-ingest/references/claude-data-format.md +0 -118
  167. package/bundled-skills/codex-history-ingest/SKILL.md +0 -248
  168. package/bundled-skills/codex-history-ingest/references/codex-data-format.md +0 -82
  169. package/bundled-skills/cold-email/SKILL.md +0 -159
  170. package/bundled-skills/cold-email/evals/evals.json +0 -94
  171. package/bundled-skills/cold-email/references/benchmarks.md +0 -83
  172. package/bundled-skills/cold-email/references/follow-up-sequences.md +0 -81
  173. package/bundled-skills/cold-email/references/frameworks.md +0 -90
  174. package/bundled-skills/cold-email/references/personalization.md +0 -79
  175. package/bundled-skills/cold-email/references/subject-lines.md +0 -53
  176. package/bundled-skills/competitor-profiling/SKILL.md +0 -415
  177. package/bundled-skills/competitor-profiling/evals/evals.json +0 -85
  178. package/bundled-skills/competitor-profiling/references/templates.md +0 -167
  179. package/bundled-skills/competitor-profiling/references/tool-reference.md +0 -179
  180. package/bundled-skills/computer-use/SKILL.md +0 -41
  181. package/bundled-skills/content-strategy/SKILL.md +0 -439
  182. package/bundled-skills/content-strategy/evals/evals.json +0 -116
  183. package/bundled-skills/content-strategy/references/content-distribution.md +0 -83
  184. package/bundled-skills/content-strategy/references/headless-cms.md +0 -194
  185. package/bundled-skills/copilot-history-ingest/SKILL.md +0 -376
  186. package/bundled-skills/copilot-history-ingest/references/copilot-data-format.md +0 -321
  187. package/bundled-skills/copy-editing/SKILL.md +0 -457
  188. package/bundled-skills/copy-editing/evals/evals.json +0 -89
  189. package/bundled-skills/copy-editing/references/checklist.md +0 -66
  190. package/bundled-skills/copy-editing/references/content-refresh.md +0 -38
  191. package/bundled-skills/copy-editing/references/plain-english-alternatives.md +0 -394
  192. package/bundled-skills/copywriting/SKILL.md +0 -256
  193. package/bundled-skills/copywriting/evals/evals.json +0 -126
  194. package/bundled-skills/copywriting/references/copy-frameworks.md +0 -433
  195. package/bundled-skills/copywriting/references/natural-transitions.md +0 -272
  196. package/bundled-skills/cro/SKILL.md +0 -187
  197. package/bundled-skills/cro/evals/evals.json +0 -111
  198. package/bundled-skills/cro/references/experiments.md +0 -248
  199. package/bundled-skills/cro/references/form.md +0 -422
  200. package/bundled-skills/cross-linker/SKILL.md +0 -309
  201. package/bundled-skills/customer-research/SKILL.md +0 -305
  202. package/bundled-skills/customer-research/evals/evals.json +0 -176
  203. package/bundled-skills/customer-research/references/interviews-and-surveys.md +0 -155
  204. package/bundled-skills/customer-research/references/source-guides.md +0 -401
  205. package/bundled-skills/daily-update/SKILL.md +0 -223
  206. package/bundled-skills/dart-add-unit-test/SKILL.md +0 -122
  207. package/bundled-skills/dart-build-cli-app/SKILL.md +0 -185
  208. package/bundled-skills/dart-collect-coverage/SKILL.md +0 -141
  209. package/bundled-skills/dart-fix-runtime-errors/SKILL.md +0 -166
  210. package/bundled-skills/dart-generate-test-mocks/SKILL.md +0 -155
  211. package/bundled-skills/dart-migrate-to-checks-package/SKILL.md +0 -528
  212. package/bundled-skills/dart-resolve-package-conflicts/SKILL.md +0 -116
  213. package/bundled-skills/dart-run-static-analysis/SKILL.md +0 -104
  214. package/bundled-skills/dart-setup-ffi-assets/SKILL.md +0 -419
  215. package/bundled-skills/dart-use-doc-examples/SKILL.md +0 -96
  216. package/bundled-skills/dart-use-ffigen/SKILL.md +0 -218
  217. package/bundled-skills/dart-use-pattern-matching/SKILL.md +0 -146
  218. package/bundled-skills/dart-use-primary-constructors/SKILL.md +0 -262
  219. package/bundled-skills/dart-write-documentation/SKILL.md +0 -136
  220. package/bundled-skills/debug-issue/SKILL.md +0 -28
  221. package/bundled-skills/dogfood/SKILL.md +0 -27
  222. package/bundled-skills/eas-app-stores/SKILL.md +0 -176
  223. package/bundled-skills/eas-app-stores/agents/openai.yaml +0 -4
  224. package/bundled-skills/eas-app-stores/references/app-store-metadata.md +0 -497
  225. package/bundled-skills/eas-app-stores/references/ios-app-store.md +0 -376
  226. package/bundled-skills/eas-app-stores/references/native-ios.md +0 -167
  227. package/bundled-skills/eas-app-stores/references/play-store.md +0 -244
  228. package/bundled-skills/eas-app-stores/references/testflight.md +0 -62
  229. package/bundled-skills/eas-app-stores/references/workflows.md +0 -120
  230. package/bundled-skills/eas-hosting/SKILL.md +0 -431
  231. package/bundled-skills/eas-hosting/agents/openai.yaml +0 -4
  232. package/bundled-skills/eas-observe/SKILL.md +0 -54
  233. package/bundled-skills/eas-observe/agents/openai.yaml +0 -4
  234. package/bundled-skills/eas-observe/references/metrics.md +0 -98
  235. package/bundled-skills/eas-observe/references/queries.md +0 -403
  236. package/bundled-skills/eas-observe/references/setup.md +0 -476
  237. package/bundled-skills/eas-observe/references/third-party.md +0 -136
  238. package/bundled-skills/eas-simulator/SKILL.md +0 -208
  239. package/bundled-skills/eas-simulator/agents/openai.yaml +0 -4
  240. package/bundled-skills/eas-simulator/references/controllers.md +0 -106
  241. package/bundled-skills/eas-simulator/references/run-your-app.md +0 -224
  242. package/bundled-skills/eas-simulator/references/troubleshooting.md +0 -44
  243. package/bundled-skills/eas-update/SKILL.md +0 -146
  244. package/bundled-skills/eas-update/agents/openai.yaml +0 -4
  245. package/bundled-skills/eas-update-insights/SKILL.md +0 -238
  246. package/bundled-skills/eas-update-insights/agents/openai.yaml +0 -4
  247. package/bundled-skills/eas-update-insights/references/channel-insights-schema.md +0 -47
  248. package/bundled-skills/eas-update-insights/references/update-insights-schema.md +0 -69
  249. package/bundled-skills/eas-workflows/SKILL.md +0 -99
  250. package/bundled-skills/eas-workflows/agents/openai.yaml +0 -4
  251. package/bundled-skills/eas-workflows/scripts/fetch.js +0 -109
  252. package/bundled-skills/eas-workflows/scripts/package.json +0 -6
  253. package/bundled-skills/emails/SKILL.md +0 -311
  254. package/bundled-skills/emails/evals/evals.json +0 -93
  255. package/bundled-skills/emails/references/copy-guidelines.md +0 -113
  256. package/bundled-skills/emails/references/email-types.md +0 -515
  257. package/bundled-skills/emails/references/sequence-templates.md +0 -168
  258. package/bundled-skills/explore-codebase/SKILL.md +0 -29
  259. package/bundled-skills/expo-animation/LICENSE +0 -21
  260. package/bundled-skills/expo-animation/RECIPES.md +0 -385
  261. package/bundled-skills/expo-animation/SKILL.md +0 -267
  262. package/bundled-skills/expo-animation/agents/openai.yaml +0 -4
  263. package/bundled-skills/expo-app-clip/SKILL.md +0 -290
  264. package/bundled-skills/expo-app-clip/agents/openai.yaml +0 -4
  265. package/bundled-skills/expo-app-clip/references/native-module.md +0 -96
  266. package/bundled-skills/expo-brownfield/SKILL.md +0 -62
  267. package/bundled-skills/expo-brownfield/agents/openai.yaml +0 -4
  268. package/bundled-skills/expo-brownfield/references/brownfield-integrated.md +0 -526
  269. package/bundled-skills/expo-brownfield/references/brownfield-isolated.md +0 -451
  270. package/bundled-skills/expo-brownfield/references/comparison.md +0 -63
  271. package/bundled-skills/expo-brownfield/references/troubleshooting.md +0 -88
  272. package/bundled-skills/expo-data-fetching/SKILL.md +0 -478
  273. package/bundled-skills/expo-data-fetching/agents/openai.yaml +0 -4
  274. package/bundled-skills/expo-data-fetching/references/expo-router-loaders.md +0 -341
  275. package/bundled-skills/expo-data-fetching/references/offline-and-cancellation.md +0 -68
  276. package/bundled-skills/expo-design-system/SKILL.md +0 -376
  277. package/bundled-skills/expo-design-system/agents/openai.yaml +0 -4
  278. package/bundled-skills/expo-design-system/references/audit.md +0 -190
  279. package/bundled-skills/expo-design-system/references/native-slop.md +0 -74
  280. package/bundled-skills/expo-dev-client/SKILL.md +0 -182
  281. package/bundled-skills/expo-dev-client/agents/openai.yaml +0 -4
  282. package/bundled-skills/expo-dom/SKILL.md +0 -425
  283. package/bundled-skills/expo-dom/agents/openai.yaml +0 -4
  284. package/bundled-skills/expo-examples/SKILL.md +0 -106
  285. package/bundled-skills/expo-examples/agents/openai.yaml +0 -4
  286. package/bundled-skills/expo-examples/references/catalog.md +0 -105
  287. package/bundled-skills/expo-migrate-module/SKILL.md +0 -113
  288. package/bundled-skills/expo-migrate-module/agents/openai.yaml +0 -4
  289. package/bundled-skills/expo-migrate-module/references/compatibility.md +0 -73
  290. package/bundled-skills/expo-migrate-module/references/example.md +0 -212
  291. package/bundled-skills/expo-migrate-module/references/migration-map.md +0 -306
  292. package/bundled-skills/expo-module/SKILL.md +0 -151
  293. package/bundled-skills/expo-module/agents/openai.yaml +0 -4
  294. package/bundled-skills/expo-module/references/config-plugin.md +0 -90
  295. package/bundled-skills/expo-module/references/create-expo-module.md +0 -206
  296. package/bundled-skills/expo-module/references/lifecycle.md +0 -127
  297. package/bundled-skills/expo-module/references/module-config.md +0 -48
  298. package/bundled-skills/expo-module/references/native-module.md +0 -286
  299. package/bundled-skills/expo-module/references/native-view.md +0 -171
  300. package/bundled-skills/expo-native-ui/SKILL.md +0 -201
  301. package/bundled-skills/expo-native-ui/agents/openai.yaml +0 -4
  302. package/bundled-skills/expo-native-ui/references/controls.md +0 -231
  303. package/bundled-skills/expo-native-ui/references/gradients.md +0 -106
  304. package/bundled-skills/expo-native-ui/references/icons.md +0 -232
  305. package/bundled-skills/expo-native-ui/references/media.md +0 -193
  306. package/bundled-skills/expo-native-ui/references/storage.md +0 -121
  307. package/bundled-skills/expo-native-ui/references/visual-effects.md +0 -198
  308. package/bundled-skills/expo-native-ui/references/webgpu-three.md +0 -605
  309. package/bundled-skills/expo-overview/SKILL.md +0 -115
  310. package/bundled-skills/expo-overview/agents/openai.yaml +0 -4
  311. package/bundled-skills/expo-project-structure/SKILL.md +0 -114
  312. package/bundled-skills/expo-project-structure/agents/openai.yaml +0 -4
  313. package/bundled-skills/expo-router/SKILL.md +0 -240
  314. package/bundled-skills/expo-router/agents/openai.yaml +0 -4
  315. package/bundled-skills/expo-router/references/form-sheet.md +0 -253
  316. package/bundled-skills/expo-router/references/route-structure.md +0 -229
  317. package/bundled-skills/expo-router/references/search.md +0 -248
  318. package/bundled-skills/expo-router/references/tabs.md +0 -433
  319. package/bundled-skills/expo-router/references/toolbar-and-headers.md +0 -284
  320. package/bundled-skills/expo-router/references/zoom-transitions.md +0 -158
  321. package/bundled-skills/expo-skill-eval/SKILL.md +0 -314
  322. package/bundled-skills/expo-skill-eval/agents/visual-grader.md +0 -52
  323. package/bundled-skills/expo-skill-eval/references/design-rubric.md +0 -45
  324. package/bundled-skills/expo-skill-eval/references/runtime-matrix.md +0 -34
  325. package/bundled-skills/expo-skill-eval/scripts/check-static.sh +0 -64
  326. package/bundled-skills/expo-skill-eval/scripts/clean-fixture.sh +0 -56
  327. package/bundled-skills/expo-skill-eval/scripts/generate_viewer.py +0 -450
  328. package/bundled-skills/expo-skill-eval/scripts/latest-sdk.sh +0 -46
  329. package/bundled-skills/expo-skill-eval/scripts/make-fixture.sh +0 -95
  330. package/bundled-skills/expo-skill-eval/scripts/make-workspace.sh +0 -31
  331. package/bundled-skills/expo-skill-eval/scripts/snapshot-android.sh +0 -284
  332. package/bundled-skills/expo-skill-eval/scripts/snapshot-ios.sh +0 -132
  333. package/bundled-skills/expo-skill-eval/scripts/snapshot-web.sh +0 -61
  334. package/bundled-skills/expo-skill-feedback/SKILL.md +0 -87
  335. package/bundled-skills/expo-skill-feedback/agents/openai.yaml +0 -4
  336. package/bundled-skills/expo-skill-feedback/scripts/skill-event.cjs +0 -180
  337. package/bundled-skills/expo-skill-feedback/scripts/telemetry.cjs +0 -65
  338. package/bundled-skills/expo-skill-feedback/scripts/telemetry_common.cjs +0 -171
  339. package/bundled-skills/expo-ui/SKILL.md +0 -101
  340. package/bundled-skills/expo-ui/agents/openai.yaml +0 -4
  341. package/bundled-skills/expo-ui/references/drop-in-replacements.md +0 -27
  342. package/bundled-skills/expo-ui/references/jetpack-compose.md +0 -73
  343. package/bundled-skills/expo-ui/references/swift-ui.md +0 -73
  344. package/bundled-skills/expo-ui/references/universal.md +0 -104
  345. package/bundled-skills/expo-ui/scripts/list-components.js +0 -193
  346. package/bundled-skills/expo-upgrade/SKILL.md +0 -150
  347. package/bundled-skills/expo-upgrade/agents/openai.yaml +0 -4
  348. package/bundled-skills/expo-upgrade/references/expo-av-to-audio.md +0 -132
  349. package/bundled-skills/expo-upgrade/references/expo-av-to-video.md +0 -160
  350. package/bundled-skills/expo-upgrade/references/native-tabs.md +0 -124
  351. package/bundled-skills/expo-upgrade/references/new-architecture.md +0 -79
  352. package/bundled-skills/expo-upgrade/references/react-19.md +0 -79
  353. package/bundled-skills/expo-upgrade/references/react-compiler.md +0 -59
  354. package/bundled-skills/expo-upgrade/references/react-navigation-to-expo-router.md +0 -61
  355. package/bundled-skills/expo-web-to-native/SKILL.md +0 -91
  356. package/bundled-skills/expo-web-to-native/agents/openai.yaml +0 -4
  357. package/bundled-skills/expo-web-to-native/references/false-friends.md +0 -119
  358. package/bundled-skills/expo-web-to-native/references/native-patterns.md +0 -39
  359. package/bundled-skills/expo-web-to-native/references/run-as-goal.md +0 -48
  360. package/bundled-skills/expo-web-to-native/references/verify-on-device.md +0 -45
  361. package/bundled-skills/find-skills/SKILL.md +0 -141
  362. package/bundled-skills/flutter-add-integration-test/SKILL.md +0 -163
  363. package/bundled-skills/flutter-add-widget-preview/SKILL.md +0 -145
  364. package/bundled-skills/flutter-add-widget-test/SKILL.md +0 -154
  365. package/bundled-skills/flutter-apply-architecture-best-practices/SKILL.md +0 -162
  366. package/bundled-skills/flutter-build-responsive-layout/SKILL.md +0 -139
  367. package/bundled-skills/flutter-fix-layout-issues/SKILL.md +0 -130
  368. package/bundled-skills/flutter-implement-json-serialization/SKILL.md +0 -153
  369. package/bundled-skills/flutter-setup-declarative-routing/SKILL.md +0 -255
  370. package/bundled-skills/flutter-setup-localization/SKILL.md +0 -210
  371. package/bundled-skills/flutter-use-http-package/SKILL.md +0 -174
  372. package/bundled-skills/graph-colorize/SKILL.md +0 -180
  373. package/bundled-skills/hermes-history-ingest/SKILL.md +0 -239
  374. package/bundled-skills/hermes-history-ingest/references/hermes-data-format.md +0 -131
  375. package/bundled-skills/impl-validator/SKILL.md +0 -118
  376. package/bundled-skills/ios-simulator/SKILL.md +0 -32
  377. package/bundled-skills/json-canvas/SKILL.md +0 -244
  378. package/bundled-skills/json-canvas/references/EXAMPLES.md +0 -329
  379. package/bundled-skills/launch/SKILL.md +0 -381
  380. package/bundled-skills/launch/evals/evals.json +0 -106
  381. package/bundled-skills/llm-wiki/SKILL.md +0 -661
  382. package/bundled-skills/llm-wiki/references/WRITING.md +0 -28
  383. package/bundled-skills/llm-wiki/references/karpathy-pattern.md +0 -45
  384. package/bundled-skills/marketing-plan/SKILL.md +0 -300
  385. package/bundled-skills/marketing-plan/evals/evals.json +0 -112
  386. package/bundled-skills/marketing-plan/references/aarrr-framework.md +0 -180
  387. package/bundled-skills/marketing-plan/references/budget-planning.md +0 -168
  388. package/bundled-skills/marketing-plan/references/client-types.md +0 -373
  389. package/bundled-skills/marketing-plan/references/current-state-rubric.md +0 -255
  390. package/bundled-skills/marketing-plan/references/example-quietude.md +0 -972
  391. package/bundled-skills/marketing-plan/references/funding-stage-unlocks.md +0 -230
  392. package/bundled-skills/marketing-plan/references/growth-patterns.md +0 -190
  393. package/bundled-skills/marketing-plan/references/idea-cross-reference.md +0 -265
  394. package/bundled-skills/marketing-plan/references/measurement-framework.md +0 -213
  395. package/bundled-skills/marketing-plan/references/methodology.md +0 -363
  396. package/bundled-skills/marketing-plan/references/ops-stack-mapping.md +0 -197
  397. package/bundled-skills/marketing-plan/references/plan-template.md +0 -494
  398. package/bundled-skills/marketing-plan/references/team-and-agency-model.md +0 -278
  399. package/bundled-skills/marketing-psychology/SKILL.md +0 -455
  400. package/bundled-skills/marketing-psychology/evals/evals.json +0 -88
  401. package/bundled-skills/memory-bridge/SKILL.md +0 -163
  402. package/bundled-skills/memory-save/SKILL.md +0 -88
  403. package/bundled-skills/obsidian-bases/SKILL.md +0 -499
  404. package/bundled-skills/obsidian-bases/references/FUNCTIONS_REFERENCE.md +0 -173
  405. package/bundled-skills/obsidian-cli/SKILL.md +0 -106
  406. package/bundled-skills/obsidian-layout-adjustment/SKILL.md +0 -226
  407. package/bundled-skills/obsidian-layout-adjustment/evals/evals.json +0 -53
  408. package/bundled-skills/obsidian-layout-adjustment/references/workflow-reference.md +0 -212
  409. package/bundled-skills/obsidian-markdown/SKILL.md +0 -196
  410. package/bundled-skills/obsidian-markdown/references/CALLOUTS.md +0 -58
  411. package/bundled-skills/obsidian-markdown/references/EMBEDS.md +0 -70
  412. package/bundled-skills/obsidian-markdown/references/PROPERTIES.md +0 -61
  413. package/bundled-skills/offers/SKILL.md +0 -155
  414. package/bundled-skills/offers/evals/evals.json +0 -21
  415. package/bundled-skills/offers/references/bonus-stacking.md +0 -150
  416. package/bundled-skills/offers/references/examples.md +0 -215
  417. package/bundled-skills/offers/references/guarantee-design.md +0 -172
  418. package/bundled-skills/offers/references/offer-anatomy.md +0 -203
  419. package/bundled-skills/offers/references/offer-formats.md +0 -240
  420. package/bundled-skills/offers/references/saas-offers.md +0 -116
  421. package/bundled-skills/offers/references/scarcity-urgency.md +0 -175
  422. package/bundled-skills/offers/references/value-equation.md +0 -134
  423. package/bundled-skills/onboarding/SKILL.md +0 -261
  424. package/bundled-skills/onboarding/evals/evals.json +0 -108
  425. package/bundled-skills/onboarding/references/activation-models.md +0 -57
  426. package/bundled-skills/onboarding/references/experiments.md +0 -258
  427. package/bundled-skills/onboarding/references/minimum-path-to-value.md +0 -49
  428. package/bundled-skills/openclaw-history-ingest/SKILL.md +0 -257
  429. package/bundled-skills/openclaw-history-ingest/references/openclaw-data-format.md +0 -154
  430. package/bundled-skills/orchestration/SKILL.md +0 -69
  431. package/bundled-skills/pi-history-ingest/SKILL.md +0 -312
  432. package/bundled-skills/pricing/SKILL.md +0 -295
  433. package/bundled-skills/pricing/evals/evals.json +0 -119
  434. package/bundled-skills/pricing/references/pricing-models.md +0 -66
  435. package/bundled-skills/pricing/references/pricing-page-teardown.md +0 -85
  436. package/bundled-skills/pricing/references/research-methods.md +0 -152
  437. package/bundled-skills/pricing/references/tier-structure.md +0 -232
  438. package/bundled-skills/product-marketing/SKILL.md +0 -255
  439. package/bundled-skills/product-marketing/evals/evals.json +0 -98
  440. package/bundled-skills/prospecting/SKILL.md +0 -262
  441. package/bundled-skills/prospecting/evals/evals.json +0 -122
  442. package/bundled-skills/prospecting/references/b2b-prospecting.md +0 -106
  443. package/bundled-skills/prospecting/references/compliance.md +0 -123
  444. package/bundled-skills/prospecting/references/data-sources.md +0 -287
  445. package/bundled-skills/prospecting/references/demand-signals.md +0 -129
  446. package/bundled-skills/prospecting/references/local-prospecting.md +0 -165
  447. package/bundled-skills/prospecting/references/saas-prospecting.md +0 -123
  448. package/bundled-skills/refactor-safely/SKILL.md +0 -29
  449. package/bundled-skills/review-delta/SKILL.md +0 -46
  450. package/bundled-skills/review-pr/SKILL.md +0 -66
  451. package/bundled-skills/sales-enablement/SKILL.md +0 -359
  452. package/bundled-skills/sales-enablement/evals/evals.json +0 -91
  453. package/bundled-skills/sales-enablement/references/deck-frameworks.md +0 -263
  454. package/bundled-skills/sales-enablement/references/demo-scripts.md +0 -355
  455. package/bundled-skills/sales-enablement/references/objection-library.md +0 -270
  456. package/bundled-skills/sales-enablement/references/one-pager-templates.md +0 -208
  457. package/bundled-skills/seo-audit/SKILL.md +0 -499
  458. package/bundled-skills/seo-audit/evals/evals.json +0 -136
  459. package/bundled-skills/seo-audit/references/ai-writing-detection.md +0 -200
  460. package/bundled-skills/seo-audit/references/international-seo.md +0 -230
  461. package/bundled-skills/session-brain/SKILL.md +0 -101
  462. package/bundled-skills/session-search/SKILL.md +0 -97
  463. package/bundled-skills/social/SKILL.md +0 -413
  464. package/bundled-skills/social/evals/evals.json +0 -107
  465. package/bundled-skills/social/references/carousel-frameworks.md +0 -141
  466. package/bundled-skills/social/references/listening-sources-template.md +0 -123
  467. package/bundled-skills/social/references/listening.md +0 -283
  468. package/bundled-skills/social/references/platform-limits.md +0 -110
  469. package/bundled-skills/social/references/platforms.md +0 -170
  470. package/bundled-skills/social/references/post-templates.md +0 -179
  471. package/bundled-skills/social/references/reverse-engineering.md +0 -195
  472. package/bundled-skills/social/references/short-form-video.md +0 -237
  473. package/bundled-skills/tag-taxonomy/SKILL.md +0 -218
  474. package/bundled-skills/vault-skill-factory/SKILL.md +0 -143
  475. package/bundled-skills/video/SKILL.md +0 -346
  476. package/bundled-skills/video/evals/evals.json +0 -103
  477. package/bundled-skills/video/references/ai-video-prompting.md +0 -175
  478. package/bundled-skills/video/references/edit-anatomy.md +0 -84
  479. package/bundled-skills/wiki-agent/SKILL.md +0 -322
  480. package/bundled-skills/wiki-capture/SKILL.md +0 -337
  481. package/bundled-skills/wiki-capture/references/RAW-FORMAT.md +0 -115
  482. package/bundled-skills/wiki-context-pack/SKILL.md +0 -89
  483. package/bundled-skills/wiki-dashboard/SKILL.md +0 -472
  484. package/bundled-skills/wiki-dedup/SKILL.md +0 -320
  485. package/bundled-skills/wiki-digest/SKILL.md +0 -243
  486. package/bundled-skills/wiki-export/SKILL.md +0 -445
  487. package/bundled-skills/wiki-history-ingest/SKILL.md +0 -61
  488. package/bundled-skills/wiki-import/SKILL.md +0 -276
  489. package/bundled-skills/wiki-ingest/SKILL.md +0 -563
  490. package/bundled-skills/wiki-ingest/references/ingest-prompts.md +0 -54
  491. package/bundled-skills/wiki-ingest/references/pageindex.md +0 -72
  492. package/bundled-skills/wiki-ingest/references/url-sources.md +0 -295
  493. package/bundled-skills/wiki-lint/SKILL.md +0 -643
  494. package/bundled-skills/wiki-narrate/SKILL.md +0 -120
  495. package/bundled-skills/wiki-narrate/references/voices.md +0 -29
  496. package/bundled-skills/wiki-query/SKILL.md +0 -307
  497. package/bundled-skills/wiki-rebuild/SKILL.md +0 -212
  498. package/bundled-skills/wiki-research/SKILL.md +0 -319
  499. package/bundled-skills/wiki-setup/SKILL.md +0 -359
  500. package/bundled-skills/wiki-stage-commit/SKILL.md +0 -164
  501. package/bundled-skills/wiki-status/SKILL.md +0 -556
  502. package/bundled-skills/wiki-switch/SKILL.md +0 -115
  503. package/bundled-skills/wiki-synthesize/SKILL.md +0 -212
  504. package/bundled-skills/wiki-update/SKILL.md +0 -271
  505. package/skills/ai-security/SKILL.md +0 -132
  506. package/skills/ai-security/references/atlas-coverage.md +0 -60
  507. package/skills/ai-security/scripts/ai_threat_scanner.py +0 -373
  508. package/skills/anti-AI-design/SKILL.md +0 -72
  509. package/skills/apple-hig-audit/SKILL.md +0 -64
  510. package/skills/apple-hig-audit/references/accessibility.md +0 -78
  511. package/skills/apple-hig-audit/references/platform-specifics.md +0 -57
  512. package/skills/apple-hig-audit/references/visual-design.md +0 -68
  513. package/skills/apple-hig-audit/scripts/hig_checker.py +0 -115
  514. package/skills/apple-hig-audit/templates/hig-audit-template.md +0 -114
  515. package/skills/dependency-auditor/README.md +0 -98
  516. package/skills/dependency-auditor/SKILL.md +0 -114
  517. package/skills/dependency-auditor/assets/sample_go.mod +0 -53
  518. package/skills/dependency-auditor/assets/sample_package.json +0 -72
  519. package/skills/dependency-auditor/assets/sample_requirements.txt +0 -71
  520. package/skills/dependency-auditor/expected_outputs/sample_license_report.txt +0 -37
  521. package/skills/dependency-auditor/expected_outputs/sample_upgrade_plan.txt +0 -59
  522. package/skills/dependency-auditor/expected_outputs/sample_vulnerability_report.json +0 -71
  523. package/skills/dependency-auditor/references/dependency_management_best_practices.md +0 -46
  524. package/skills/dependency-auditor/references/license_compatibility_matrix.md +0 -57
  525. package/skills/dependency-auditor/references/vulnerability_assessment_guide.md +0 -62
  526. package/skills/dependency-auditor/scripts/dep_scanner.py +0 -595
  527. package/skills/dependency-auditor/scripts/license_checker.py +0 -442
  528. package/skills/dependency-auditor/scripts/upgrade_planner.py +0 -490
  529. package/skills/dependency-auditor/test-inventory.json +0 -421
  530. package/skills/dependency-auditor/test-project/package.json +0 -72
  531. package/skills/design-system-tokens/SKILL.md +0 -90
  532. package/skills/design-system-tokens/assets/design_system_doc_template.md +0 -129
  533. package/skills/design-system-tokens/references/component-architecture.md +0 -88
  534. package/skills/design-system-tokens/references/developer-handoff.md +0 -61
  535. package/skills/design-system-tokens/references/responsive-calculations.md +0 -72
  536. package/skills/design-system-tokens/references/token-generation.md +0 -62
  537. package/skills/design-system-tokens/scripts/design_token_generator.py +0 -315
  538. package/skills/engineering-code-standards/SKILL.md +0 -174
  539. package/skills/env-secrets-manager/SKILL.md +0 -102
  540. package/skills/env-secrets-manager/references/secret-patterns.md +0 -56
  541. package/skills/env-secrets-manager/references/validation-detection-rotation.md +0 -76
  542. package/skills/env-secrets-manager/scripts/env_auditor.py +0 -340
  543. package/skills/frontend-design-taste/SKILL.md +0 -62
  544. package/skills/mcp-server-builder/README.md +0 -42
  545. package/skills/mcp-server-builder/SKILL.md +0 -144
  546. package/skills/mcp-server-builder/references/openapi-extraction-guide.md +0 -34
  547. package/skills/mcp-server-builder/references/production-hardening-guide.md +0 -80
  548. package/skills/mcp-server-builder/references/python-server-template.md +0 -22
  549. package/skills/mcp-server-builder/references/spec-compatibility.md +0 -129
  550. package/skills/mcp-server-builder/references/typescript-server-template.md +0 -19
  551. package/skills/mcp-server-builder/references/validation-checklist.md +0 -30
  552. package/skills/mcp-server-builder/scripts/mcp_validator.py +0 -180
  553. package/skills/mcp-server-builder/scripts/openapi_to_mcp.py +0 -279
  554. package/skills/playwright-agent/SKILL.md +0 -147
  555. package/skills/skill-creator/LICENSE.txt +0 -14
  556. package/skills/skill-creator/SKILL.md +0 -471
  557. package/skills/skill-creator/agents/analyzer.md +0 -274
  558. package/skills/skill-creator/agents/comparator.md +0 -202
  559. package/skills/skill-creator/agents/grader.md +0 -223
  560. package/skills/skill-creator/assets/eval_review.html +0 -143
  561. package/skills/skill-creator/eval-viewer/generate_review.py +0 -471
  562. package/skills/skill-creator/eval-viewer/viewer.html +0 -1321
  563. package/skills/skill-creator/references/schemas.md +0 -430
  564. package/skills/skill-creator/scripts/__init__.py +0 -0
  565. package/skills/skill-creator/scripts/aggregate_benchmark.py +0 -401
  566. package/skills/skill-creator/scripts/generate_report.py +0 -322
  567. package/skills/skill-creator/scripts/improve_description.py +0 -247
  568. package/skills/skill-creator/scripts/package_skill.py +0 -136
  569. package/skills/skill-creator/scripts/quick_validate.py +0 -103
  570. package/skills/skill-creator/scripts/run_eval.py +0 -310
  571. package/skills/skill-creator/scripts/run_loop.py +0 -328
  572. package/skills/skill-creator/scripts/utils.py +0 -47
  573. package/skills/write-a-skill/SKILL.md +0 -111
  574. package/skills/write-a-skill/references/companion_tooling.md +0 -61
  575. package/skills/write-a-skill/references/description_design_patterns.md +0 -136
  576. package/skills/write-a-skill/references/progressive_disclosure_principles.md +0 -87
  577. package/skills/write-a-skill/references/quality_gates_for_skills.md +0 -158
  578. package/skills/write-a-skill/scripts/skill_description_validator.py +0 -241
  579. package/skills/write-a-skill/scripts/skill_review_checklist_runner.py +0 -257
  580. package/skills/write-a-skill/scripts/skill_structure_validator.py +0 -265
@@ -1,328 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Run the eval + improve loop until all pass or max iterations reached.
3
-
4
- Combines run_eval.py and improve_description.py in a loop, tracking history
5
- and returning the best description found. Supports train/test split to prevent
6
- overfitting.
7
- """
8
-
9
- import argparse
10
- import json
11
- import random
12
- import sys
13
- import tempfile
14
- import time
15
- import webbrowser
16
- from pathlib import Path
17
-
18
- from scripts.generate_report import generate_html
19
- from scripts.improve_description import improve_description
20
- from scripts.run_eval import find_project_root, run_eval
21
- from scripts.utils import parse_skill_md
22
-
23
-
24
- def split_eval_set(eval_set: list[dict], holdout: float, seed: int = 42) -> tuple[list[dict], list[dict]]:
25
- """Split eval set into train and test sets, stratified by should_trigger."""
26
- random.seed(seed)
27
-
28
- # Separate by should_trigger
29
- trigger = [e for e in eval_set if e["should_trigger"]]
30
- no_trigger = [e for e in eval_set if not e["should_trigger"]]
31
-
32
- # Shuffle each group
33
- random.shuffle(trigger)
34
- random.shuffle(no_trigger)
35
-
36
- # Calculate split points
37
- n_trigger_test = max(1, int(len(trigger) * holdout))
38
- n_no_trigger_test = max(1, int(len(no_trigger) * holdout))
39
-
40
- # Split
41
- test_set = trigger[:n_trigger_test] + no_trigger[:n_no_trigger_test]
42
- train_set = trigger[n_trigger_test:] + no_trigger[n_no_trigger_test:]
43
-
44
- return train_set, test_set
45
-
46
-
47
- def run_loop(
48
- eval_set: list[dict],
49
- skill_path: Path,
50
- description_override: str | None,
51
- num_workers: int,
52
- timeout: int,
53
- max_iterations: int,
54
- runs_per_query: int,
55
- trigger_threshold: float,
56
- holdout: float,
57
- model: str,
58
- verbose: bool,
59
- live_report_path: Path | None = None,
60
- log_dir: Path | None = None,
61
- ) -> dict:
62
- """Run the eval + improvement loop."""
63
- project_root = find_project_root()
64
- name, original_description, content = parse_skill_md(skill_path)
65
- current_description = description_override or original_description
66
-
67
- # Split into train/test if holdout > 0
68
- if holdout > 0:
69
- train_set, test_set = split_eval_set(eval_set, holdout)
70
- if verbose:
71
- print(f"Split: {len(train_set)} train, {len(test_set)} test (holdout={holdout})", file=sys.stderr)
72
- else:
73
- train_set = eval_set
74
- test_set = []
75
-
76
- history = []
77
- exit_reason = "unknown"
78
-
79
- for iteration in range(1, max_iterations + 1):
80
- if verbose:
81
- print(f"\n{'='*60}", file=sys.stderr)
82
- print(f"Iteration {iteration}/{max_iterations}", file=sys.stderr)
83
- print(f"Description: {current_description}", file=sys.stderr)
84
- print(f"{'='*60}", file=sys.stderr)
85
-
86
- # Evaluate train + test together in one batch for parallelism
87
- all_queries = train_set + test_set
88
- t0 = time.time()
89
- all_results = run_eval(
90
- eval_set=all_queries,
91
- skill_name=name,
92
- description=current_description,
93
- num_workers=num_workers,
94
- timeout=timeout,
95
- project_root=project_root,
96
- runs_per_query=runs_per_query,
97
- trigger_threshold=trigger_threshold,
98
- model=model,
99
- )
100
- eval_elapsed = time.time() - t0
101
-
102
- # Split results back into train/test by matching queries
103
- train_queries_set = {q["query"] for q in train_set}
104
- train_result_list = [r for r in all_results["results"] if r["query"] in train_queries_set]
105
- test_result_list = [r for r in all_results["results"] if r["query"] not in train_queries_set]
106
-
107
- train_passed = sum(1 for r in train_result_list if r["pass"])
108
- train_total = len(train_result_list)
109
- train_summary = {"passed": train_passed, "failed": train_total - train_passed, "total": train_total}
110
- train_results = {"results": train_result_list, "summary": train_summary}
111
-
112
- if test_set:
113
- test_passed = sum(1 for r in test_result_list if r["pass"])
114
- test_total = len(test_result_list)
115
- test_summary = {"passed": test_passed, "failed": test_total - test_passed, "total": test_total}
116
- test_results = {"results": test_result_list, "summary": test_summary}
117
- else:
118
- test_results = None
119
- test_summary = None
120
-
121
- history.append({
122
- "iteration": iteration,
123
- "description": current_description,
124
- "train_passed": train_summary["passed"],
125
- "train_failed": train_summary["failed"],
126
- "train_total": train_summary["total"],
127
- "train_results": train_results["results"],
128
- "test_passed": test_summary["passed"] if test_summary else None,
129
- "test_failed": test_summary["failed"] if test_summary else None,
130
- "test_total": test_summary["total"] if test_summary else None,
131
- "test_results": test_results["results"] if test_results else None,
132
- # For backward compat with report generator
133
- "passed": train_summary["passed"],
134
- "failed": train_summary["failed"],
135
- "total": train_summary["total"],
136
- "results": train_results["results"],
137
- })
138
-
139
- # Write live report if path provided
140
- if live_report_path:
141
- partial_output = {
142
- "original_description": original_description,
143
- "best_description": current_description,
144
- "best_score": "in progress",
145
- "iterations_run": len(history),
146
- "holdout": holdout,
147
- "train_size": len(train_set),
148
- "test_size": len(test_set),
149
- "history": history,
150
- }
151
- live_report_path.write_text(generate_html(partial_output, auto_refresh=True, skill_name=name))
152
-
153
- if verbose:
154
- def print_eval_stats(label, results, elapsed):
155
- pos = [r for r in results if r["should_trigger"]]
156
- neg = [r for r in results if not r["should_trigger"]]
157
- tp = sum(r["triggers"] for r in pos)
158
- pos_runs = sum(r["runs"] for r in pos)
159
- fn = pos_runs - tp
160
- fp = sum(r["triggers"] for r in neg)
161
- neg_runs = sum(r["runs"] for r in neg)
162
- tn = neg_runs - fp
163
- total = tp + tn + fp + fn
164
- precision = tp / (tp + fp) if (tp + fp) > 0 else 1.0
165
- recall = tp / (tp + fn) if (tp + fn) > 0 else 1.0
166
- accuracy = (tp + tn) / total if total > 0 else 0.0
167
- print(f"{label}: {tp+tn}/{total} correct, precision={precision:.0%} recall={recall:.0%} accuracy={accuracy:.0%} ({elapsed:.1f}s)", file=sys.stderr)
168
- for r in results:
169
- status = "PASS" if r["pass"] else "FAIL"
170
- rate_str = f"{r['triggers']}/{r['runs']}"
171
- print(f" [{status}] rate={rate_str} expected={r['should_trigger']}: {r['query'][:60]}", file=sys.stderr)
172
-
173
- print_eval_stats("Train", train_results["results"], eval_elapsed)
174
- if test_summary:
175
- print_eval_stats("Test ", test_results["results"], 0)
176
-
177
- if train_summary["failed"] == 0:
178
- exit_reason = f"all_passed (iteration {iteration})"
179
- if verbose:
180
- print(f"\nAll train queries passed on iteration {iteration}!", file=sys.stderr)
181
- break
182
-
183
- if iteration == max_iterations:
184
- exit_reason = f"max_iterations ({max_iterations})"
185
- if verbose:
186
- print(f"\nMax iterations reached ({max_iterations}).", file=sys.stderr)
187
- break
188
-
189
- # Improve the description based on train results
190
- if verbose:
191
- print(f"\nImproving description...", file=sys.stderr)
192
-
193
- t0 = time.time()
194
- # Strip test scores from history so improvement model can't see them
195
- blinded_history = [
196
- {k: v for k, v in h.items() if not k.startswith("test_")}
197
- for h in history
198
- ]
199
- new_description = improve_description(
200
- skill_name=name,
201
- skill_content=content,
202
- current_description=current_description,
203
- eval_results=train_results,
204
- history=blinded_history,
205
- model=model,
206
- log_dir=log_dir,
207
- iteration=iteration,
208
- )
209
- improve_elapsed = time.time() - t0
210
-
211
- if verbose:
212
- print(f"Proposed ({improve_elapsed:.1f}s): {new_description}", file=sys.stderr)
213
-
214
- current_description = new_description
215
-
216
- # Find the best iteration by TEST score (or train if no test set)
217
- if test_set:
218
- best = max(history, key=lambda h: h["test_passed"] or 0)
219
- best_score = f"{best['test_passed']}/{best['test_total']}"
220
- else:
221
- best = max(history, key=lambda h: h["train_passed"])
222
- best_score = f"{best['train_passed']}/{best['train_total']}"
223
-
224
- if verbose:
225
- print(f"\nExit reason: {exit_reason}", file=sys.stderr)
226
- print(f"Best score: {best_score} (iteration {best['iteration']})", file=sys.stderr)
227
-
228
- return {
229
- "exit_reason": exit_reason,
230
- "original_description": original_description,
231
- "best_description": best["description"],
232
- "best_score": best_score,
233
- "best_train_score": f"{best['train_passed']}/{best['train_total']}",
234
- "best_test_score": f"{best['test_passed']}/{best['test_total']}" if test_set else None,
235
- "final_description": current_description,
236
- "iterations_run": len(history),
237
- "holdout": holdout,
238
- "train_size": len(train_set),
239
- "test_size": len(test_set),
240
- "history": history,
241
- }
242
-
243
-
244
- def main():
245
- parser = argparse.ArgumentParser(description="Run eval + improve loop")
246
- parser.add_argument("--eval-set", required=True, help="Path to eval set JSON file")
247
- parser.add_argument("--skill-path", required=True, help="Path to skill directory")
248
- parser.add_argument("--description", default=None, help="Override starting description")
249
- parser.add_argument("--num-workers", type=int, default=10, help="Number of parallel workers")
250
- parser.add_argument("--timeout", type=int, default=30, help="Timeout per query in seconds")
251
- parser.add_argument("--max-iterations", type=int, default=5, help="Max improvement iterations")
252
- parser.add_argument("--runs-per-query", type=int, default=3, help="Number of runs per query")
253
- parser.add_argument("--trigger-threshold", type=float, default=0.5, help="Trigger rate threshold")
254
- parser.add_argument("--holdout", type=float, default=0.4, help="Fraction of eval set to hold out for testing (0 to disable)")
255
- parser.add_argument("--model", required=True, help="Model for improvement")
256
- parser.add_argument("--verbose", action="store_true", help="Print progress to stderr")
257
- parser.add_argument("--report", default="auto", help="Generate HTML report at this path (default: 'auto' for temp file, 'none' to disable)")
258
- parser.add_argument("--results-dir", default=None, help="Save all outputs (results.json, report.html, log.txt) to a timestamped subdirectory here")
259
- args = parser.parse_args()
260
-
261
- eval_set = json.loads(Path(args.eval_set).read_text())
262
- skill_path = Path(args.skill_path)
263
-
264
- if not (skill_path / "SKILL.md").exists():
265
- print(f"Error: No SKILL.md found at {skill_path}", file=sys.stderr)
266
- sys.exit(1)
267
-
268
- name, _, _ = parse_skill_md(skill_path)
269
-
270
- # Set up live report path
271
- if args.report != "none":
272
- if args.report == "auto":
273
- timestamp = time.strftime("%Y%m%d_%H%M%S")
274
- live_report_path = Path(tempfile.gettempdir()) / f"skill_description_report_{skill_path.name}_{timestamp}.html"
275
- else:
276
- live_report_path = Path(args.report)
277
- # Open the report immediately so the user can watch
278
- live_report_path.write_text("<html><body><h1>Starting optimization loop...</h1><meta http-equiv='refresh' content='5'></body></html>")
279
- webbrowser.open(str(live_report_path))
280
- else:
281
- live_report_path = None
282
-
283
- # Determine output directory (create before run_loop so logs can be written)
284
- if args.results_dir:
285
- timestamp = time.strftime("%Y-%m-%d_%H%M%S")
286
- results_dir = Path(args.results_dir) / timestamp
287
- results_dir.mkdir(parents=True, exist_ok=True)
288
- else:
289
- results_dir = None
290
-
291
- log_dir = results_dir / "logs" if results_dir else None
292
-
293
- output = run_loop(
294
- eval_set=eval_set,
295
- skill_path=skill_path,
296
- description_override=args.description,
297
- num_workers=args.num_workers,
298
- timeout=args.timeout,
299
- max_iterations=args.max_iterations,
300
- runs_per_query=args.runs_per_query,
301
- trigger_threshold=args.trigger_threshold,
302
- holdout=args.holdout,
303
- model=args.model,
304
- verbose=args.verbose,
305
- live_report_path=live_report_path,
306
- log_dir=log_dir,
307
- )
308
-
309
- # Save JSON output
310
- json_output = json.dumps(output, indent=2)
311
- print(json_output)
312
- if results_dir:
313
- (results_dir / "results.json").write_text(json_output)
314
-
315
- # Write final HTML report (without auto-refresh)
316
- if live_report_path:
317
- live_report_path.write_text(generate_html(output, auto_refresh=False, skill_name=name))
318
- print(f"\nReport: {live_report_path}", file=sys.stderr)
319
-
320
- if results_dir and live_report_path:
321
- (results_dir / "report.html").write_text(generate_html(output, auto_refresh=False, skill_name=name))
322
-
323
- if results_dir:
324
- print(f"Results saved to: {results_dir}", file=sys.stderr)
325
-
326
-
327
- if __name__ == "__main__":
328
- main()
@@ -1,47 +0,0 @@
1
- """Shared utilities for skill-creator scripts."""
2
-
3
- from pathlib import Path
4
-
5
-
6
-
7
- def parse_skill_md(skill_path: Path) -> tuple[str, str, str]:
8
- """Parse a SKILL.md file, returning (name, description, full_content)."""
9
- content = (skill_path / "SKILL.md").read_text()
10
- lines = content.split("\n")
11
-
12
- if lines[0].strip() != "---":
13
- raise ValueError("SKILL.md missing frontmatter (no opening ---)")
14
-
15
- end_idx = None
16
- for i, line in enumerate(lines[1:], start=1):
17
- if line.strip() == "---":
18
- end_idx = i
19
- break
20
-
21
- if end_idx is None:
22
- raise ValueError("SKILL.md missing frontmatter (no closing ---)")
23
-
24
- name = ""
25
- description = ""
26
- frontmatter_lines = lines[1:end_idx]
27
- i = 0
28
- while i < len(frontmatter_lines):
29
- line = frontmatter_lines[i]
30
- if line.startswith("name:"):
31
- name = line[len("name:"):].strip().strip('"').strip("'")
32
- elif line.startswith("description:"):
33
- value = line[len("description:"):].strip()
34
- # Handle YAML multiline indicators (>, |, >-, |-)
35
- if value in (">", "|", ">-", "|-"):
36
- continuation_lines: list[str] = []
37
- i += 1
38
- while i < len(frontmatter_lines) and (frontmatter_lines[i].startswith(" ") or frontmatter_lines[i].startswith("\t")):
39
- continuation_lines.append(frontmatter_lines[i].strip())
40
- i += 1
41
- description = " ".join(continuation_lines)
42
- continue
43
- else:
44
- description = value.strip('"').strip("'")
45
- i += 1
46
-
47
- return name, description, content
@@ -1,111 +0,0 @@
1
- ---
2
- name: write-a-skill
3
- description: Create new agent skills with proper structure, progressive disclosure, and bundled resources. Use when user wants to create, write, build, or author a new skill.
4
- license: Apache-2.0
5
- metadata:
6
- author: Novahiz
7
- organization: Novahiz
8
- version: "2.0.0"
9
- date: September 2026
10
- ---
11
-
12
- # Writing skills
13
-
14
- ## Process
15
-
16
- 1. **Clarify the need.** Ask about the domain, the concrete cases the skill must handle, whether scripts are required, and what reference material should ship with it.
17
- 2. **Draft the files.** Write SKILL.md first. Push anything past 100 lines into a reference file. Add scripts only for deterministic work the model should not re-derive each time.
18
- 3. **Review with the author.** Present the draft. Ask what is missing, what is unclear, and which sections are too heavy or too thin.
19
-
20
- ## Layout
21
-
22
- ```
23
- skill-name/
24
- ├── SKILL.md # required entry point
25
- ├── REFERENCE.md # optional long-form detail
26
- ├── EXAMPLES.md # optional worked cases
27
- └── scripts/ # optional helpers
28
- └── helper.js
29
- ```
30
-
31
- ## SKILL.md skeleton
32
-
33
- ```md
34
- ---
35
- name: skill-name
36
- description: What the skill does. Use when [specific triggers].
37
- ---
38
-
39
- # Skill name
40
-
41
- ## Quick start
42
-
43
- [Minimal working example]
44
-
45
- ## Workflows
46
-
47
- [Step-by-step processes for complex tasks]
48
-
49
- ## Advanced features
50
-
51
- [See REFERENCE.md](REFERENCE.md)
52
- ```
53
-
54
- ## Description rules
55
-
56
- The description is the only text the agent sees when it decides whether to load the skill. It sits in the system prompt next to every other installed skill, and the agent picks from it alone.
57
-
58
- Two facts must land:
59
-
60
- 1. What the skill does.
61
- 2. When to fire: keywords, contexts, file types.
62
-
63
- Constraints:
64
-
65
- - Hard cap at 1024 characters.
66
- - Third person throughout.
67
- - Sentence one: the action.
68
- - Sentence two: `Use when [specific triggers]`.
69
-
70
- Good:
71
-
72
- ```
73
- Extract text and tables from PDF files, fill forms, merge documents. Use when working with PDF files or when user mentions PDFs, forms, or document extraction.
74
- ```
75
-
76
- Bad:
77
-
78
- ```
79
- Helps with documents.
80
- ```
81
-
82
- The weak version gives the agent nothing to match against. Every document skill in the list looks the same from it.
83
-
84
- ## When scripts belong in the skill
85
-
86
- Ship a script when the work is deterministic (validation, formatting), when the same code would otherwise be generated over and over, or when failure needs explicit handling. Scripts cut token use and remove the chance of two runs inventing two different implementations.
87
-
88
- ## When to split the file
89
-
90
- Split when SKILL.md crosses 100 lines, when the content covers distinct domains (finance schemas next to sales schemas), or when advanced paths are rarely needed. The agent should read SKILL.md in full and stop there unless a pointer sends it deeper.
91
-
92
- ## Review checklist
93
-
94
- - [ ] Description carries triggers (`Use when ...`)
95
- - [ ] SKILL.md stays under 100 lines
96
- - [ ] No dates, versions, or `as of YYYY` claims
97
- - [ ] One word per concept throughout
98
- - [ ] At least one concrete example
99
- - [ ] References sit one level deep
100
-
101
- ## Tooling
102
-
103
- Three stdlib validators sit next to this skill:
104
-
105
- ```
106
- python scripts/skill_description_validator.py path/to/SKILL.md
107
- python scripts/skill_structure_validator.py path/to/skill-folder
108
- python scripts/skill_review_checklist_runner.py path/to/skill-folder
109
- ```
110
-
111
- Catalogue and companions: [references/companion_tooling.md](references/companion_tooling.md).
@@ -1,61 +0,0 @@
1
- # Companion Tooling
2
-
3
- Validation tools layered on top of the core skill-authoring process. Use these when authoring a new skill in this repo.
4
-
5
- ## Validation Tools (stdlib Python)
6
-
7
- | Tool | Purpose | Run before |
8
- |---|---|---|
9
- | `scripts/skill_description_validator.py` | Validates description: ≤1024 chars, third person, "Use when" trigger, action verb in first sentence | First draft of SKILL.md |
10
- | `scripts/skill_structure_validator.py` | Validates folder structure: SKILL.md present, ≤100 lines, references one level deep, no circular refs | Pre-commit |
11
- | `scripts/skill_review_checklist_runner.py` | Runs all 6 review-checklist items against a skill folder | Final check before PR |
12
-
13
- All three tools:
14
- - Stdlib-only (no external dependencies)
15
- - Run with embedded sample if no path provided
16
- - Output text or JSON (`--output json`)
17
- - Exit code: 0 if PASS, 1 if FAIL/WARN
18
-
19
- ## cs-skill-author Persona Agent
20
-
21
- Lives at `../agents/cs-skill-author.md`. Voice: forcing-question interrogator. Surfaces the skill-authoring workflow as an interrogation before any new skill commit.
22
-
23
- **Opening question:** "What capability does this skill provide, and what's the trigger phrase that distinguishes it from existing skills?"
24
-
25
- **Six forcing questions** (matches the review checklist):
26
- 1. What's the description? Is it ≤1024 chars + third person + has "Use when ..."?
27
- 2. Is SKILL.md under 100 lines? If not, where will the split land (REFERENCE.md / EXAMPLES.md / references/)?
28
- 3. Are there time-sensitive claims (dates, "as of YYYY")?
29
- 4. Is terminology consistent: same word for the same concept throughout?
30
- 5. Concrete examples: at least 1 code block, ideally good/bad contrast?
31
- 6. References one level deep, no circular refs?
32
-
33
- ## `/cs:write-a-skill` Slash Command
34
-
35
- Lives at `../commands/cs-write-a-skill.md`. Three-step flow:
36
-
37
- 1. Run `cs-skill-author` interrogation (6 questions)
38
- 2. Draft skill files per the structure pattern
39
- 3. Run all 3 validation tools; show verdict; fix until PASS
40
-
41
- Use when: starting a new skill in this repo from scratch.
42
-
43
- ## Why Wrap the Core Process
44
-
45
- The core write-a-skill process is a tight, principled skill: perfect as-is for individual authoring sessions. The wrapper layers add three things this repo benefits from at scale:
46
-
47
- 1. **Programmatic enforcement** of the review checklist (the validation tools): prevents human review-checklist drift across 100+ skills.
48
- 2. **Forcing-question interrogation** (the cs-skill-author persona): adapts the "review with user" phase to the cs-* persona pattern used elsewhere in this repo.
49
- 3. **Citation-backed references**: the core process links to principles; the wrapper adds 5+ authoritative external sources per reference for newcomers learning the pattern.
50
-
51
- ---
52
-
53
- **Source authorities (non-exhaustive):**
54
-
55
- - **Novahiz skill-authoring standard** (this repo): the core process
56
- - **Skills documentation**: official guidance on skill structure
57
- - **Engineering blog: Skills patterns** (continuously updated): patterns for skill authoring
58
- - **Karpathy, A.: "Software 3.0" + LLM coding pitfalls** (X.com posts 2024-2025): discipline reference applied throughout this repo's karpathy-coder skill
59
- - **Pareto principle applied to documentation**: concise = trustworthy; 80% of value in 20% of words
60
- - **Hyrum's Law** as applied to skill descriptions: once a description shape is observed, downstream agents depend on it
61
- - **Conway's Law as applied to skill libraries**: skill organization mirrors team responsibilities; progressive disclosure mirrors information needs across team boundaries
@@ -1,136 +0,0 @@
1
- # Description Design Patterns for Skills
2
-
3
- This reference answers exactly one decision: how do we write a skill description that an agent actually picks correctly when faced with a long skill list?
4
-
5
- Pair with `scripts/skill_description_validator.py` for automated enforcement.
6
-
7
- ## The Core Rule
8
-
9
- The description is the only thing your agent sees when deciding which skill to load.
10
-
11
- Implication: the description is not marketing copy. It is a routing signal for the agent. Every word competes with every other skill's description for activation attention.
12
-
13
- ## The Four Format Rules
14
-
15
- 1. **Max 1024 chars**: beyond this, agents lose the early sentences when condensing context
16
- 2. **Third person**: first-person ("I help with...") confuses agent self-identification; second-person ("You can...") confuses pronoun reference
17
- 3. **First sentence: what it does**: front-load the verb + object
18
- 4. **Second sentence: "Use when [specific triggers]"**: agent's most reliable activation cue
19
-
20
- ## Good vs Bad Examples
21
-
22
- **Good:**
23
-
24
- ```
25
- Extract text and tables from PDF files, fill forms, merge documents. Use when working with PDF files or when user mentions PDFs, forms, or document extraction.
26
- ```
27
-
28
- **Why good:**
29
- - Front-loaded verbs: Extract, fill, merge
30
- - Concrete objects: text, tables, PDF files, forms
31
- - Explicit trigger: "Use when working with PDF files"
32
- - Specific keywords for matching: "PDFs", "forms", "document extraction"
33
-
34
- **Bad:**
35
-
36
- ```
37
- Helps with documents.
38
- ```
39
-
40
- **Why bad:**
41
- - "Helps" is content-free
42
- - "Documents" is generic: every doc skill has this
43
- - No trigger
44
- - No keyword variety
45
-
46
- **Bad in a different way (over-specified):**
47
-
48
- ```
49
- This skill performs comprehensive PDF document processing including but not limited to extraction, manipulation, format conversion, content analysis, metadata management, and security operations on PDF files, with support for various PDF versions and embedded media types.
50
- ```
51
-
52
- **Why bad:** verbose, no triggers, agent can't extract the key keywords from the wall of text.
53
-
54
- ## The Trigger Sentence Pattern
55
-
56
- The "Use when" sentence is the highest-impact part of the description. Patterns that work:
57
-
58
- **Keyword triggers** (when user types specific words):
59
- ```
60
- Use when user mentions PDFs, forms, or document extraction.
61
- ```
62
-
63
- **File-type triggers** (when agent sees specific files):
64
- ```
65
- Use when working with `.tsx` files or React component tests.
66
- ```
67
-
68
- **Context triggers** (when agent is in a specific state):
69
- ```
70
- Use when the user requests a code review of a pull request.
71
- ```
72
-
73
- **Workflow triggers** (when agent is mid-workflow):
74
- ```
75
- Use after running tests and before committing changes.
76
- ```
77
-
78
- ## Vocabulary Selection
79
-
80
- The description's words must overlap with words users + agents naturally use for the task.
81
-
82
- | Bad keyword | Better keyword | Why |
83
- |---|---|---|
84
- | "documents" | "PDF files" / "Word docs" | More specific = less collision |
85
- | "improve" | "refactor" / "fix" / "optimize" | Specific verb = clearer routing |
86
- | "various" | (delete; just list them) | Hedge language = no info |
87
- | "modern" | (cite the actual tool/version) | Trend words age badly |
88
- | "comprehensive" | (delete; just list capabilities) | Adjective inflation |
89
-
90
- ## Length Optimization
91
-
92
- Below 1024 chars, shorter is usually better. Target: 100-300 chars for most skills.
93
-
94
- Where complexity demands more chars, prioritize:
95
- 1. The verb-object pair (what it does): never compress
96
- 2. The trigger phrase: never compress
97
- 3. Keyword variety (different ways users describe it): expand here if space allows
98
- 4. Anti-keyword (what it does NOT do): only if there's a frequently-confused sibling skill
99
-
100
- ## Anti-Patterns to Avoid
101
-
102
- 1. **First-person voice**: "I extract PDFs": confuses agent self-reference
103
- 2. **Marketing language**: "fast, powerful, intuitive": agent doesn't care, ignores adjectives
104
- 3. **Trigger-less descriptions**: every skill needs "Use when X"
105
- 4. **Multi-purpose dumping**: if your skill does 10 unrelated things, it's probably 10 skills
106
- 5. **Pronouns and hedges**: "you can also use this if you want to": drop entirely
107
- 6. **Recursive descriptions**: "Use this skill when you need this skill": adds nothing
108
- 7. **Implementation details**: "Built on Python + stdlib": agent doesn't care; matters for README, not description
109
-
110
- ## Pre-Commit Discipline
111
-
112
- Run before every skill PR:
113
-
114
- ```bash
115
- python scripts/skill_description_validator.py path/to/SKILL.md
116
- ```
117
-
118
- If validator returns FAIL, fix before merging. If WARN, justify and document the trade-off.
119
-
120
- ## When This Reference Doesn't Help
121
-
122
- - **Naming the skill itself**: different concern; see naming-conventions guidance per-repo
123
- - **Skill discovery in marketplaces**: different audience (humans browsing), different rules
124
- - **System-prompt design for the agent that loads skills**: separate concern
125
-
126
- ---
127
-
128
- **Source authorities (non-exhaustive):**
129
-
130
- - **Novahiz authoring standard** (this repo): the 4 format rules + good/bad example pattern
131
- - **Skill registry documentation**: how a skill-loader uses descriptions for routing
132
- - **Effective system prompt writing guides**: same principles applied to system-prompt design
133
- - **Karpathy, A.: public commentary on LLM prompt design**: emphasis on specificity + lack of ambiguity
134
- - **Garrett, J.J.: "The Elements of User Experience"** (2002) + information architecture principles: labels must match user mental models
135
- - **Nielsen Norman Group: Microcontent guidelines**: applies to skill descriptions: front-load value, hard-cap length, scannable structure
136
- - **Search-engine + SEO patterns adapted for agent routing**: keyword density, intent matching, semantic field coverage