dsh-aris-panel 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (442) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +98 -0
  3. package/README_CN.md +87 -0
  4. package/dsh/checkout.patch.yml +38 -0
  5. package/dsh/client.js +634 -0
  6. package/dsh/cordis.patch.yml +44 -0
  7. package/dsh/index.mjs +76 -0
  8. package/dsh/run-status.mjs +182 -0
  9. package/dsh/scope-limits.mjs +50 -0
  10. package/dsh/workbench.mjs +291 -0
  11. package/mcp-servers/claude-review/README.md +93 -0
  12. package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
  13. package/mcp-servers/claude-review/server.py +718 -0
  14. package/mcp-servers/codex-image2/README.md +65 -0
  15. package/mcp-servers/codex-image2/server.py +893 -0
  16. package/mcp-servers/feishu-bridge/requirements.txt +1 -0
  17. package/mcp-servers/feishu-bridge/server.py +240 -0
  18. package/mcp-servers/gemini-review/README.md +171 -0
  19. package/mcp-servers/gemini-review/server.py +1856 -0
  20. package/mcp-servers/llm-chat/requirements.txt +1 -0
  21. package/mcp-servers/llm-chat/server.py +664 -0
  22. package/mcp-servers/manual-review/README.md +133 -0
  23. package/mcp-servers/manual-review/server.py +910 -0
  24. package/mcp-servers/manual-review/ui.html +279 -0
  25. package/mcp-servers/minimax-chat/requirements.txt +1 -0
  26. package/mcp-servers/minimax-chat/server.py +381 -0
  27. package/package.json +51 -0
  28. package/skills/ablation-planner/SKILL.md +123 -0
  29. package/skills/alphaxiv/SKILL.md +196 -0
  30. package/skills/analyze-results/SKILL.md +46 -0
  31. package/skills/arxiv/SKILL.md +248 -0
  32. package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
  33. package/skills/auto-review-loop/SKILL.md +1137 -0
  34. package/skills/auto-review-loop-llm/SKILL.md +259 -0
  35. package/skills/auto-review-loop-minimax/SKILL.md +302 -0
  36. package/skills/citation-audit/SKILL.md +502 -0
  37. package/skills/claims-drafting/SKILL.md +227 -0
  38. package/skills/comm-lit-review/SKILL.md +297 -0
  39. package/skills/deepxiv/SKILL.md +263 -0
  40. package/skills/dse-loop/SKILL.md +296 -0
  41. package/skills/embodiment-description/SKILL.md +129 -0
  42. package/skills/exa-search/SKILL.md +205 -0
  43. package/skills/experiment-audit/SKILL.md +311 -0
  44. package/skills/experiment-bridge/SKILL.md +376 -0
  45. package/skills/experiment-plan/SKILL.md +249 -0
  46. package/skills/experiment-queue/SKILL.md +431 -0
  47. package/skills/experiment-queue/scripts/build_manifest.py +142 -0
  48. package/skills/experiment-queue/scripts/queue_manager.py +433 -0
  49. package/skills/feishu-notify/SKILL.md +156 -0
  50. package/skills/figure-description/SKILL.md +138 -0
  51. package/skills/figure-spec/SKILL.md +262 -0
  52. package/skills/figure-spec/scripts/figure_renderer.py +799 -0
  53. package/skills/formula-derivation/SKILL.md +280 -0
  54. package/skills/gemini-search/SKILL.md +231 -0
  55. package/skills/grant-proposal/SKILL.md +698 -0
  56. package/skills/idea-creator/SKILL.md +542 -0
  57. package/skills/idea-discovery/SKILL.md +521 -0
  58. package/skills/idea-discovery-robot/SKILL.md +363 -0
  59. package/skills/integrity-forensics/SKILL.md +284 -0
  60. package/skills/interview-cheatsheet/SKILL.md +245 -0
  61. package/skills/invention-structuring/SKILL.md +188 -0
  62. package/skills/jurisdiction-format/SKILL.md +192 -0
  63. package/skills/kill-argument/SKILL.md +437 -0
  64. package/skills/mermaid-diagram/SKILL.md +419 -0
  65. package/skills/meta-apply/SKILL.md +141 -0
  66. package/skills/meta-optimize/SKILL.md +437 -0
  67. package/skills/monitor-experiment/SKILL.md +140 -0
  68. package/skills/novelty-check/SKILL.md +101 -0
  69. package/skills/openalex/SKILL.md +237 -0
  70. package/skills/overleaf-sync/SKILL.md +220 -0
  71. package/skills/paper-claim-audit/SKILL.md +348 -0
  72. package/skills/paper-compile/SKILL.md +266 -0
  73. package/skills/paper-figure/SKILL.md +312 -0
  74. package/skills/paper-illustration/SKILL.md +736 -0
  75. package/skills/paper-illustration-image2/SKILL.md +391 -0
  76. package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  77. package/skills/paper-plan/SKILL.md +386 -0
  78. package/skills/paper-poster/SKILL.md +19 -0
  79. package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
  80. package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
  81. package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
  82. package/skills/paper-poster-html/NOTICE.md +57 -0
  83. package/skills/paper-poster-html/SKILL.md +323 -0
  84. package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
  85. package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
  86. package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
  87. package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
  88. package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
  89. package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
  90. package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
  91. package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
  92. package/skills/paper-poster-html/scripts/asset_check.py +897 -0
  93. package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
  94. package/skills/paper-poster-html/scripts/poster_check.py +251 -0
  95. package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
  96. package/skills/paper-poster-html/scripts/render_preview.py +217 -0
  97. package/skills/paper-poster-html/scripts/run_gates.py +556 -0
  98. package/skills/paper-poster-html/scripts/style_check.py +1324 -0
  99. package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
  100. package/skills/paper-poster-html/templates/README.md +170 -0
  101. package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
  102. package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
  103. package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
  104. package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
  105. package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
  106. package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
  107. package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
  108. package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
  109. package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
  110. package/skills/paper-slides/SKILL.md +635 -0
  111. package/skills/paper-talk/SKILL.md +381 -0
  112. package/skills/paper-write/SKILL.md +604 -0
  113. package/skills/paper-write/templates/IEEEtran.bst +2409 -0
  114. package/skills/paper-write/templates/IEEEtran.cls +6347 -0
  115. package/skills/paper-write/templates/iclr2026.tex +84 -0
  116. package/skills/paper-write/templates/icml2025.tex +87 -0
  117. package/skills/paper-write/templates/ieee_conference.tex +89 -0
  118. package/skills/paper-write/templates/ieee_journal.tex +93 -0
  119. package/skills/paper-write/templates/math_commands.tex +48 -0
  120. package/skills/paper-write/templates/neurips2025.tex +80 -0
  121. package/skills/paper-writing/SKILL.md +916 -0
  122. package/skills/patent-novelty-check/SKILL.md +153 -0
  123. package/skills/patent-pipeline/SKILL.md +344 -0
  124. package/skills/patent-review/SKILL.md +203 -0
  125. package/skills/pixel-art/SKILL.md +137 -0
  126. package/skills/prior-art-search/SKILL.md +146 -0
  127. package/skills/proof-checker/SKILL.md +866 -0
  128. package/skills/proof-orchestrator/NOTICE.md +24 -0
  129. package/skills/proof-orchestrator/SKILL.md +254 -0
  130. package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
  131. package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
  132. package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
  133. package/skills/proof-orchestrator/references/notation-audit.md +135 -0
  134. package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  135. package/skills/proof-orchestrator/references/stress-tests.md +38 -0
  136. package/skills/proof-writer/SKILL.md +223 -0
  137. package/skills/qzcli/SKILL.md +324 -0
  138. package/skills/rebuttal/SKILL.md +376 -0
  139. package/skills/render-html/SKILL.md +316 -0
  140. package/skills/render-html/scripts/render_html.py +1006 -0
  141. package/skills/render-html/scripts/templates/academic.html +703 -0
  142. package/skills/render-html/scripts/templates/dashboard.html +333 -0
  143. package/skills/research-lit/SKILL.md +756 -0
  144. package/skills/research-pipeline/SKILL.md +384 -0
  145. package/skills/research-refine/SKILL.md +770 -0
  146. package/skills/research-refine-pipeline/SKILL.md +186 -0
  147. package/skills/research-review/SKILL.md +198 -0
  148. package/skills/research-wiki/SKILL.md +461 -0
  149. package/skills/resubmit-pipeline/SKILL.md +447 -0
  150. package/skills/result-to-claim/SKILL.md +311 -0
  151. package/skills/run-experiment/SKILL.md +313 -0
  152. package/skills/semantic-scholar/SKILL.md +236 -0
  153. package/skills/serverless-modal/SKILL.md +335 -0
  154. package/skills/shared-references/acceptance-gate.md +324 -0
  155. package/skills/shared-references/assurance-contract.md +248 -0
  156. package/skills/shared-references/capture-antipatterns.md +78 -0
  157. package/skills/shared-references/citation-discipline.md +583 -0
  158. package/skills/shared-references/compute-env-contract.md +163 -0
  159. package/skills/shared-references/effort-contract.md +183 -0
  160. package/skills/shared-references/evidence-precheck.md +65 -0
  161. package/skills/shared-references/experiment-integrity.md +49 -0
  162. package/skills/shared-references/external-cadence.md +326 -0
  163. package/skills/shared-references/fan-out-pattern.md +366 -0
  164. package/skills/shared-references/injection-hygiene.md +127 -0
  165. package/skills/shared-references/integration-contract.md +461 -0
  166. package/skills/shared-references/output-composition.md +93 -0
  167. package/skills/shared-references/output-language.md +45 -0
  168. package/skills/shared-references/output-manifest.md +49 -0
  169. package/skills/shared-references/output-versioning.md +111 -0
  170. package/skills/shared-references/patent-format-cn.md +199 -0
  171. package/skills/shared-references/patent-format-ep.md +173 -0
  172. package/skills/shared-references/patent-format-us.md +161 -0
  173. package/skills/shared-references/patent-writing-principles.md +197 -0
  174. package/skills/shared-references/prior-art-databases.md +141 -0
  175. package/skills/shared-references/resumable-runs.md +109 -0
  176. package/skills/shared-references/review-scope-limits.md +81 -0
  177. package/skills/shared-references/review-tracing.md +391 -0
  178. package/skills/shared-references/reviewer-independence.md +79 -0
  179. package/skills/shared-references/reviewer-routing.md +852 -0
  180. package/skills/shared-references/skill-governance.md +104 -0
  181. package/skills/shared-references/taste-calibration.md +85 -0
  182. package/skills/shared-references/venue-checklists.md +114 -0
  183. package/skills/shared-references/wiki-helper-resolution.md +134 -0
  184. package/skills/shared-references/writing-principles.md +525 -0
  185. package/skills/skills-codex/README.md +102 -0
  186. package/skills/skills-codex/README_CN.md +100 -0
  187. package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
  188. package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
  189. package/skills/skills-codex/analyze-results/SKILL.md +45 -0
  190. package/skills/skills-codex/arxiv/SKILL.md +210 -0
  191. package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
  192. package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
  193. package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
  194. package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
  195. package/skills/skills-codex/citation-audit/SKILL.md +504 -0
  196. package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
  197. package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
  198. package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
  199. package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
  200. package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
  201. package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
  202. package/skills/skills-codex/deepxiv/SKILL.md +142 -0
  203. package/skills/skills-codex/dse-loop/SKILL.md +285 -0
  204. package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
  205. package/skills/skills-codex/exa-search/SKILL.md +192 -0
  206. package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
  207. package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
  208. package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
  209. package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
  210. package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
  211. package/skills/skills-codex/figure-description/SKILL.md +138 -0
  212. package/skills/skills-codex/figure-spec/SKILL.md +252 -0
  213. package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
  214. package/skills/skills-codex/gemini-search/SKILL.md +205 -0
  215. package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
  216. package/skills/skills-codex/idea-creator/SKILL.md +405 -0
  217. package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
  218. package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
  219. package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
  220. package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
  221. package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
  222. package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
  223. package/skills/skills-codex/kill-argument/SKILL.md +403 -0
  224. package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
  225. package/skills/skills-codex/meta-apply/SKILL.md +154 -0
  226. package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
  227. package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
  228. package/skills/skills-codex/novelty-check/SKILL.md +89 -0
  229. package/skills/skills-codex/openalex/SKILL.md +228 -0
  230. package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
  231. package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
  232. package/skills/skills-codex/paper-compile/SKILL.md +253 -0
  233. package/skills/skills-codex/paper-figure/SKILL.md +311 -0
  234. package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
  235. package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
  236. package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  237. package/skills/skills-codex/paper-plan/SKILL.md +278 -0
  238. package/skills/skills-codex/paper-poster/SKILL.md +19 -0
  239. package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
  240. package/skills/skills-codex/paper-slides/SKILL.md +571 -0
  241. package/skills/skills-codex/paper-talk/SKILL.md +381 -0
  242. package/skills/skills-codex/paper-write/SKILL.md +411 -0
  243. package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
  244. package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
  245. package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
  246. package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
  247. package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
  248. package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
  249. package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
  250. package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
  251. package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
  252. package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
  253. package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
  254. package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
  255. package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
  256. package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
  257. package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
  258. package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
  259. package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
  260. package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
  261. package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
  262. package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
  263. package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
  264. package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
  265. package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
  266. package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
  267. package/skills/skills-codex/paper-writing/SKILL.md +731 -0
  268. package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
  269. package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
  270. package/skills/skills-codex/patent-review/SKILL.md +202 -0
  271. package/skills/skills-codex/pixel-art/SKILL.md +139 -0
  272. package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
  273. package/skills/skills-codex/proof-checker/SKILL.md +554 -0
  274. package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
  275. package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
  276. package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
  277. package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
  278. package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
  279. package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  280. package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
  281. package/skills/skills-codex/proof-writer/SKILL.md +222 -0
  282. package/skills/skills-codex/qzcli/SKILL.md +324 -0
  283. package/skills/skills-codex/rebuttal/SKILL.md +305 -0
  284. package/skills/skills-codex/render-html/SKILL.md +305 -0
  285. package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
  286. package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
  287. package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
  288. package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
  289. package/skills/skills-codex/research-lit/SKILL.md +464 -0
  290. package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
  291. package/skills/skills-codex/research-refine/SKILL.md +721 -0
  292. package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
  293. package/skills/skills-codex/research-review/SKILL.md +135 -0
  294. package/skills/skills-codex/research-wiki/SKILL.md +421 -0
  295. package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
  296. package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
  297. package/skills/skills-codex/run-experiment/SKILL.md +236 -0
  298. package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
  299. package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
  300. package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
  301. package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
  302. package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
  303. package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
  304. package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
  305. package/skills/skills-codex/shared-references/effort-contract.md +143 -0
  306. package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
  307. package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
  308. package/skills/skills-codex/shared-references/external-cadence.md +334 -0
  309. package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
  310. package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
  311. package/skills/skills-codex/shared-references/integration-contract.md +372 -0
  312. package/skills/skills-codex/shared-references/output-composition.md +98 -0
  313. package/skills/skills-codex/shared-references/output-language.md +45 -0
  314. package/skills/skills-codex/shared-references/output-manifest.md +40 -0
  315. package/skills/skills-codex/shared-references/output-versioning.md +111 -0
  316. package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
  317. package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
  318. package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
  319. package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
  320. package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
  321. package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
  322. package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
  323. package/skills/skills-codex/shared-references/review-tracing.md +144 -0
  324. package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
  325. package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
  326. package/skills/skills-codex/shared-references/skill-governance.md +119 -0
  327. package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
  328. package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
  329. package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
  330. package/skills/skills-codex/shared-references/writing-principles.md +525 -0
  331. package/skills/skills-codex/slides-polish/SKILL.md +563 -0
  332. package/skills/skills-codex/specification-writing/SKILL.md +211 -0
  333. package/skills/skills-codex/system-profile/SKILL.md +103 -0
  334. package/skills/skills-codex/training-check/SKILL.md +83 -0
  335. package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
  336. package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
  337. package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
  338. package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
  339. package/skills/skills-codex-claude-review/README.md +79 -0
  340. package/skills/skills-codex-claude-review/README_CN.md +78 -0
  341. package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
  342. package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
  343. package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
  344. package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
  345. package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
  346. package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
  347. package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
  348. package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
  349. package/skills/skills-codex-gemini-review/README.md +176 -0
  350. package/skills/skills-codex-gemini-review/README_CN.md +175 -0
  351. package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
  352. package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
  353. package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
  354. package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
  355. package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
  356. package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
  357. package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
  358. package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
  359. package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
  360. package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
  361. package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
  362. package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
  363. package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
  364. package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
  365. package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
  366. package/skills/slides-polish/SKILL.md +565 -0
  367. package/skills/specification-writing/SKILL.md +211 -0
  368. package/skills/system-profile/SKILL.md +103 -0
  369. package/skills/training-check/SKILL.md +132 -0
  370. package/skills/vast-gpu/SKILL.md +394 -0
  371. package/skills/web-debug-search/SKILL.md +334 -0
  372. package/skills/wiki-enrich/SKILL.md +257 -0
  373. package/skills/writing-systems-papers/SKILL.md +184 -0
  374. package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
  375. package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
  376. package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
  377. package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
  378. package/templates/FINDINGS_TEMPLATE.md +52 -0
  379. package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
  380. package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
  381. package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
  382. package/templates/MANIFEST_TEMPLATE.md +7 -0
  383. package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
  384. package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
  385. package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
  386. package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
  387. package/templates/README.md +57 -0
  388. package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
  389. package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
  390. package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
  391. package/templates/claude-hooks/corpus_write_guard.json +16 -0
  392. package/templates/claude-hooks/corpus_write_guard.py +85 -0
  393. package/templates/claude-hooks/meta_logging.json +74 -0
  394. package/templates/gitignore-trace.txt +3 -0
  395. package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
  396. package/tools/arxiv_fetch.py +311 -0
  397. package/tools/capture_filter.py +126 -0
  398. package/tools/check_skills_inventory.py +273 -0
  399. package/tools/convert_skills_to_llm_chat.py +282 -0
  400. package/tools/copilot_native_evidence.py +818 -0
  401. package/tools/deepxiv_fetch.py +213 -0
  402. package/tools/evidence_check.py +212 -0
  403. package/tools/exa_search.py +425 -0
  404. package/tools/experiment_queue/README.md +118 -0
  405. package/tools/experiment_queue/build_manifest.py +44 -0
  406. package/tools/experiment_queue/queue_manager.py +44 -0
  407. package/tools/extract_paper_style.py +560 -0
  408. package/tools/figure_renderer.py +69 -0
  409. package/tools/forensics_gate.py +669 -0
  410. package/tools/generate_codex_claude_review_overrides.py +299 -0
  411. package/tools/idea_discovery_gate.py +256 -0
  412. package/tools/install_aris.ps1 +1372 -0
  413. package/tools/install_aris.sh +1370 -0
  414. package/tools/install_aris_codex.sh +1023 -0
  415. package/tools/install_aris_copilot.sh +1052 -0
  416. package/tools/iteration_log.py +143 -0
  417. package/tools/lint_skills_helpers.sh +84 -0
  418. package/tools/meta_opt/check_ready.sh +80 -0
  419. package/tools/meta_opt/log_event.sh +91 -0
  420. package/tools/meta_opt/trigger_eval.py +280 -0
  421. package/tools/meta_opt/trigger_evals.sample.json +28 -0
  422. package/tools/openalex_fetch.py +326 -0
  423. package/tools/overleaf_audit.sh +104 -0
  424. package/tools/overleaf_setup.sh +150 -0
  425. package/tools/paper_illustration_image2.py +62 -0
  426. package/tools/provenance.py +294 -0
  427. package/tools/research_wiki.py +1720 -0
  428. package/tools/review_gate.py +502 -0
  429. package/tools/run_state.py +399 -0
  430. package/tools/save_trace.sh +477 -0
  431. package/tools/semantic_scholar_fetch.py +438 -0
  432. package/tools/skill-groups.tsv +116 -0
  433. package/tools/skill_picker.py +238 -0
  434. package/tools/smart_update.ps1 +521 -0
  435. package/tools/smart_update.sh +591 -0
  436. package/tools/smart_update_codex.sh +419 -0
  437. package/tools/smart_update_copilot.sh +605 -0
  438. package/tools/threat_scan.py +222 -0
  439. package/tools/verify_paper_audits.sh +487 -0
  440. package/tools/verify_papers.py +613 -0
  441. package/tools/verify_wiki_coverage.sh +176 -0
  442. package/tools/watchdog.py +485 -0
@@ -0,0 +1,143 @@
1
+ #!/usr/bin/env python3
2
+ """iteration_log.py — overnight-loop stall detection → forced structural pivot.
3
+
4
+ Append-only per-iteration ledger for an unattended research loop. Each tick the
5
+ orchestrator records how many NEW findings the iteration produced — where a "finding"
6
+ is a concrete added entry (new evidence, a falsified hypothesis, a candidate direction),
7
+ NOT a subjective "valuable result". Consecutive zero-finding iterations accumulate a
8
+ stale_count, which drives a forced pivot:
9
+
10
+ stale_count >= 2 → pivot = "structural" (change a STRUCTURAL constraint, not tactical params)
11
+ stale_count >= 4 → pivot = "human" (flag for human attention)
12
+
13
+ This is a **Type-A signal**: it COUNTS entries and changes *direction*; it does NOT judge
14
+ quality — quality/correctness stays with the cross-model jury (shared-references/
15
+ acceptance-gate.md). It only ever says "keep going / change direction," never "good enough".
16
+
17
+ The ledger is a sidecar at `.aris/runs/<run_id>.iterations.jsonl`; it deliberately does
18
+ NOT import or touch run_state.py's done/accepted state machine (only shares the `.aris/runs/`
19
+ dir, with a distinct `.iterations.jsonl` suffix). An optional `direction` per record lets
20
+ the loop's re-generation step reject candidates too close to a tried direction. See
21
+ shared-references/external-cadence.md → "Stall detection & forced structural pivot".
22
+
23
+ Usage:
24
+ python3 iteration_log.py note <root> <run_id> <phase> <new_findings> [--direction "..."]
25
+ python3 iteration_log.py show <root> <run_id>
26
+ """
27
+ from __future__ import annotations
28
+
29
+ import argparse
30
+ import json
31
+ import sys
32
+ from contextlib import contextmanager
33
+ from datetime import datetime, timezone
34
+ from pathlib import Path
35
+ from typing import Iterator, Optional
36
+
37
+ try:
38
+ import fcntl
39
+ except ImportError: # pragma: no cover - non-POSIX
40
+ fcntl = None # type: ignore
41
+
42
+ PIVOT_STRUCTURAL_AT = 2 # consecutive zero-finding iterations → force a structural pivot
43
+ ESCALATE_HUMAN_AT = 4 # still stalled → flag for human attention
44
+
45
+
46
+ def _now() -> str:
47
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
48
+
49
+
50
+ def _log_path(root: str, run_id: str) -> Path:
51
+ # Same run_id discipline as run_state.py: no path escape.
52
+ safe = "".join(c for c in run_id if c.isalnum() or c in "-_.")
53
+ if not safe or safe != run_id or run_id in (".", ".."):
54
+ raise ValueError(f"invalid run_id {run_id!r} (use [A-Za-z0-9-_.])")
55
+ return Path(root) / ".aris" / "runs" / f"{run_id}.iterations.jsonl"
56
+
57
+
58
+ @contextmanager
59
+ def _lock(path: Path) -> Iterator[None]:
60
+ """Best-effort advisory lock (single-orchestrator contract; guards a stray resumer)."""
61
+ path.parent.mkdir(parents=True, exist_ok=True)
62
+ if fcntl is None:
63
+ yield
64
+ return
65
+ fh = open(path.with_suffix(".jsonl.lock"), "w")
66
+ try:
67
+ fcntl.flock(fh, fcntl.LOCK_EX)
68
+ yield
69
+ finally:
70
+ try:
71
+ fcntl.flock(fh, fcntl.LOCK_UN)
72
+ finally:
73
+ fh.close()
74
+
75
+
76
+ def _last_stale(path: Path) -> int:
77
+ """Read the most recent stale_count from the append-only ledger (0 if none)."""
78
+ if not path.is_file():
79
+ return 0
80
+ last = 0
81
+ try:
82
+ for line in path.read_text(encoding="utf-8").splitlines():
83
+ line = line.strip()
84
+ if not line:
85
+ continue
86
+ try:
87
+ last = int(json.loads(line).get("stale_count", last))
88
+ except (json.JSONDecodeError, ValueError, TypeError):
89
+ continue # tolerate a partial/garbled line, keep the last good count
90
+ except OSError:
91
+ return 0
92
+ return last
93
+
94
+
95
+ def pivot_for(stale_count: int) -> str:
96
+ if stale_count >= ESCALATE_HUMAN_AT:
97
+ return "human"
98
+ if stale_count >= PIVOT_STRUCTURAL_AT:
99
+ return "structural"
100
+ return "none"
101
+
102
+
103
+ def note(root: str, run_id: str, phase: str, new_findings: int,
104
+ direction: Optional[str] = None) -> dict:
105
+ """Record one iteration; return {stale_count, pivot}. Append-only; never blocks work."""
106
+ new_findings = int(new_findings)
107
+ if new_findings < 0:
108
+ raise ValueError(f"new_findings must be >= 0, got {new_findings}")
109
+ path = _log_path(root, run_id)
110
+ with _lock(path):
111
+ stale_count = 0 if new_findings > 0 else _last_stale(path) + 1
112
+ pivot = pivot_for(stale_count)
113
+ rec = {"ts": _now(), "phase": phase, "new_findings": new_findings,
114
+ "stale_count": stale_count, "pivot": pivot}
115
+ if direction is not None:
116
+ rec["direction"] = direction
117
+ with open(path, "a", encoding="utf-8") as f:
118
+ f.write(json.dumps(rec, ensure_ascii=False) + "\n")
119
+ return {"stale_count": stale_count, "pivot": pivot}
120
+
121
+
122
+ def show(root: str, run_id: str) -> str:
123
+ path = _log_path(root, run_id)
124
+ return path.read_text(encoding="utf-8") if path.is_file() else ""
125
+
126
+
127
+ def main() -> int:
128
+ ap = argparse.ArgumentParser(description="overnight-loop stall detection → forced structural pivot")
129
+ sub = ap.add_subparsers(dest="cmd", required=True)
130
+ n = sub.add_parser("note")
131
+ n.add_argument("root"); n.add_argument("run_id"); n.add_argument("phase")
132
+ n.add_argument("new_findings", type=int); n.add_argument("--direction", default=None)
133
+ s = sub.add_parser("show"); s.add_argument("root"); s.add_argument("run_id")
134
+ a = ap.parse_args()
135
+ if a.cmd == "note":
136
+ print(json.dumps(note(a.root, a.run_id, a.phase, a.new_findings, a.direction)))
137
+ elif a.cmd == "show":
138
+ sys.stdout.write(show(a.root, a.run_id))
139
+ return 0
140
+
141
+
142
+ if __name__ == "__main__":
143
+ raise SystemExit(main())
@@ -0,0 +1,84 @@
1
+ #!/usr/bin/env bash
2
+ # lint_skills_helpers.sh — Advisory lint for hardcoded `tools/<helper>` references.
3
+ #
4
+ # Per shared-references/integration-contract.md §2, SKILL.md files must
5
+ # resolve helpers via the canonical strict-safe chain
6
+ # .aris/tools/<helper> → tools/<helper> → $ARIS_REPO/tools/<helper>
7
+ # → $ARIS_REPO/tools/<helper> via the global pointer file ~/.aris/repo (#366)
8
+ # (Codex mirror uses the mirror-side chain), NOT hardcode `python3 tools/foo.py`
9
+ # or `bash tools/foo.sh` directly.
10
+ #
11
+ # This script is ADVISORY: it always exits 0 and only prints findings.
12
+ # A future enforcement layer (issue #178) may fail CI on new violations,
13
+ # but Phase 2 keeps the contract gentle so the maintainer is not blocked.
14
+ #
15
+ # Run from the ARIS repo root:
16
+ # bash tools/lint_skills_helpers.sh
17
+
18
+ set -u
19
+
20
+ # Patterns that indicate hardcoded helper invocation (no resolver).
21
+ INVOCATION_PY='python3 tools/(verify_papers|extract_paper_style|paper_illustration_image2|figure_renderer|arxiv_fetch|semantic_scholar_fetch|deepxiv_fetch|exa_search|openalex_fetch|research_wiki|iteration_log)\.py'
22
+ INVOCATION_SH='bash tools/(verify_paper_audits|save_trace|verify_wiki_coverage|overleaf_audit)\.sh'
23
+
24
+ # Files exempted from the lint:
25
+ # - integration-contract.md (canonical docs include ❌ anti-pattern examples)
26
+ # - wiki-helper-resolution.md (defines the chain; layer-2 reference is intentional)
27
+ # - skills-codex/paper-writing/SKILL.md L525 hook JSON example (placeholder for user
28
+ # ~/.claude/settings.json or ~/.codex/config hook, not a SKILL bash block)
29
+ EXEMPTIONS="\
30
+ skills/shared-references/integration-contract.md
31
+ skills/skills-codex/shared-references/integration-contract.md
32
+ skills/shared-references/wiki-helper-resolution.md
33
+ skills/skills-codex/shared-references/wiki-helper-resolution.md
34
+ skills/skills-codex/paper-writing/SKILL.md"
35
+
36
+ is_exempt() {
37
+ case "$EXEMPTIONS" in
38
+ *"$1"*) return 0 ;;
39
+ esac
40
+ return 1
41
+ }
42
+
43
+ violation_count=0
44
+ violation_report=""
45
+
46
+ while IFS= read -r f; do
47
+ if is_exempt "$f"; then
48
+ continue
49
+ fi
50
+ py_hits=$(grep -nE "$INVOCATION_PY" "$f" 2>/dev/null || true)
51
+ sh_hits=$(grep -nE "$INVOCATION_SH" "$f" 2>/dev/null || true)
52
+ if [ -n "$py_hits" ] || [ -n "$sh_hits" ]; then
53
+ violation_count=$((violation_count + 1))
54
+ violation_report="${violation_report}
55
+ === $f ==="
56
+ [ -n "$py_hits" ] && violation_report="${violation_report}
57
+ ${py_hits}"
58
+ [ -n "$sh_hits" ] && violation_report="${violation_report}
59
+ ${sh_hits}"
60
+ fi
61
+ done < <(find skills -name '*.md' -type f 2>/dev/null)
62
+
63
+ echo "ARIS helper-resolution lint (advisory)"
64
+ echo "======================================="
65
+ echo "Files with hardcoded \`tools/<helper>\` references: $violation_count"
66
+
67
+ if [ "$violation_count" -gt 0 ]; then
68
+ printf '%s\n\n' "$violation_report"
69
+ echo "Resolution:"
70
+ echo " Migrate each violating SKILL.md to the canonical strict-safe resolver"
71
+ echo " per shared-references/integration-contract.md §2 (assign a semantic"
72
+ echo " variable like \$AUDIT_VERIFIER / \$TRACE_HELPER / \$<NAME>_FETCHER from"
73
+ echo " the four-layer chain, then invoke as \`python3 \"\$VAR\" ...\` or"
74
+ echo " \`bash \"\$VAR\" ...\`)."
75
+ echo ""
76
+ echo " Per-helper policy (Policy A gate / B side-effect / C forensic /"
77
+ echo " D1 cascade / D2 multi-source / E diagnostic) is documented in the"
78
+ echo " \"Per-helper policy assignments\" table of integration-contract.md §2."
79
+ fi
80
+
81
+ echo ""
82
+ echo "Status: advisory (this script never fails CI; warnings only)."
83
+
84
+ exit 0
@@ -0,0 +1,80 @@
1
+ #!/usr/bin/env bash
2
+ # ARIS Meta-Optimize: Readiness Check
3
+ # Called by SessionEnd hook. If enough data has accumulated,
4
+ # outputs a reminder to stdout (injected into Claude's context).
5
+ #
6
+ # Trigger: ≥5 skill invocations since last /meta-optimize run
7
+
8
+ set -euo pipefail
9
+
10
+ ARIS_META_DIR="${CLAUDE_PROJECT_DIR:-.}/.aris/meta"
11
+ EVENTS_FILE="$ARIS_META_DIR/events.jsonl"
12
+ LAST_RUN_FILE="$ARIS_META_DIR/.last_optimize"
13
+
14
+ # No log = nothing to check
15
+ [ -f "$EVENTS_FILE" ] || exit 0
16
+
17
+ # Count skill invocations. Match the full event-field key so a stray
18
+ # "skill_invoke" substring inside args/prompt values can't false-match.
19
+ # Note: `grep -c` prints "0" then exits 1 on no match, so `... || echo 0`
20
+ # would produce a two-line string "0\n0" that breaks the later integer
21
+ # comparison. Use `|| true` to absorb the non-zero exit and keep the
22
+ # captured "0" alone.
23
+ TOTAL_SKILLS=$(grep -cE '"event": *"skill_invoke"' "$EVENTS_FILE" 2>/dev/null || true)
24
+ TOTAL_SKILLS=${TOTAL_SKILLS:-0}
25
+
26
+ # Check when meta-optimize was last run
27
+ if [ -f "$LAST_RUN_FILE" ]; then
28
+ LAST_TS=$(cat "$LAST_RUN_FILE")
29
+ # Count skill invocations AFTER last run.
30
+ # WARNING: do NOT use `$0 > ts` here — every JSONL row starts with `{`
31
+ # (ASCII 0x7B), which sorts greater than every digit in an ISO 8601
32
+ # timestamp, so the comparison would degenerate to "always true" and
33
+ # SINCE_LAST would equal TOTAL_SKILLS. Extract the embedded "ts" value
34
+ # and compare that instead. log_event.sh emits Python json.dumps default
35
+ # format (`"ts": "..."` with a space after the colon), so the regex
36
+ # tolerates 0+ spaces between key and value to stay compatible if that
37
+ # ever changes.
38
+ SINCE_LAST=$(awk -v ts="$LAST_TS" '
39
+ /"event": *"skill_invoke"/ {
40
+ if (match($0, /"ts": *"[^"]+"/)) {
41
+ event_ts = substr($0, RSTART, RLENGTH)
42
+ sub(/^"ts": *"/, "", event_ts)
43
+ sub(/"$/, "", event_ts)
44
+ if (event_ts > ts) count++
45
+ }
46
+ }
47
+ END { print count + 0 }
48
+ ' "$EVENTS_FILE")
49
+ else
50
+ SINCE_LAST=$TOTAL_SKILLS
51
+ fi
52
+
53
+ # Threshold: 5 skill invocations since last optimize
54
+ if [ "$SINCE_LAST" -ge 5 ]; then
55
+ echo "📊 ARIS has logged $SINCE_LAST skill runs since last optimization. Run /meta-optimize to check for improvement opportunities."
56
+ fi
57
+
58
+ # Model-delta trigger (harness diet): a model bump makes existing scaffolding a
59
+ # deletion candidate, independent of usage volume. /meta-optimize records the
60
+ # session model at each run in .last_optimize_model; compare against the LATEST
61
+ # session_start event's model. Absent files (older installs, no session_start
62
+ # yet) silently skip.
63
+ LAST_MODEL_FILE="$ARIS_META_DIR/.last_optimize_model"
64
+ if [ -f "$LAST_MODEL_FILE" ]; then
65
+ CURRENT_MODEL=$(awk '
66
+ /"event": *"session_start"/ {
67
+ if (match($0, /"model": *"[^"]+"/)) {
68
+ m = substr($0, RSTART, RLENGTH)
69
+ sub(/^"model": *"/, "", m)
70
+ sub(/"$/, "", m)
71
+ latest = m
72
+ }
73
+ }
74
+ END { if (latest != "") print latest }
75
+ ' "$EVENTS_FILE")
76
+ LAST_MODEL=$(cat "$LAST_MODEL_FILE")
77
+ if [ -n "$CURRENT_MODEL" ] && [ -n "$LAST_MODEL" ] && [ "$CURRENT_MODEL" != "$LAST_MODEL" ]; then
78
+ echo "🔁 Model changed since last optimization ($LAST_MODEL → $CURRENT_MODEL). Run /meta-optimize — a model bump makes existing scaffolding a deletion candidate (harness diet)."
79
+ fi
80
+ fi
@@ -0,0 +1,91 @@
1
+ #!/usr/bin/env bash
2
+ # ARIS Meta-Optimize: Event Logger
3
+ # Reads Claude Code hook JSON from stdin, extracts key fields,
4
+ # appends structured event to BOTH project-level and global logs.
5
+ #
6
+ # Called automatically by Claude Code hooks (PostToolUse, UserPromptSubmit, etc.)
7
+ # Input: JSON via stdin (Claude Code hook payload)
8
+ # Output:
9
+ # Project: $CLAUDE_PROJECT_DIR/.aris/meta/events.jsonl (project-specific details)
10
+ # Global: ~/.aris/meta/events.jsonl (cross-project trends)
11
+
12
+ set -euo pipefail
13
+
14
+ PROJECT_META="${CLAUDE_PROJECT_DIR:-.}/.aris/meta"
15
+ GLOBAL_META="$HOME/.aris/meta"
16
+ mkdir -p "$PROJECT_META" "$GLOBAL_META"
17
+
18
+ # Read stdin payload into env var (cannot use heredoc + herestring simultaneously)
19
+ export ARIS_HOOK_PAYLOAD="$(cat)"
20
+
21
+ python3 - "$PROJECT_META/events.jsonl" "$GLOBAL_META/events.jsonl" << 'PYEOF'
22
+ import json, sys, os
23
+ from datetime import datetime, timezone
24
+
25
+ project_log = sys.argv[1]
26
+ global_log = sys.argv[2]
27
+
28
+ raw = os.environ.get("ARIS_HOOK_PAYLOAD", "").strip()
29
+ if not raw:
30
+ sys.exit(0)
31
+ try:
32
+ p = json.loads(raw)
33
+ except json.JSONDecodeError:
34
+ sys.exit(0)
35
+
36
+ event_name = p.get("hook_event_name", "unknown")
37
+ session_id = p.get("session_id", "")
38
+ project_dir = os.environ.get("CLAUDE_PROJECT_DIR", "")
39
+ ts = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
40
+
41
+ record = {"ts": ts, "session": session_id, "event": event_name}
42
+
43
+ if event_name in ("PostToolUse", "PostToolUseFailure"):
44
+ tool_name = p.get("tool_name", "")
45
+ tool_input = p.get("tool_input", {})
46
+ record["tool"] = tool_name
47
+
48
+ if event_name == "PostToolUseFailure":
49
+ record["event"] = "tool_failure"
50
+
51
+ if tool_name == "Skill":
52
+ record["event"] = "skill_invoke"
53
+ record["skill"] = tool_input.get("skill", "")
54
+ record["args"] = tool_input.get("args", "")
55
+ elif tool_name == "Bash":
56
+ record["input_summary"] = tool_input.get("command", "")[:200]
57
+ elif tool_name in ("Edit", "Write", "Read"):
58
+ record["input_summary"] = tool_input.get("file_path", "")
59
+ elif tool_name.startswith("mcp__codex__"):
60
+ record["event"] = "codex_call"
61
+ record["input_summary"] = tool_input.get("prompt", "")[:150]
62
+
63
+ elif event_name == "UserPromptSubmit":
64
+ prompt = p.get("prompt", "")
65
+ if prompt.startswith("/"):
66
+ parts = prompt.split(None, 1)
67
+ record["event"] = "slash_command"
68
+ record["command"] = parts[0]
69
+ record["args"] = parts[1] if len(parts) > 1 else ""
70
+ else:
71
+ record["event"] = "user_prompt"
72
+ record["prompt_preview"] = prompt[:100]
73
+
74
+ elif event_name == "SessionStart":
75
+ record["event"] = "session_start"
76
+ record["source"] = p.get("source", "")
77
+ record["model"] = p.get("model", "")
78
+
79
+ elif event_name == "SessionEnd":
80
+ record["event"] = "session_end"
81
+
82
+ # Write to project-level log
83
+ with open(project_log, "a") as f:
84
+ f.write(json.dumps(record, ensure_ascii=False) + "\n")
85
+
86
+ # Write to global log with project tag
87
+ global_record = record.copy()
88
+ global_record["project"] = os.path.basename(project_dir) if project_dir else "unknown"
89
+ with open(global_log, "a") as f:
90
+ f.write(json.dumps(global_record, ensure_ascii=False) + "\n")
91
+ PYEOF
@@ -0,0 +1,280 @@
1
+ #!/usr/bin/env python3
2
+ """trigger_eval.py — measure whether a skill's `description` actually triggers.
3
+
4
+ ARIS's known pain: with 80+ skills installed, Claude Code sometimes fails to
5
+ invoke the right skill for a query it should handle — and until now the only
6
+ lever (the frontmatter `description`) was tuned by pure judgment, with zero
7
+ measurement. This tool turns trigger behavior into a number.
8
+
9
+ MEASURE-ONLY BY DESIGN. It never rewrites a description. Its report is
10
+ EVIDENCE for a /meta-optimize proposal (which lands only via the human-gated
11
+ /meta-apply) — "a loop can drive, never acquit" applies to description tuning
12
+ too.
13
+
14
+ How it works (adapted from Anthropic's Claude Science `skill-creator`
15
+ run_eval.py — Apache-2.0; ported off its host.* runtime onto plain `claude -p`):
16
+ - For each (skill, query), run `claude -p <query> --output-format stream-json
17
+ --max-turns 1 --permission-mode plan --disallowed-tools Bash Write Edit …`
18
+ as a subprocess FROM A NEUTRAL TEMP CWD. The user-level ~/.claude/skills
19
+ corpus is loaded as usual, so the measurement happens under the REALISTIC
20
+ long installed list — the exact condition under which omission happens (an
21
+ isolated one-skill sandbox would trivially inflate trigger rates).
22
+ - Parse the stream for the first assistant turn's tool_use blocks. A `Skill`
23
+ tool call with input.skill == target counts as a TRIGGER; a Skill call for a
24
+ different skill is a CONFUSION (recorded by name — the confusion matrix is
25
+ the interesting part for the long-list problem); no Skill call is a MISS.
26
+ Reading the target's SKILL.md via the Read tool counts as a trigger too
27
+ (secondary signal).
28
+ - SAFETY: `--permission-mode plan` blocks every side-effecting tool (Bash,
29
+ Write, Edit, …) from executing, so a probed skill's own commands (e.g.
30
+ check-gpu's ssh, vast-gpu's rentals) do NOT run — we observe only which tool
31
+ the model REACHED FOR. The read-only tools we score on (a `Skill` load, a
32
+ `Read` of a SKILL.md) may execute, and both are side-effect-free.
33
+ `--disallowed-tools` denies the stateful tools explicitly as belt-and-braces,
34
+ and `--no-session-persistence` avoids leaving session artifacts. This is a
35
+ measurement, not a sandbox — it does not stop the user's own SessionStart
36
+ hooks (their normal per-session behavior), it stops the PROBED WORK.
37
+
38
+ Query-set methodology (see trigger_evals.sample.json): queries must PARAPHRASE
39
+ user intent, never quote the description's own trigger phrases verbatim — a
40
+ query containing the literal trigger string is trivially positive and measures
41
+ nothing. Optional negative queries (expect: none) measure false-triggering.
42
+
43
+ Usage:
44
+ python3 tools/meta_opt/trigger_eval.py --eval-file tools/meta_opt/trigger_evals.sample.json \\
45
+ [--skills check-gpu,research-lit] [--samples 2] [--model haiku] \\
46
+ [--out .aris/meta/trigger_report.json] [--timeout 120]
47
+
48
+ Exit code: 0 on completed run (regardless of rates), 2 on setup error.
49
+ """
50
+
51
+ import argparse
52
+ import json
53
+ import os
54
+ import subprocess
55
+ import sys
56
+ import tempfile
57
+ from pathlib import Path
58
+
59
+
60
+ # ---------------------------------------------------------------- pure logic
61
+
62
+ def parse_stream_tool_uses(stream_text: str):
63
+ """Extract (tool_name, tool_input) pairs from `claude -p` stream-json output.
64
+
65
+ Each line is a JSON event; assistant events carry message.content lists in
66
+ which tool_use blocks appear. Malformed lines are skipped (the stream can
67
+ interleave non-JSON stderr noise when things go wrong).
68
+ """
69
+ uses = []
70
+ for line in stream_text.splitlines():
71
+ line = line.strip()
72
+ if not line or not line.startswith("{"):
73
+ continue
74
+ try:
75
+ ev = json.loads(line)
76
+ except json.JSONDecodeError:
77
+ continue
78
+ if ev.get("type") != "assistant":
79
+ continue
80
+ content = (ev.get("message") or {}).get("content") or []
81
+ for block in content:
82
+ if isinstance(block, dict) and block.get("type") == "tool_use":
83
+ uses.append((block.get("name") or "", block.get("input") or {}))
84
+ return uses
85
+
86
+
87
+ def classify(tool_uses, target_skill: str):
88
+ """Classify one probe run: ('trigger'|'confusion'|'miss', detail).
89
+
90
+ Trigger: a Skill call for the target (exact id, or a `plugin:target`
91
+ namespaced form — the latter tagged in detail so a namespaced match is
92
+ never silently indistinguishable from an exact one), or a Read of the
93
+ target's SKILL.md.
94
+ Confusion: the FIRST Skill call named a different skill (detail = its name).
95
+ Miss: no skill engagement at all.
96
+ """
97
+ for name, inp in tool_uses:
98
+ if name == "Skill":
99
+ invoked = (inp.get("skill") or "").strip()
100
+ if invoked == target_skill:
101
+ return "trigger", invoked
102
+ # plugin-namespaced form "plugin:skill": a trigger only if the tail
103
+ # equals the target AND the target itself is bare (not namespaced),
104
+ # surfaced distinctly so a human can spot a plugin/bare collision.
105
+ if ":" in invoked and invoked.split(":")[-1] == target_skill \
106
+ and ":" not in target_skill:
107
+ return "trigger", f"{invoked} (namespaced→{target_skill})"
108
+ return "confusion", invoked
109
+ if name == "Read":
110
+ path = str(inp.get("file_path") or "")
111
+ if f"/skills/{target_skill}/SKILL.md" in path:
112
+ return "trigger", path
113
+ return "miss", ""
114
+
115
+
116
+ def aggregate(records):
117
+ """records: list of {skill, query, outcome, detail} → per-skill summary."""
118
+ out = {}
119
+ for r in records:
120
+ s = out.setdefault(r["skill"], {
121
+ "probes": 0, "triggers": 0, "misses": 0, "errors": 0,
122
+ "confusions": {}, "queries": {},
123
+ })
124
+ s["probes"] += 1
125
+ q = s["queries"].setdefault(r["query"], {"trigger": 0, "confusion": 0,
126
+ "miss": 0, "error": 0})
127
+ q[r["outcome"]] += 1
128
+ if r["outcome"] == "trigger":
129
+ s["triggers"] += 1
130
+ elif r["outcome"] == "miss":
131
+ s["misses"] += 1
132
+ elif r["outcome"] == "error":
133
+ s["errors"] += 1
134
+ elif r["outcome"] == "confusion":
135
+ s["confusions"][r["detail"]] = s["confusions"].get(r["detail"], 0) + 1
136
+ for s in out.values():
137
+ graded = s["probes"] - s["errors"]
138
+ s["trigger_rate"] = round(s["triggers"] / graded, 3) if graded else None
139
+ return out
140
+
141
+
142
+ # ------------------------------------------------------------------- probing
143
+
144
+ # Stateful tools that must never execute during a probe (belt-and-braces on top
145
+ # of --permission-mode plan, which already blocks side-effecting tools).
146
+ _DENY_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch"]
147
+
148
+
149
+ # `--max-turns 1` deliberately caps the probe at one turn, so the CLI ends with
150
+ # result subtype `error_max_turns` and a NONZERO exit — that is the EXPECTED,
151
+ # successful termination for a probe, NOT a failure. Only other errors (auth,
152
+ # startup/hook failure, execution error) count as a real error.
153
+ _EXPECTED_TERMINATION = "error_max_turns"
154
+
155
+
156
+ def _stream_real_error(stream_text: str) -> bool:
157
+ """True iff the stream carries a genuine terminal error — an `is_error`
158
+ result whose subtype is NOT the expected max-turns cap. Auth/hook failures
159
+ that still emit JSON are caught here so they are graded `error`, never a
160
+ `miss` that would silently corrupt the trigger rate."""
161
+ for line in stream_text.splitlines():
162
+ line = line.strip()
163
+ if not line.startswith("{"):
164
+ continue
165
+ try:
166
+ ev = json.loads(line)
167
+ except json.JSONDecodeError:
168
+ continue
169
+ if ev.get("type") == "result" and ev.get("is_error") \
170
+ and ev.get("subtype") != _EXPECTED_TERMINATION:
171
+ return True
172
+ return False
173
+
174
+
175
+ def _stream_has_assistant(stream_text: str) -> bool:
176
+ """True iff the model produced at least one assistant turn — i.e. the probe
177
+ ran far enough to be gradeable (even if it then hit the max-turns cap)."""
178
+ for line in stream_text.splitlines():
179
+ line = line.strip()
180
+ if not line.startswith("{"):
181
+ continue
182
+ try:
183
+ if json.loads(line).get("type") == "assistant":
184
+ return True
185
+ except json.JSONDecodeError:
186
+ continue
187
+ return False
188
+
189
+
190
+ def run_probe(query: str, model: str | None, timeout: int, cwd: str) -> str:
191
+ """One `claude -p` probe; returns raw stream-json text. Raises RuntimeError
192
+ on a REAL failure (genuine error event, or nonzero exit with no assistant
193
+ turn at all) so the caller records `error` rather than a rate-corrupting
194
+ `miss`. The expected max-turns termination (nonzero exit + assistant turn
195
+ present) is a normal, gradeable result."""
196
+ cmd = ["claude", "-p", "--output-format", "stream-json", "--verbose",
197
+ "--max-turns", "1", "--permission-mode", "plan",
198
+ "--no-session-persistence", "--disallowed-tools", *_DENY_TOOLS]
199
+ if model:
200
+ cmd += ["--model", model]
201
+ # Allow nesting claude -p inside a Claude Code session (same pattern as the
202
+ # Apache-2.0 source): the CLAUDECODE guard is for interactive terminals.
203
+ env = {k: v for k, v in os.environ.items() if k != "CLAUDECODE"}
204
+ result = subprocess.run(cmd, input=query, capture_output=True, text=True,
205
+ env=env, timeout=timeout, cwd=cwd)
206
+ if _stream_real_error(result.stdout):
207
+ raise RuntimeError("claude -p stream carried a terminal error result event")
208
+ if _stream_has_assistant(result.stdout):
209
+ return result.stdout # gradeable (max-turns cap is fine)
210
+ if result.returncode != 0: # no assistant turn AND failed = real
211
+ raise RuntimeError(f"claude -p exited {result.returncode} with no assistant "
212
+ f"turn: {result.stderr.strip()[:300]}")
213
+ return result.stdout # clean, no tool call → graded miss
214
+
215
+
216
+ def main(argv=None) -> int:
217
+ ap = argparse.ArgumentParser(description="Measure skill-description trigger rates.")
218
+ ap.add_argument("--eval-file", required=True,
219
+ help='JSON: {"<skill>": ["query", ...], ...}')
220
+ ap.add_argument("--skills", default="",
221
+ help="comma-separated subset of skills to probe (default: all in file)")
222
+ ap.add_argument("--samples", type=int, default=3,
223
+ help="probes per query (trigger behavior is stochastic; the "
224
+ "default 3 matches the upstream eval — samples=1 is too "
225
+ "noisy to act on)")
226
+ ap.add_argument("--model", default=None,
227
+ help="model override for probes (default: claude CLI default). "
228
+ "NB: trigger behavior is model-dependent — compare like with like.")
229
+ ap.add_argument("--timeout", type=int, default=120)
230
+ ap.add_argument("--out", default=".aris/meta/trigger_report.json")
231
+ args = ap.parse_args(argv)
232
+
233
+ try:
234
+ evals = json.loads(Path(args.eval_file).read_text(encoding="utf-8"))
235
+ except (OSError, json.JSONDecodeError) as e:
236
+ print(f"ERROR: cannot read eval file: {e}", file=sys.stderr)
237
+ return 2
238
+ subset = {s.strip() for s in args.skills.split(",") if s.strip()}
239
+ targets = {k: v for k, v in evals.items()
240
+ if (not subset or k in subset) and not k.startswith("_")}
241
+ if not targets:
242
+ print("ERROR: no skills selected", file=sys.stderr)
243
+ return 2
244
+
245
+ records = []
246
+ # Neutral cwd: no project-level .claude/, so probes see exactly the
247
+ # user-level installed corpus — the realistic long list.
248
+ with tempfile.TemporaryDirectory(prefix="trigger-eval-") as neutral_cwd:
249
+ for skill, queries in targets.items():
250
+ for query in queries:
251
+ for _ in range(args.samples):
252
+ try:
253
+ stream = run_probe(query, args.model, args.timeout, neutral_cwd)
254
+ outcome, detail = classify(parse_stream_tool_uses(stream), skill)
255
+ except (RuntimeError, subprocess.TimeoutExpired) as e:
256
+ outcome, detail = "error", str(e)[:200]
257
+ records.append({"skill": skill, "query": query,
258
+ "outcome": outcome, "detail": detail})
259
+ print(f" [{outcome:9}] {skill} ← {query[:60]!r}"
260
+ + (f" → {detail}" if outcome == "confusion" else ""))
261
+
262
+ summary = aggregate(records)
263
+ report = {"model": args.model or "cli-default", "samples": args.samples,
264
+ "skills": summary, "records": records}
265
+ out = Path(args.out)
266
+ out.parent.mkdir(parents=True, exist_ok=True)
267
+ out.write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
268
+
269
+ print("\nskill rate probes confusions")
270
+ for name, s in sorted(summary.items()):
271
+ conf = ", ".join(f"{k}×{v}" for k, v in
272
+ sorted(s["confusions"].items(), key=lambda kv: -kv[1])) or "-"
273
+ rate = "n/a " if s["trigger_rate"] is None else f"{s['trigger_rate']:.2f}"
274
+ print(f"{name:30} {rate} {s['probes']:4} {conf}")
275
+ print(f"\nreport → {out}")
276
+ return 0
277
+
278
+
279
+ if __name__ == "__main__":
280
+ sys.exit(main())