dsh-aris-panel 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (442) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +98 -0
  3. package/README_CN.md +87 -0
  4. package/dsh/checkout.patch.yml +38 -0
  5. package/dsh/client.js +634 -0
  6. package/dsh/cordis.patch.yml +44 -0
  7. package/dsh/index.mjs +76 -0
  8. package/dsh/run-status.mjs +182 -0
  9. package/dsh/scope-limits.mjs +50 -0
  10. package/dsh/workbench.mjs +291 -0
  11. package/mcp-servers/claude-review/README.md +93 -0
  12. package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
  13. package/mcp-servers/claude-review/server.py +718 -0
  14. package/mcp-servers/codex-image2/README.md +65 -0
  15. package/mcp-servers/codex-image2/server.py +893 -0
  16. package/mcp-servers/feishu-bridge/requirements.txt +1 -0
  17. package/mcp-servers/feishu-bridge/server.py +240 -0
  18. package/mcp-servers/gemini-review/README.md +171 -0
  19. package/mcp-servers/gemini-review/server.py +1856 -0
  20. package/mcp-servers/llm-chat/requirements.txt +1 -0
  21. package/mcp-servers/llm-chat/server.py +664 -0
  22. package/mcp-servers/manual-review/README.md +133 -0
  23. package/mcp-servers/manual-review/server.py +910 -0
  24. package/mcp-servers/manual-review/ui.html +279 -0
  25. package/mcp-servers/minimax-chat/requirements.txt +1 -0
  26. package/mcp-servers/minimax-chat/server.py +381 -0
  27. package/package.json +51 -0
  28. package/skills/ablation-planner/SKILL.md +123 -0
  29. package/skills/alphaxiv/SKILL.md +196 -0
  30. package/skills/analyze-results/SKILL.md +46 -0
  31. package/skills/arxiv/SKILL.md +248 -0
  32. package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
  33. package/skills/auto-review-loop/SKILL.md +1137 -0
  34. package/skills/auto-review-loop-llm/SKILL.md +259 -0
  35. package/skills/auto-review-loop-minimax/SKILL.md +302 -0
  36. package/skills/citation-audit/SKILL.md +502 -0
  37. package/skills/claims-drafting/SKILL.md +227 -0
  38. package/skills/comm-lit-review/SKILL.md +297 -0
  39. package/skills/deepxiv/SKILL.md +263 -0
  40. package/skills/dse-loop/SKILL.md +296 -0
  41. package/skills/embodiment-description/SKILL.md +129 -0
  42. package/skills/exa-search/SKILL.md +205 -0
  43. package/skills/experiment-audit/SKILL.md +311 -0
  44. package/skills/experiment-bridge/SKILL.md +376 -0
  45. package/skills/experiment-plan/SKILL.md +249 -0
  46. package/skills/experiment-queue/SKILL.md +431 -0
  47. package/skills/experiment-queue/scripts/build_manifest.py +142 -0
  48. package/skills/experiment-queue/scripts/queue_manager.py +433 -0
  49. package/skills/feishu-notify/SKILL.md +156 -0
  50. package/skills/figure-description/SKILL.md +138 -0
  51. package/skills/figure-spec/SKILL.md +262 -0
  52. package/skills/figure-spec/scripts/figure_renderer.py +799 -0
  53. package/skills/formula-derivation/SKILL.md +280 -0
  54. package/skills/gemini-search/SKILL.md +231 -0
  55. package/skills/grant-proposal/SKILL.md +698 -0
  56. package/skills/idea-creator/SKILL.md +542 -0
  57. package/skills/idea-discovery/SKILL.md +521 -0
  58. package/skills/idea-discovery-robot/SKILL.md +363 -0
  59. package/skills/integrity-forensics/SKILL.md +284 -0
  60. package/skills/interview-cheatsheet/SKILL.md +245 -0
  61. package/skills/invention-structuring/SKILL.md +188 -0
  62. package/skills/jurisdiction-format/SKILL.md +192 -0
  63. package/skills/kill-argument/SKILL.md +437 -0
  64. package/skills/mermaid-diagram/SKILL.md +419 -0
  65. package/skills/meta-apply/SKILL.md +141 -0
  66. package/skills/meta-optimize/SKILL.md +437 -0
  67. package/skills/monitor-experiment/SKILL.md +140 -0
  68. package/skills/novelty-check/SKILL.md +101 -0
  69. package/skills/openalex/SKILL.md +237 -0
  70. package/skills/overleaf-sync/SKILL.md +220 -0
  71. package/skills/paper-claim-audit/SKILL.md +348 -0
  72. package/skills/paper-compile/SKILL.md +266 -0
  73. package/skills/paper-figure/SKILL.md +312 -0
  74. package/skills/paper-illustration/SKILL.md +736 -0
  75. package/skills/paper-illustration-image2/SKILL.md +391 -0
  76. package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  77. package/skills/paper-plan/SKILL.md +386 -0
  78. package/skills/paper-poster/SKILL.md +19 -0
  79. package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
  80. package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
  81. package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
  82. package/skills/paper-poster-html/NOTICE.md +57 -0
  83. package/skills/paper-poster-html/SKILL.md +323 -0
  84. package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
  85. package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
  86. package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
  87. package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
  88. package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
  89. package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
  90. package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
  91. package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
  92. package/skills/paper-poster-html/scripts/asset_check.py +897 -0
  93. package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
  94. package/skills/paper-poster-html/scripts/poster_check.py +251 -0
  95. package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
  96. package/skills/paper-poster-html/scripts/render_preview.py +217 -0
  97. package/skills/paper-poster-html/scripts/run_gates.py +556 -0
  98. package/skills/paper-poster-html/scripts/style_check.py +1324 -0
  99. package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
  100. package/skills/paper-poster-html/templates/README.md +170 -0
  101. package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
  102. package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
  103. package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
  104. package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
  105. package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
  106. package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
  107. package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
  108. package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
  109. package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
  110. package/skills/paper-slides/SKILL.md +635 -0
  111. package/skills/paper-talk/SKILL.md +381 -0
  112. package/skills/paper-write/SKILL.md +604 -0
  113. package/skills/paper-write/templates/IEEEtran.bst +2409 -0
  114. package/skills/paper-write/templates/IEEEtran.cls +6347 -0
  115. package/skills/paper-write/templates/iclr2026.tex +84 -0
  116. package/skills/paper-write/templates/icml2025.tex +87 -0
  117. package/skills/paper-write/templates/ieee_conference.tex +89 -0
  118. package/skills/paper-write/templates/ieee_journal.tex +93 -0
  119. package/skills/paper-write/templates/math_commands.tex +48 -0
  120. package/skills/paper-write/templates/neurips2025.tex +80 -0
  121. package/skills/paper-writing/SKILL.md +916 -0
  122. package/skills/patent-novelty-check/SKILL.md +153 -0
  123. package/skills/patent-pipeline/SKILL.md +344 -0
  124. package/skills/patent-review/SKILL.md +203 -0
  125. package/skills/pixel-art/SKILL.md +137 -0
  126. package/skills/prior-art-search/SKILL.md +146 -0
  127. package/skills/proof-checker/SKILL.md +866 -0
  128. package/skills/proof-orchestrator/NOTICE.md +24 -0
  129. package/skills/proof-orchestrator/SKILL.md +254 -0
  130. package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
  131. package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
  132. package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
  133. package/skills/proof-orchestrator/references/notation-audit.md +135 -0
  134. package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  135. package/skills/proof-orchestrator/references/stress-tests.md +38 -0
  136. package/skills/proof-writer/SKILL.md +223 -0
  137. package/skills/qzcli/SKILL.md +324 -0
  138. package/skills/rebuttal/SKILL.md +376 -0
  139. package/skills/render-html/SKILL.md +316 -0
  140. package/skills/render-html/scripts/render_html.py +1006 -0
  141. package/skills/render-html/scripts/templates/academic.html +703 -0
  142. package/skills/render-html/scripts/templates/dashboard.html +333 -0
  143. package/skills/research-lit/SKILL.md +756 -0
  144. package/skills/research-pipeline/SKILL.md +384 -0
  145. package/skills/research-refine/SKILL.md +770 -0
  146. package/skills/research-refine-pipeline/SKILL.md +186 -0
  147. package/skills/research-review/SKILL.md +198 -0
  148. package/skills/research-wiki/SKILL.md +461 -0
  149. package/skills/resubmit-pipeline/SKILL.md +447 -0
  150. package/skills/result-to-claim/SKILL.md +311 -0
  151. package/skills/run-experiment/SKILL.md +313 -0
  152. package/skills/semantic-scholar/SKILL.md +236 -0
  153. package/skills/serverless-modal/SKILL.md +335 -0
  154. package/skills/shared-references/acceptance-gate.md +324 -0
  155. package/skills/shared-references/assurance-contract.md +248 -0
  156. package/skills/shared-references/capture-antipatterns.md +78 -0
  157. package/skills/shared-references/citation-discipline.md +583 -0
  158. package/skills/shared-references/compute-env-contract.md +163 -0
  159. package/skills/shared-references/effort-contract.md +183 -0
  160. package/skills/shared-references/evidence-precheck.md +65 -0
  161. package/skills/shared-references/experiment-integrity.md +49 -0
  162. package/skills/shared-references/external-cadence.md +326 -0
  163. package/skills/shared-references/fan-out-pattern.md +366 -0
  164. package/skills/shared-references/injection-hygiene.md +127 -0
  165. package/skills/shared-references/integration-contract.md +461 -0
  166. package/skills/shared-references/output-composition.md +93 -0
  167. package/skills/shared-references/output-language.md +45 -0
  168. package/skills/shared-references/output-manifest.md +49 -0
  169. package/skills/shared-references/output-versioning.md +111 -0
  170. package/skills/shared-references/patent-format-cn.md +199 -0
  171. package/skills/shared-references/patent-format-ep.md +173 -0
  172. package/skills/shared-references/patent-format-us.md +161 -0
  173. package/skills/shared-references/patent-writing-principles.md +197 -0
  174. package/skills/shared-references/prior-art-databases.md +141 -0
  175. package/skills/shared-references/resumable-runs.md +109 -0
  176. package/skills/shared-references/review-scope-limits.md +81 -0
  177. package/skills/shared-references/review-tracing.md +391 -0
  178. package/skills/shared-references/reviewer-independence.md +79 -0
  179. package/skills/shared-references/reviewer-routing.md +852 -0
  180. package/skills/shared-references/skill-governance.md +104 -0
  181. package/skills/shared-references/taste-calibration.md +85 -0
  182. package/skills/shared-references/venue-checklists.md +114 -0
  183. package/skills/shared-references/wiki-helper-resolution.md +134 -0
  184. package/skills/shared-references/writing-principles.md +525 -0
  185. package/skills/skills-codex/README.md +102 -0
  186. package/skills/skills-codex/README_CN.md +100 -0
  187. package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
  188. package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
  189. package/skills/skills-codex/analyze-results/SKILL.md +45 -0
  190. package/skills/skills-codex/arxiv/SKILL.md +210 -0
  191. package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
  192. package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
  193. package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
  194. package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
  195. package/skills/skills-codex/citation-audit/SKILL.md +504 -0
  196. package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
  197. package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
  198. package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
  199. package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
  200. package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
  201. package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
  202. package/skills/skills-codex/deepxiv/SKILL.md +142 -0
  203. package/skills/skills-codex/dse-loop/SKILL.md +285 -0
  204. package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
  205. package/skills/skills-codex/exa-search/SKILL.md +192 -0
  206. package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
  207. package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
  208. package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
  209. package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
  210. package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
  211. package/skills/skills-codex/figure-description/SKILL.md +138 -0
  212. package/skills/skills-codex/figure-spec/SKILL.md +252 -0
  213. package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
  214. package/skills/skills-codex/gemini-search/SKILL.md +205 -0
  215. package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
  216. package/skills/skills-codex/idea-creator/SKILL.md +405 -0
  217. package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
  218. package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
  219. package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
  220. package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
  221. package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
  222. package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
  223. package/skills/skills-codex/kill-argument/SKILL.md +403 -0
  224. package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
  225. package/skills/skills-codex/meta-apply/SKILL.md +154 -0
  226. package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
  227. package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
  228. package/skills/skills-codex/novelty-check/SKILL.md +89 -0
  229. package/skills/skills-codex/openalex/SKILL.md +228 -0
  230. package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
  231. package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
  232. package/skills/skills-codex/paper-compile/SKILL.md +253 -0
  233. package/skills/skills-codex/paper-figure/SKILL.md +311 -0
  234. package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
  235. package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
  236. package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  237. package/skills/skills-codex/paper-plan/SKILL.md +278 -0
  238. package/skills/skills-codex/paper-poster/SKILL.md +19 -0
  239. package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
  240. package/skills/skills-codex/paper-slides/SKILL.md +571 -0
  241. package/skills/skills-codex/paper-talk/SKILL.md +381 -0
  242. package/skills/skills-codex/paper-write/SKILL.md +411 -0
  243. package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
  244. package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
  245. package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
  246. package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
  247. package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
  248. package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
  249. package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
  250. package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
  251. package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
  252. package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
  253. package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
  254. package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
  255. package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
  256. package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
  257. package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
  258. package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
  259. package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
  260. package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
  261. package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
  262. package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
  263. package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
  264. package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
  265. package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
  266. package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
  267. package/skills/skills-codex/paper-writing/SKILL.md +731 -0
  268. package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
  269. package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
  270. package/skills/skills-codex/patent-review/SKILL.md +202 -0
  271. package/skills/skills-codex/pixel-art/SKILL.md +139 -0
  272. package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
  273. package/skills/skills-codex/proof-checker/SKILL.md +554 -0
  274. package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
  275. package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
  276. package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
  277. package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
  278. package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
  279. package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  280. package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
  281. package/skills/skills-codex/proof-writer/SKILL.md +222 -0
  282. package/skills/skills-codex/qzcli/SKILL.md +324 -0
  283. package/skills/skills-codex/rebuttal/SKILL.md +305 -0
  284. package/skills/skills-codex/render-html/SKILL.md +305 -0
  285. package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
  286. package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
  287. package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
  288. package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
  289. package/skills/skills-codex/research-lit/SKILL.md +464 -0
  290. package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
  291. package/skills/skills-codex/research-refine/SKILL.md +721 -0
  292. package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
  293. package/skills/skills-codex/research-review/SKILL.md +135 -0
  294. package/skills/skills-codex/research-wiki/SKILL.md +421 -0
  295. package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
  296. package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
  297. package/skills/skills-codex/run-experiment/SKILL.md +236 -0
  298. package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
  299. package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
  300. package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
  301. package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
  302. package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
  303. package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
  304. package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
  305. package/skills/skills-codex/shared-references/effort-contract.md +143 -0
  306. package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
  307. package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
  308. package/skills/skills-codex/shared-references/external-cadence.md +334 -0
  309. package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
  310. package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
  311. package/skills/skills-codex/shared-references/integration-contract.md +372 -0
  312. package/skills/skills-codex/shared-references/output-composition.md +98 -0
  313. package/skills/skills-codex/shared-references/output-language.md +45 -0
  314. package/skills/skills-codex/shared-references/output-manifest.md +40 -0
  315. package/skills/skills-codex/shared-references/output-versioning.md +111 -0
  316. package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
  317. package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
  318. package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
  319. package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
  320. package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
  321. package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
  322. package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
  323. package/skills/skills-codex/shared-references/review-tracing.md +144 -0
  324. package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
  325. package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
  326. package/skills/skills-codex/shared-references/skill-governance.md +119 -0
  327. package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
  328. package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
  329. package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
  330. package/skills/skills-codex/shared-references/writing-principles.md +525 -0
  331. package/skills/skills-codex/slides-polish/SKILL.md +563 -0
  332. package/skills/skills-codex/specification-writing/SKILL.md +211 -0
  333. package/skills/skills-codex/system-profile/SKILL.md +103 -0
  334. package/skills/skills-codex/training-check/SKILL.md +83 -0
  335. package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
  336. package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
  337. package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
  338. package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
  339. package/skills/skills-codex-claude-review/README.md +79 -0
  340. package/skills/skills-codex-claude-review/README_CN.md +78 -0
  341. package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
  342. package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
  343. package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
  344. package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
  345. package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
  346. package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
  347. package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
  348. package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
  349. package/skills/skills-codex-gemini-review/README.md +176 -0
  350. package/skills/skills-codex-gemini-review/README_CN.md +175 -0
  351. package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
  352. package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
  353. package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
  354. package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
  355. package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
  356. package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
  357. package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
  358. package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
  359. package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
  360. package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
  361. package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
  362. package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
  363. package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
  364. package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
  365. package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
  366. package/skills/slides-polish/SKILL.md +565 -0
  367. package/skills/specification-writing/SKILL.md +211 -0
  368. package/skills/system-profile/SKILL.md +103 -0
  369. package/skills/training-check/SKILL.md +132 -0
  370. package/skills/vast-gpu/SKILL.md +394 -0
  371. package/skills/web-debug-search/SKILL.md +334 -0
  372. package/skills/wiki-enrich/SKILL.md +257 -0
  373. package/skills/writing-systems-papers/SKILL.md +184 -0
  374. package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
  375. package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
  376. package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
  377. package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
  378. package/templates/FINDINGS_TEMPLATE.md +52 -0
  379. package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
  380. package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
  381. package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
  382. package/templates/MANIFEST_TEMPLATE.md +7 -0
  383. package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
  384. package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
  385. package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
  386. package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
  387. package/templates/README.md +57 -0
  388. package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
  389. package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
  390. package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
  391. package/templates/claude-hooks/corpus_write_guard.json +16 -0
  392. package/templates/claude-hooks/corpus_write_guard.py +85 -0
  393. package/templates/claude-hooks/meta_logging.json +74 -0
  394. package/templates/gitignore-trace.txt +3 -0
  395. package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
  396. package/tools/arxiv_fetch.py +311 -0
  397. package/tools/capture_filter.py +126 -0
  398. package/tools/check_skills_inventory.py +273 -0
  399. package/tools/convert_skills_to_llm_chat.py +282 -0
  400. package/tools/copilot_native_evidence.py +818 -0
  401. package/tools/deepxiv_fetch.py +213 -0
  402. package/tools/evidence_check.py +212 -0
  403. package/tools/exa_search.py +425 -0
  404. package/tools/experiment_queue/README.md +118 -0
  405. package/tools/experiment_queue/build_manifest.py +44 -0
  406. package/tools/experiment_queue/queue_manager.py +44 -0
  407. package/tools/extract_paper_style.py +560 -0
  408. package/tools/figure_renderer.py +69 -0
  409. package/tools/forensics_gate.py +669 -0
  410. package/tools/generate_codex_claude_review_overrides.py +299 -0
  411. package/tools/idea_discovery_gate.py +256 -0
  412. package/tools/install_aris.ps1 +1372 -0
  413. package/tools/install_aris.sh +1370 -0
  414. package/tools/install_aris_codex.sh +1023 -0
  415. package/tools/install_aris_copilot.sh +1052 -0
  416. package/tools/iteration_log.py +143 -0
  417. package/tools/lint_skills_helpers.sh +84 -0
  418. package/tools/meta_opt/check_ready.sh +80 -0
  419. package/tools/meta_opt/log_event.sh +91 -0
  420. package/tools/meta_opt/trigger_eval.py +280 -0
  421. package/tools/meta_opt/trigger_evals.sample.json +28 -0
  422. package/tools/openalex_fetch.py +326 -0
  423. package/tools/overleaf_audit.sh +104 -0
  424. package/tools/overleaf_setup.sh +150 -0
  425. package/tools/paper_illustration_image2.py +62 -0
  426. package/tools/provenance.py +294 -0
  427. package/tools/research_wiki.py +1720 -0
  428. package/tools/review_gate.py +502 -0
  429. package/tools/run_state.py +399 -0
  430. package/tools/save_trace.sh +477 -0
  431. package/tools/semantic_scholar_fetch.py +438 -0
  432. package/tools/skill-groups.tsv +116 -0
  433. package/tools/skill_picker.py +238 -0
  434. package/tools/smart_update.ps1 +521 -0
  435. package/tools/smart_update.sh +591 -0
  436. package/tools/smart_update_codex.sh +419 -0
  437. package/tools/smart_update_copilot.sh +605 -0
  438. package/tools/threat_scan.py +222 -0
  439. package/tools/verify_paper_audits.sh +487 -0
  440. package/tools/verify_papers.py +613 -0
  441. package/tools/verify_wiki_coverage.sh +176 -0
  442. package/tools/watchdog.py +485 -0
@@ -0,0 +1,246 @@
1
+ ---
2
+ name: result-to-claim
3
+ description: Use when experiments complete to judge what claims the results support, what they don't, and what evidence is still missing. A secondary Codex agent evaluates results against intended claims and routes to next action (pivot, supplement, or confirm). Use after experiments finish — before writing the paper or running ablations.
4
+ argument-hint: "[experiment-description-or-wandb-run]"
5
+ allowed-tools: Bash(*), Read, Grep, Glob, Write, Edit
6
+ ---
7
+
8
+ # Result-to-Claim Gate
9
+
10
+ > **Codex assurance:** deterministic evidence existence can be accepted, while
11
+ > the base semantic claim judgment records `review_independence: same-family`
12
+ > and `acceptance_status: provisional`. Cross-family overlays may record
13
+ > accepted; reviewer failure emits BLOCKED.
14
+
15
+ Experiments produce numbers; this gate decides what those numbers *mean*. Collect results from available sources, get a secondary Codex judgment, then auto-route based on the verdict.
16
+
17
+ ## Context: $ARGUMENTS
18
+
19
+ ## When to Use
20
+
21
+ - After a set of experiments completes (main results, not just sanity checks)
22
+ - Before committing to claims in a paper or review response
23
+ - When results are ambiguous and you need an objective second opinion
24
+
25
+ ## Workflow
26
+
27
+ ### Step 1: Collect Results
28
+
29
+ Gather experiment data from whatever sources are available in the project:
30
+
31
+ 1. **W&B** (preferred): `wandb.Api().run("<entity>/<project>/<run_id>").history()` — metrics, training curves, comparisons
32
+ 2. **EXPERIMENT_LOG.md**: full results table with baselines and verdicts
33
+ 3. **EXPERIMENT_TRACKER.md**: check which experiments are DONE vs still running
34
+ 4. **Log files**: `ssh server "tail -100 /path/to/training.log"` if no other source
35
+ 5. **`idea-stage/docs/research_contract.md`** (legacy fallback: `docs/research_contract.md`): intended claims and experiment design
36
+
37
+ Assemble the key information:
38
+ - What experiments were run (method, dataset, config)
39
+ - Main metrics and baseline comparisons (deltas)
40
+ - The intended claim these experiments were designed to test
41
+ - Any known confounds or caveats
42
+
43
+ ### Step 1.5: Deterministic evidence pre-check
44
+
45
+ Before the reviewer call, resolve and run `evidence_check.py` per
46
+ [`evidence-precheck.md`](../shared-references/evidence-precheck.md):
47
+
48
+ ```bash
49
+ if [ -z "${ARIS_REPO:-}" ] && [ -f .aris/installed-skills-codex.txt ]; then
50
+ ARIS_REPO=$(awk -F'\t' '$1=="repo_root"{print $2; exit}' .aris/installed-skills-codex.txt 2>/dev/null) || true
51
+ fi
52
+ EVIDENCE_CHECK=""
53
+ [ -n "${ARIS_REPO:-}" ] && [ -f "$ARIS_REPO/tools/evidence_check.py" ] && EVIDENCE_CHECK="$ARIS_REPO/tools/evidence_check.py"
54
+ [ -z "$EVIDENCE_CHECK" ] && [ -f tools/evidence_check.py ] && EVIDENCE_CHECK="tools/evidence_check.py"
55
+ mkdir -p .aris
56
+ if [ -n "$EVIDENCE_CHECK" ]; then
57
+ python3 "$EVIDENCE_CHECK" . --batch .aris/claims.json \
58
+ > .aris/evidence_precheck.json 2>.aris/evidence_precheck.err || true
59
+ else
60
+ echo "WARN: evidence_check.py unresolved; semantic review will still run" >&2
61
+ fi
62
+ ```
63
+
64
+ Treat `path_missing` and `value_not_found` as unsupported evidence before the
65
+ semantic review. `verified` means only that the cited value exists; it does not
66
+ prove the claim. Pass the pre-check JSON path to the fresh reviewer. The Codex
67
+ reviewer's positive result remains `review_independence: same-family` and
68
+ `acceptance_status: provisional`; a deterministic evidence check never upgrades
69
+ a semantic claim to accepted by itself.
70
+
71
+ ### Step 2: Codex Judgment
72
+
73
+ Send the collected results to a secondary Codex agent for objective evaluation:
74
+
75
+ ```text
76
+ spawn_agent:
77
+ model: gpt-5.6-sol
78
+ reasoning_effort: ultra
79
+ message: |
80
+ RESULT-TO-CLAIM EVALUATION
81
+
82
+ I need you to judge whether experimental results support the intended claim.
83
+
84
+ Intended claim: [the claim these experiments test]
85
+
86
+ Experiments run:
87
+ [list experiments with method, dataset, metrics]
88
+
89
+ Results:
90
+ [paste key numbers, comparison deltas, significance]
91
+
92
+ Baselines:
93
+ [baseline numbers and sources — reproduced or from paper]
94
+
95
+ Known caveats:
96
+ [any confounding factors, limited datasets, missing comparisons]
97
+
98
+ Please evaluate:
99
+ 1. claim_supported: yes | partial | no
100
+ 2. what_results_support: what the data actually shows
101
+ 3. what_results_dont_support: where the data falls short of the claim
102
+ 4. missing_evidence: specific evidence gaps
103
+ 5. suggested_claim_revision: if the claim should be strengthened, weakened, or reframed
104
+ 6. next_experiments_needed: specific experiments to fill gaps (if any)
105
+ 7. confidence: high | medium | low
106
+
107
+ Be honest. Do not inflate claims beyond what the data supports.
108
+ A single positive result on one dataset does not support a general claim.
109
+ ```
110
+
111
+ ### Step 3: Parse and Normalize
112
+
113
+ Extract structured fields from the secondary Codex response:
114
+
115
+ ```markdown
116
+ - claim_supported: yes | partial | no
117
+ - what_results_support: "..."
118
+ - what_results_dont_support: "..."
119
+ - missing_evidence: "..."
120
+ - suggested_claim_revision: "..."
121
+ - next_experiments_needed: "..."
122
+ - confidence: high | medium | low
123
+ ```
124
+
125
+ ### Step 3.5: Check Experiment Integrity (if audit exists)
126
+
127
+ **Skip this step if `EXPERIMENT_AUDIT.json` does not exist.**
128
+
129
+ ```
130
+ if EXPERIMENT_AUDIT.json exists:
131
+ read integrity_status from file
132
+ attach to verdict output:
133
+ integrity_status: pass | warn | fail
134
+
135
+ if integrity_status == "fail":
136
+ append to verdict: "[INTEGRITY CONCERN] — audit found issues, see EXPERIMENT_AUDIT.md"
137
+ downgrade confidence to "low" regardless of Codex judgment
138
+
139
+ if integrity_status == "warn":
140
+ append to verdict: "[INTEGRITY: WARN] — audit flagged potential issues"
141
+ else:
142
+ integrity_status = "unavailable"
143
+ verdict is labeled "provisional — no integrity audit run"
144
+ (this does NOT block anything — pipeline continues normally)
145
+ ```
146
+
147
+ See `shared-references/experiment-integrity.md` for the full integrity protocol.
148
+
149
+ ### Step 4: Route Based on Verdict
150
+
151
+ #### `no` — Claim not supported
152
+
153
+ 1. Record postmortem in findings.md (Research Findings section):
154
+ - What was tested, what failed, hypotheses for why
155
+ - Constraints for future attempts (what NOT to try again)
156
+ 2. Update the project pipeline status in `AGENTS.md` or project notes
157
+ 3. Decide whether to pivot to next idea from IDEA_CANDIDATES.md or try an alternative approach
158
+
159
+ #### `partial` — Claim partially supported
160
+
161
+ 1. Update the working claim to reflect what IS supported
162
+ 2. Record the gap in findings.md
163
+ 3. Design and run supplementary experiments to fill evidence gaps
164
+ 4. Re-run result-to-claim after supplementary experiments complete
165
+ 5. **Multiple rounds of `partial` on the same claim** → record analysis in findings.md, consider whether to narrow the claim scope or switch ideas
166
+
167
+ #### `yes` — Claim supported
168
+
169
+ 1. Record confirmed claim in project notes
170
+ 2. If ablation studies are incomplete → trigger `/ablation-planner`
171
+ 3. If all evidence is in → ready for paper writing
172
+
173
+ ### Step 5: Update Research Wiki (if active)
174
+
175
+ **Skip this step entirely if `research-wiki/` does not exist.**
176
+
177
+ ```
178
+ if research-wiki/ exists:
179
+ # Resolve the helper (Codex chain). If unavailable, skip wiki writes; still report verdict.
180
+ ARIS_REPO="${ARIS_REPO:-$(awk -F'\t' '$1=="repo_root"{print $2; exit}' .aris/installed-skills-codex.txt 2>/dev/null)}"
181
+ WIKI_SCRIPT=""
182
+ [ -n "$ARIS_REPO" ] && [ -f "$ARIS_REPO/tools/research_wiki.py" ] && WIKI_SCRIPT="$ARIS_REPO/tools/research_wiki.py"
183
+ [ -z "$WIKI_SCRIPT" ] && [ -f tools/research_wiki.py ] && WIKI_SCRIPT="tools/research_wiki.py"
184
+ [ -z "$WIKI_SCRIPT" ] && [ -f ~/.codex/skills/research-wiki/research_wiki.py ] && WIKI_SCRIPT="$HOME/.codex/skills/research-wiki/research_wiki.py"
185
+ [ -n "$WIKI_SCRIPT" ] || echo "WARN: research_wiki.py unreachable; skipping wiki writes (verdict still reported)." >&2
186
+
187
+ # 1. Create/refresh the experiment node FIRST (verdict OWNER → --update-on-exist so a
188
+ # re-judge overwrites the stale verdict). The supports/invalidates edges in #2 point
189
+ # FROM exp:<id> and add_edge does NOT verify node existence, so only add them if the
190
+ # experiment node was born (EXP_NODE_OK); otherwise skip the wiki edges.
191
+ EXP_NODE_OK=0
192
+ [ -n "$WIKI_SCRIPT" ] && python3 "$WIKI_SCRIPT" add_experiment research-wiki/ \
193
+ --slug "<exp_id>" --idea "idea:<active_idea>" \
194
+ --verdict "<yes|partial|no>" --confidence "<high|medium|low>" \
195
+ --date "<date>" --hardware "<hw>" --duration "<dur>" \
196
+ --metrics "<key metrics>" --reasoning "<one-line why this verdict>" \
197
+ --provenance "<EXPERIMENT_AUDIT.md / run dir>" --update-on-exist && EXP_NODE_OK=1
198
+
199
+ # 2. Record empirical support as EDGES ONLY, and ONLY if EXP_NODE_OK. NEVER edit a
200
+ # claim page's `status`: that is the PROOF axis (verified / refuted / unproven /
201
+ # sound-modulo-imports / drafted / retracted), owned by /proof-checker (the claim
202
+ # birth point) — the ARIS helper REJECTS "supported"/"partial"/"invalidated".
203
+ if [ "$EXP_NODE_OK" = 1 ]:
204
+ for each claim resolved by this verdict:
205
+ if verdict == "yes":
206
+ python3 "$WIKI_SCRIPT" add_edge research-wiki/ --from "exp:<id>" --to "claim:<cid>" --type supports --evidence "<metric>"
207
+ elif verdict == "partial":
208
+ python3 "$WIKI_SCRIPT" add_edge research-wiki/ --from "exp:<id>" --to "claim:<cid>" --type supports --evidence "partial: <metric>"
209
+ else:
210
+ python3 "$WIKI_SCRIPT" add_edge research-wiki/ --from "exp:<id>" --to "claim:<cid>" --type invalidates --evidence "<why>"
211
+
212
+ # 3. Update idea outcome (raw markdown, helper-free — preserves the rich idea body)
213
+ Update research-wiki/ideas/<idea_id>.md:
214
+ - outcome: positive | mixed | negative
215
+ - If negative: fill "Failure / Risk Notes" and "Lessons Learned"
216
+ - If positive: fill "Actual Outcome" and "Reusable Components"
217
+
218
+ # 4. Rebuild + log (reflect the new edges; only if WIKI_SCRIPT resolved)
219
+ [ -n "$WIKI_SCRIPT" ] && python3 "$WIKI_SCRIPT" rebuild_query_pack research-wiki/
220
+ [ -n "$WIKI_SCRIPT" ] && python3 "$WIKI_SCRIPT" log research-wiki/ "result-to-claim: exp:<id> verdict=<verdict> for idea:<idea_id>"
221
+
222
+ # 5. Re-ideation suggestion
223
+ Count failed/partial ideas since last /idea-creator run.
224
+ If >= 3: print "💡 3+ ideas tested since last ideation. Consider re-running /idea-creator — the wiki now knows what doesn't work."
225
+ ```
226
+
227
+ ## Rules
228
+
229
+ - **The secondary Codex agent is the judge, not the local executor.** The local executor collects evidence and routes; the reviewer agent evaluates. This prevents post-hoc rationalization.
230
+ - Do not inflate claims beyond what the data supports. If Codex says "partial", do not round up to "yes".
231
+ - A single positive result on one dataset does not support a general claim. Be honest about scope.
232
+ - If `confidence` is low, treat the judgment as inconclusive and add experiments rather than committing to a claim.
233
+ - **Fail closed if the reviewer is unavailable.** Follow the capability fallback
234
+ in `reviewer-routing.md` (`gpt-5.6-sol` + `ultra` → `gpt-5.6-sol` + `xhigh`
235
+ → `gpt-5.5` + `xhigh`), and never downgrade on timeout, rate-limit, auth,
236
+ transport, server, or context errors. If no allowed pair succeeds, write a
237
+ traced `BLOCKED` review record with the unavailable route and evidence paths, write
238
+ `CLAIMS_FROM_RESULTS.md` containing only `verdict: REVIEW_UNAVAILABLE`, record
239
+ the same in findings.md, and stop. Do not emit a local PASS/WARN substitute or
240
+ advance a submission-facing claim; only an explicitly non-submission
241
+ evidence-gathering phase may continue.
242
+ - Always record the verdict and reasoning in findings.md, regardless of outcome.
243
+
244
+ ## Review Tracing
245
+
246
+ After the secondary Codex judgment, save a trace following `../shared-references/review-tracing.md`. Write files directly to `.aris/traces/result-to-claim/<date>_run<NN>/` and include the prompt, raw reviewer response, parsed verdict, routing action, and whether the result is `[pending external review]`. Respect the `--- trace:` parameter when present (default: `full`).
@@ -0,0 +1,236 @@
1
+ ---
2
+ name: "run-experiment"
3
+ description: "Deploy and run ML experiments on local or remote GPU servers. Use when user says \"run experiment\", \"deploy to server\", \"\u8dd1\u5b9e\u9a8c\", or needs to launch training jobs."
4
+ ---
5
+
6
+ # Run Experiment
7
+
8
+ Deploy and run ML experiment: $ARGUMENTS
9
+
10
+ ## Workflow
11
+
12
+ ### Step 1: Detect Environment
13
+
14
+ Read the project's `AGENTS.md` to determine the experiment environment:
15
+
16
+ - **Local GPU**: Look for local CUDA/MPS setup info
17
+ - **Remote server**: Look for SSH alias, conda env, code directory
18
+ - **Vast.ai instance**: Look for `gpu: vast`, `vast_instance`, SSH host/port, remote path, and optional `auto_destroy`
19
+ - **Modal serverless**: Look for `gpu: modal`, app/function name, image/dependency setup, and secrets
20
+
21
+ If no server info is found in `AGENTS.md`, ask the user.
22
+
23
+ **Environment contract** (`../shared-references/compute-env-contract.md`): before
24
+ building or trusting any environment, read the provider's env ledger
25
+ (`.aris/compute/<provider>.md`) — an unchanged spec hash means warm-reuse, a
26
+ changed one means rebuild. New env → write the declarative spec first, render it
27
+ for this provider's shape, and never declare it ready on import-success alone:
28
+ run the seeded kernel witness, and after any rebuild/doc edit run the
29
+ agent-follows-doc pass (a fresh subagent executes the documented invocation
30
+ verbatim and reports doc-vs-reality divergence).
31
+
32
+ ### Step 2: Pre-flight Check
33
+
34
+ Check GPU availability on the target machine:
35
+
36
+ **Remote:**
37
+ ```bash
38
+ ssh <server> nvidia-smi --query-gpu=index,memory.used,memory.total --format=csv,noheader
39
+ ```
40
+
41
+ **Local:**
42
+ ```bash
43
+ nvidia-smi --query-gpu=index,memory.used,memory.total --format=csv,noheader
44
+ # or for Mac MPS:
45
+ python -c "import torch; print('MPS available:', torch.backends.mps.is_available())"
46
+ ```
47
+
48
+ Free GPU = memory.used < 500 MiB.
49
+
50
+ ### Step 3: Sync Code (Remote Only)
51
+
52
+ Check the project's `AGENTS.md` for a `code_sync` setting. If not specified, default to `rsync`.
53
+
54
+ #### Option A: rsync (default)
55
+
56
+ Only sync necessary files — NOT data, checkpoints, or large files:
57
+ ```bash
58
+ rsync -avz --include='*.py' --exclude='*' <local_src>/ <server>:<remote_dst>/
59
+ ```
60
+
61
+ #### Option B: git (when `code_sync: git` is set in AGENTS.md)
62
+
63
+ Push local changes to remote repo, then pull on the server:
64
+ ```bash
65
+ # 1. Push from local
66
+ git add -A && git commit -m "sync: experiment deployment" && git push
67
+
68
+ # 2. Pull on server
69
+ ssh <server> "cd <remote_dst> && git pull"
70
+ ```
71
+
72
+ Benefits: version-tracked, multi-server sync with one push, no rsync include/exclude rules needed.
73
+
74
+ #### Option C: Vast.ai instance
75
+
76
+ If `gpu: vast` is configured, treat the Vast.ai machine as a remote server with an explicit lifecycle:
77
+
78
+ 1. Verify the instance is running and reachable.
79
+ 2. Sync code to the configured remote path.
80
+ 3. Confirm data/checkpoints are already mounted or intentionally copied.
81
+ 4. Record the instance id in the launch summary for later cleanup.
82
+
83
+ Do not silently ignore a requested Vast.ai route. If Vast.ai CLI credentials or instance metadata are missing, stop and ask the user to configure them.
84
+
85
+ ### Step 3.5: W&B Integration (when `wandb: true` in AGENTS.md)
86
+
87
+ **Skip this step entirely if `wandb` is not set or is `false` in AGENTS.md.**
88
+
89
+ Before deploying, ensure the experiment scripts have W&B logging:
90
+
91
+ 1. **Check if wandb is already in the script** — look for `import wandb` or `wandb.init`. If present, skip to Step 4.
92
+
93
+ 2. **If not present, add W&B logging** to the training script:
94
+ ```python
95
+ import wandb
96
+ wandb.init(project=WANDB_PROJECT, name=EXP_NAME, config={...hyperparams...})
97
+
98
+ # Inside training loop:
99
+ wandb.log({"train/loss": loss, "train/lr": lr, "step": step})
100
+
101
+ # After eval:
102
+ wandb.log({"eval/loss": eval_loss, "eval/ppl": ppl, "eval/accuracy": acc})
103
+
104
+ # At end:
105
+ wandb.finish()
106
+ ```
107
+
108
+ 3. **Metrics to log** (add whichever apply to the experiment):
109
+ - `train/loss` — training loss per step
110
+ - `train/lr` — learning rate
111
+ - `eval/loss`, `eval/ppl`, `eval/accuracy` — eval metrics per epoch
112
+ - `gpu/memory_used` — GPU memory (via `torch.cuda.max_memory_allocated()`)
113
+ - `speed/samples_per_sec` — throughput
114
+ - Any custom metrics the experiment already computes
115
+
116
+ 4. **Verify wandb login on the target machine:**
117
+ ```bash
118
+ ssh <server> "wandb status" # should show logged in
119
+ # If not logged in:
120
+ ssh <server> "wandb login <WANDB_API_KEY>"
121
+ ```
122
+
123
+ > The W&B project name and API key come from `AGENTS.md` (see example below). The experiment name is auto-generated from the script name + timestamp.
124
+
125
+ ### Step 4: Deploy
126
+
127
+ #### Remote (via SSH + screen)
128
+
129
+ For each experiment, create a dedicated screen session with GPU binding:
130
+ ```bash
131
+ ssh <server> "screen -dmS <exp_name> bash -c '\
132
+ eval \"\$(<conda_path>/conda shell.bash hook)\" && \
133
+ conda activate <env> && \
134
+ CUDA_VISIBLE_DEVICES=<gpu_id> python <script> <args> 2>&1 | tee <log_file>'"
135
+ ```
136
+
137
+ #### Vast.ai instance
138
+
139
+ Use the same SSH + screen pattern, but include the Vast.ai instance id, public SSH endpoint, and remote working directory in the report. If `auto_destroy: true`, write a cleanup command to the run notes before launch.
140
+
141
+ Record the estimated hourly cost, expected run duration, and cleanup owner. If the command fails to start or the instance becomes unreachable, do not relaunch blindly; capture logs and ask for a rescue / second opinion before spending more GPU time.
142
+
143
+ #### Modal (serverless)
144
+
145
+ If `gpu: modal` is configured, deploy through Modal instead of SSH:
146
+
147
+ ```bash
148
+ modal run <module_or_app>.py -- <args>
149
+ ```
150
+
151
+ Before launch, verify required secrets, volumes, image dependencies, and output persistence. If Modal is requested but the project lacks Modal configuration, stop and ask the user to configure it rather than falling back to local execution.
152
+
153
+ Record the Modal app/function name, GPU type, timeout, mounted volumes, and where results will be stored. If Modal reports an image, secret, or volume error, preserve the exact error and run a configuration fix before retrying.
154
+
155
+ #### Local
156
+
157
+ ```bash
158
+ # Linux with CUDA
159
+ CUDA_VISIBLE_DEVICES=<gpu_id> python <script> <args> 2>&1 | tee <log_file>
160
+
161
+ # Mac with MPS (PyTorch uses MPS automatically)
162
+ python <script> <args> 2>&1 | tee <log_file>
163
+ ```
164
+
165
+ For local long-running jobs, use `run_in_background: true` to keep the conversation responsive.
166
+
167
+ ### Step 5: Verify Launch
168
+
169
+ **Remote:**
170
+ ```bash
171
+ ssh <server> "screen -ls"
172
+ ```
173
+
174
+ **Local:**
175
+ Check process is running and GPU is allocated.
176
+
177
+ ### Step 6: Feishu Notification (if configured)
178
+
179
+ After deployment is verified, check `~/.codex/feishu.json`:
180
+ - Send `experiment_done` notification: which experiments launched, which GPUs, estimated time
181
+ - If config absent or mode `"off"`: skip entirely (no-op)
182
+
183
+ ### Step 7: Auto-Destroy Vast.ai Instance (when `gpu: vast` and `auto_destroy: true`)
184
+
185
+ Only run this after the experiment has completed and results/logs/checkpoints have been copied or otherwise persisted.
186
+
187
+ 1. Verify the target process has exited.
188
+ 2. Copy result files and logs to the configured durable location.
189
+ 3. Ask for confirmation unless AGENTS.md explicitly says `auto_destroy: true`.
190
+ 4. Destroy only the recorded instance id for this run.
191
+
192
+ If any artifact copy fails, do not destroy the instance.
193
+
194
+ ## Key Rules
195
+
196
+ - ALWAYS check GPU availability first — never blindly assign GPUs
197
+ - Each experiment gets its own screen session + GPU (remote) or background process (local)
198
+ - Use `tee` to save logs for later inspection
199
+ - Run deployment commands with `run_in_background: true` to keep conversation responsive
200
+ - Report back: which GPU, which screen/process, what command, estimated time
201
+ - If multiple experiments, launch them in parallel on different GPUs
202
+
203
+ ## AGENTS.md Example
204
+
205
+ Users should add their server info to their project's `AGENTS.md`:
206
+
207
+ ```markdown
208
+ ## Remote Server
209
+ - SSH: `ssh my-gpu-server`
210
+ - GPU: 4x A100 (80GB each)
211
+ - Conda: `eval "$(/opt/conda/bin/conda shell.bash hook)" && conda activate research`
212
+ - Code dir: `/home/user/experiments/`
213
+ - code_sync: rsync # default. Or set to "git" for git push/pull workflow
214
+ - wandb: false # set to "true" to auto-add W&B logging to experiment scripts
215
+ - wandb_project: my-project # W&B project name (required if wandb: true)
216
+ - wandb_entity: my-team # W&B team/user (optional, uses default if omitted)
217
+
218
+ ## Vast.ai
219
+ - gpu: vast
220
+ - vast_instance: 123456
221
+ - SSH: `ssh -p 12345 root@ssh.vast.ai`
222
+ - Code dir: `/workspace/experiments/`
223
+ - auto_destroy: false
224
+
225
+ ## Modal
226
+ - gpu: modal
227
+ - modal_app: `train.py`
228
+ - modal_secrets: `wandb-secret`
229
+ - modal_volume: `experiment-results`
230
+
231
+ ## Local Environment
232
+ - Mac MPS / Linux CUDA
233
+ - Conda env: `ml` (Python 3.10 + PyTorch)
234
+ ```
235
+
236
+ > **W&B setup**: Run `wandb login` on your server once (or set `WANDB_API_KEY` env var). The skill reads project/entity from `AGENTS.md` and adds `wandb.init()` + `wandb.log()` to your training scripts automatically. Dashboard: `https://wandb.ai/<entity>/<project>`.