dsh-aris-panel 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (442) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +98 -0
  3. package/README_CN.md +87 -0
  4. package/dsh/checkout.patch.yml +38 -0
  5. package/dsh/client.js +634 -0
  6. package/dsh/cordis.patch.yml +44 -0
  7. package/dsh/index.mjs +76 -0
  8. package/dsh/run-status.mjs +182 -0
  9. package/dsh/scope-limits.mjs +50 -0
  10. package/dsh/workbench.mjs +291 -0
  11. package/mcp-servers/claude-review/README.md +93 -0
  12. package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
  13. package/mcp-servers/claude-review/server.py +718 -0
  14. package/mcp-servers/codex-image2/README.md +65 -0
  15. package/mcp-servers/codex-image2/server.py +893 -0
  16. package/mcp-servers/feishu-bridge/requirements.txt +1 -0
  17. package/mcp-servers/feishu-bridge/server.py +240 -0
  18. package/mcp-servers/gemini-review/README.md +171 -0
  19. package/mcp-servers/gemini-review/server.py +1856 -0
  20. package/mcp-servers/llm-chat/requirements.txt +1 -0
  21. package/mcp-servers/llm-chat/server.py +664 -0
  22. package/mcp-servers/manual-review/README.md +133 -0
  23. package/mcp-servers/manual-review/server.py +910 -0
  24. package/mcp-servers/manual-review/ui.html +279 -0
  25. package/mcp-servers/minimax-chat/requirements.txt +1 -0
  26. package/mcp-servers/minimax-chat/server.py +381 -0
  27. package/package.json +51 -0
  28. package/skills/ablation-planner/SKILL.md +123 -0
  29. package/skills/alphaxiv/SKILL.md +196 -0
  30. package/skills/analyze-results/SKILL.md +46 -0
  31. package/skills/arxiv/SKILL.md +248 -0
  32. package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
  33. package/skills/auto-review-loop/SKILL.md +1137 -0
  34. package/skills/auto-review-loop-llm/SKILL.md +259 -0
  35. package/skills/auto-review-loop-minimax/SKILL.md +302 -0
  36. package/skills/citation-audit/SKILL.md +502 -0
  37. package/skills/claims-drafting/SKILL.md +227 -0
  38. package/skills/comm-lit-review/SKILL.md +297 -0
  39. package/skills/deepxiv/SKILL.md +263 -0
  40. package/skills/dse-loop/SKILL.md +296 -0
  41. package/skills/embodiment-description/SKILL.md +129 -0
  42. package/skills/exa-search/SKILL.md +205 -0
  43. package/skills/experiment-audit/SKILL.md +311 -0
  44. package/skills/experiment-bridge/SKILL.md +376 -0
  45. package/skills/experiment-plan/SKILL.md +249 -0
  46. package/skills/experiment-queue/SKILL.md +431 -0
  47. package/skills/experiment-queue/scripts/build_manifest.py +142 -0
  48. package/skills/experiment-queue/scripts/queue_manager.py +433 -0
  49. package/skills/feishu-notify/SKILL.md +156 -0
  50. package/skills/figure-description/SKILL.md +138 -0
  51. package/skills/figure-spec/SKILL.md +262 -0
  52. package/skills/figure-spec/scripts/figure_renderer.py +799 -0
  53. package/skills/formula-derivation/SKILL.md +280 -0
  54. package/skills/gemini-search/SKILL.md +231 -0
  55. package/skills/grant-proposal/SKILL.md +698 -0
  56. package/skills/idea-creator/SKILL.md +542 -0
  57. package/skills/idea-discovery/SKILL.md +521 -0
  58. package/skills/idea-discovery-robot/SKILL.md +363 -0
  59. package/skills/integrity-forensics/SKILL.md +284 -0
  60. package/skills/interview-cheatsheet/SKILL.md +245 -0
  61. package/skills/invention-structuring/SKILL.md +188 -0
  62. package/skills/jurisdiction-format/SKILL.md +192 -0
  63. package/skills/kill-argument/SKILL.md +437 -0
  64. package/skills/mermaid-diagram/SKILL.md +419 -0
  65. package/skills/meta-apply/SKILL.md +141 -0
  66. package/skills/meta-optimize/SKILL.md +437 -0
  67. package/skills/monitor-experiment/SKILL.md +140 -0
  68. package/skills/novelty-check/SKILL.md +101 -0
  69. package/skills/openalex/SKILL.md +237 -0
  70. package/skills/overleaf-sync/SKILL.md +220 -0
  71. package/skills/paper-claim-audit/SKILL.md +348 -0
  72. package/skills/paper-compile/SKILL.md +266 -0
  73. package/skills/paper-figure/SKILL.md +312 -0
  74. package/skills/paper-illustration/SKILL.md +736 -0
  75. package/skills/paper-illustration-image2/SKILL.md +391 -0
  76. package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  77. package/skills/paper-plan/SKILL.md +386 -0
  78. package/skills/paper-poster/SKILL.md +19 -0
  79. package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
  80. package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
  81. package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
  82. package/skills/paper-poster-html/NOTICE.md +57 -0
  83. package/skills/paper-poster-html/SKILL.md +323 -0
  84. package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
  85. package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
  86. package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
  87. package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
  88. package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
  89. package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
  90. package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
  91. package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
  92. package/skills/paper-poster-html/scripts/asset_check.py +897 -0
  93. package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
  94. package/skills/paper-poster-html/scripts/poster_check.py +251 -0
  95. package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
  96. package/skills/paper-poster-html/scripts/render_preview.py +217 -0
  97. package/skills/paper-poster-html/scripts/run_gates.py +556 -0
  98. package/skills/paper-poster-html/scripts/style_check.py +1324 -0
  99. package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
  100. package/skills/paper-poster-html/templates/README.md +170 -0
  101. package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
  102. package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
  103. package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
  104. package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
  105. package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
  106. package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
  107. package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
  108. package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
  109. package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
  110. package/skills/paper-slides/SKILL.md +635 -0
  111. package/skills/paper-talk/SKILL.md +381 -0
  112. package/skills/paper-write/SKILL.md +604 -0
  113. package/skills/paper-write/templates/IEEEtran.bst +2409 -0
  114. package/skills/paper-write/templates/IEEEtran.cls +6347 -0
  115. package/skills/paper-write/templates/iclr2026.tex +84 -0
  116. package/skills/paper-write/templates/icml2025.tex +87 -0
  117. package/skills/paper-write/templates/ieee_conference.tex +89 -0
  118. package/skills/paper-write/templates/ieee_journal.tex +93 -0
  119. package/skills/paper-write/templates/math_commands.tex +48 -0
  120. package/skills/paper-write/templates/neurips2025.tex +80 -0
  121. package/skills/paper-writing/SKILL.md +916 -0
  122. package/skills/patent-novelty-check/SKILL.md +153 -0
  123. package/skills/patent-pipeline/SKILL.md +344 -0
  124. package/skills/patent-review/SKILL.md +203 -0
  125. package/skills/pixel-art/SKILL.md +137 -0
  126. package/skills/prior-art-search/SKILL.md +146 -0
  127. package/skills/proof-checker/SKILL.md +866 -0
  128. package/skills/proof-orchestrator/NOTICE.md +24 -0
  129. package/skills/proof-orchestrator/SKILL.md +254 -0
  130. package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
  131. package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
  132. package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
  133. package/skills/proof-orchestrator/references/notation-audit.md +135 -0
  134. package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  135. package/skills/proof-orchestrator/references/stress-tests.md +38 -0
  136. package/skills/proof-writer/SKILL.md +223 -0
  137. package/skills/qzcli/SKILL.md +324 -0
  138. package/skills/rebuttal/SKILL.md +376 -0
  139. package/skills/render-html/SKILL.md +316 -0
  140. package/skills/render-html/scripts/render_html.py +1006 -0
  141. package/skills/render-html/scripts/templates/academic.html +703 -0
  142. package/skills/render-html/scripts/templates/dashboard.html +333 -0
  143. package/skills/research-lit/SKILL.md +756 -0
  144. package/skills/research-pipeline/SKILL.md +384 -0
  145. package/skills/research-refine/SKILL.md +770 -0
  146. package/skills/research-refine-pipeline/SKILL.md +186 -0
  147. package/skills/research-review/SKILL.md +198 -0
  148. package/skills/research-wiki/SKILL.md +461 -0
  149. package/skills/resubmit-pipeline/SKILL.md +447 -0
  150. package/skills/result-to-claim/SKILL.md +311 -0
  151. package/skills/run-experiment/SKILL.md +313 -0
  152. package/skills/semantic-scholar/SKILL.md +236 -0
  153. package/skills/serverless-modal/SKILL.md +335 -0
  154. package/skills/shared-references/acceptance-gate.md +324 -0
  155. package/skills/shared-references/assurance-contract.md +248 -0
  156. package/skills/shared-references/capture-antipatterns.md +78 -0
  157. package/skills/shared-references/citation-discipline.md +583 -0
  158. package/skills/shared-references/compute-env-contract.md +163 -0
  159. package/skills/shared-references/effort-contract.md +183 -0
  160. package/skills/shared-references/evidence-precheck.md +65 -0
  161. package/skills/shared-references/experiment-integrity.md +49 -0
  162. package/skills/shared-references/external-cadence.md +326 -0
  163. package/skills/shared-references/fan-out-pattern.md +366 -0
  164. package/skills/shared-references/injection-hygiene.md +127 -0
  165. package/skills/shared-references/integration-contract.md +461 -0
  166. package/skills/shared-references/output-composition.md +93 -0
  167. package/skills/shared-references/output-language.md +45 -0
  168. package/skills/shared-references/output-manifest.md +49 -0
  169. package/skills/shared-references/output-versioning.md +111 -0
  170. package/skills/shared-references/patent-format-cn.md +199 -0
  171. package/skills/shared-references/patent-format-ep.md +173 -0
  172. package/skills/shared-references/patent-format-us.md +161 -0
  173. package/skills/shared-references/patent-writing-principles.md +197 -0
  174. package/skills/shared-references/prior-art-databases.md +141 -0
  175. package/skills/shared-references/resumable-runs.md +109 -0
  176. package/skills/shared-references/review-scope-limits.md +81 -0
  177. package/skills/shared-references/review-tracing.md +391 -0
  178. package/skills/shared-references/reviewer-independence.md +79 -0
  179. package/skills/shared-references/reviewer-routing.md +852 -0
  180. package/skills/shared-references/skill-governance.md +104 -0
  181. package/skills/shared-references/taste-calibration.md +85 -0
  182. package/skills/shared-references/venue-checklists.md +114 -0
  183. package/skills/shared-references/wiki-helper-resolution.md +134 -0
  184. package/skills/shared-references/writing-principles.md +525 -0
  185. package/skills/skills-codex/README.md +102 -0
  186. package/skills/skills-codex/README_CN.md +100 -0
  187. package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
  188. package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
  189. package/skills/skills-codex/analyze-results/SKILL.md +45 -0
  190. package/skills/skills-codex/arxiv/SKILL.md +210 -0
  191. package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
  192. package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
  193. package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
  194. package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
  195. package/skills/skills-codex/citation-audit/SKILL.md +504 -0
  196. package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
  197. package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
  198. package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
  199. package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
  200. package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
  201. package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
  202. package/skills/skills-codex/deepxiv/SKILL.md +142 -0
  203. package/skills/skills-codex/dse-loop/SKILL.md +285 -0
  204. package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
  205. package/skills/skills-codex/exa-search/SKILL.md +192 -0
  206. package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
  207. package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
  208. package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
  209. package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
  210. package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
  211. package/skills/skills-codex/figure-description/SKILL.md +138 -0
  212. package/skills/skills-codex/figure-spec/SKILL.md +252 -0
  213. package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
  214. package/skills/skills-codex/gemini-search/SKILL.md +205 -0
  215. package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
  216. package/skills/skills-codex/idea-creator/SKILL.md +405 -0
  217. package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
  218. package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
  219. package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
  220. package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
  221. package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
  222. package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
  223. package/skills/skills-codex/kill-argument/SKILL.md +403 -0
  224. package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
  225. package/skills/skills-codex/meta-apply/SKILL.md +154 -0
  226. package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
  227. package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
  228. package/skills/skills-codex/novelty-check/SKILL.md +89 -0
  229. package/skills/skills-codex/openalex/SKILL.md +228 -0
  230. package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
  231. package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
  232. package/skills/skills-codex/paper-compile/SKILL.md +253 -0
  233. package/skills/skills-codex/paper-figure/SKILL.md +311 -0
  234. package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
  235. package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
  236. package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  237. package/skills/skills-codex/paper-plan/SKILL.md +278 -0
  238. package/skills/skills-codex/paper-poster/SKILL.md +19 -0
  239. package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
  240. package/skills/skills-codex/paper-slides/SKILL.md +571 -0
  241. package/skills/skills-codex/paper-talk/SKILL.md +381 -0
  242. package/skills/skills-codex/paper-write/SKILL.md +411 -0
  243. package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
  244. package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
  245. package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
  246. package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
  247. package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
  248. package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
  249. package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
  250. package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
  251. package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
  252. package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
  253. package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
  254. package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
  255. package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
  256. package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
  257. package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
  258. package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
  259. package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
  260. package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
  261. package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
  262. package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
  263. package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
  264. package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
  265. package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
  266. package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
  267. package/skills/skills-codex/paper-writing/SKILL.md +731 -0
  268. package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
  269. package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
  270. package/skills/skills-codex/patent-review/SKILL.md +202 -0
  271. package/skills/skills-codex/pixel-art/SKILL.md +139 -0
  272. package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
  273. package/skills/skills-codex/proof-checker/SKILL.md +554 -0
  274. package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
  275. package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
  276. package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
  277. package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
  278. package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
  279. package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  280. package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
  281. package/skills/skills-codex/proof-writer/SKILL.md +222 -0
  282. package/skills/skills-codex/qzcli/SKILL.md +324 -0
  283. package/skills/skills-codex/rebuttal/SKILL.md +305 -0
  284. package/skills/skills-codex/render-html/SKILL.md +305 -0
  285. package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
  286. package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
  287. package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
  288. package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
  289. package/skills/skills-codex/research-lit/SKILL.md +464 -0
  290. package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
  291. package/skills/skills-codex/research-refine/SKILL.md +721 -0
  292. package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
  293. package/skills/skills-codex/research-review/SKILL.md +135 -0
  294. package/skills/skills-codex/research-wiki/SKILL.md +421 -0
  295. package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
  296. package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
  297. package/skills/skills-codex/run-experiment/SKILL.md +236 -0
  298. package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
  299. package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
  300. package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
  301. package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
  302. package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
  303. package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
  304. package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
  305. package/skills/skills-codex/shared-references/effort-contract.md +143 -0
  306. package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
  307. package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
  308. package/skills/skills-codex/shared-references/external-cadence.md +334 -0
  309. package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
  310. package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
  311. package/skills/skills-codex/shared-references/integration-contract.md +372 -0
  312. package/skills/skills-codex/shared-references/output-composition.md +98 -0
  313. package/skills/skills-codex/shared-references/output-language.md +45 -0
  314. package/skills/skills-codex/shared-references/output-manifest.md +40 -0
  315. package/skills/skills-codex/shared-references/output-versioning.md +111 -0
  316. package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
  317. package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
  318. package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
  319. package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
  320. package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
  321. package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
  322. package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
  323. package/skills/skills-codex/shared-references/review-tracing.md +144 -0
  324. package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
  325. package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
  326. package/skills/skills-codex/shared-references/skill-governance.md +119 -0
  327. package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
  328. package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
  329. package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
  330. package/skills/skills-codex/shared-references/writing-principles.md +525 -0
  331. package/skills/skills-codex/slides-polish/SKILL.md +563 -0
  332. package/skills/skills-codex/specification-writing/SKILL.md +211 -0
  333. package/skills/skills-codex/system-profile/SKILL.md +103 -0
  334. package/skills/skills-codex/training-check/SKILL.md +83 -0
  335. package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
  336. package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
  337. package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
  338. package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
  339. package/skills/skills-codex-claude-review/README.md +79 -0
  340. package/skills/skills-codex-claude-review/README_CN.md +78 -0
  341. package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
  342. package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
  343. package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
  344. package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
  345. package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
  346. package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
  347. package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
  348. package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
  349. package/skills/skills-codex-gemini-review/README.md +176 -0
  350. package/skills/skills-codex-gemini-review/README_CN.md +175 -0
  351. package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
  352. package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
  353. package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
  354. package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
  355. package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
  356. package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
  357. package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
  358. package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
  359. package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
  360. package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
  361. package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
  362. package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
  363. package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
  364. package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
  365. package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
  366. package/skills/slides-polish/SKILL.md +565 -0
  367. package/skills/specification-writing/SKILL.md +211 -0
  368. package/skills/system-profile/SKILL.md +103 -0
  369. package/skills/training-check/SKILL.md +132 -0
  370. package/skills/vast-gpu/SKILL.md +394 -0
  371. package/skills/web-debug-search/SKILL.md +334 -0
  372. package/skills/wiki-enrich/SKILL.md +257 -0
  373. package/skills/writing-systems-papers/SKILL.md +184 -0
  374. package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
  375. package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
  376. package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
  377. package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
  378. package/templates/FINDINGS_TEMPLATE.md +52 -0
  379. package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
  380. package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
  381. package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
  382. package/templates/MANIFEST_TEMPLATE.md +7 -0
  383. package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
  384. package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
  385. package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
  386. package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
  387. package/templates/README.md +57 -0
  388. package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
  389. package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
  390. package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
  391. package/templates/claude-hooks/corpus_write_guard.json +16 -0
  392. package/templates/claude-hooks/corpus_write_guard.py +85 -0
  393. package/templates/claude-hooks/meta_logging.json +74 -0
  394. package/templates/gitignore-trace.txt +3 -0
  395. package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
  396. package/tools/arxiv_fetch.py +311 -0
  397. package/tools/capture_filter.py +126 -0
  398. package/tools/check_skills_inventory.py +273 -0
  399. package/tools/convert_skills_to_llm_chat.py +282 -0
  400. package/tools/copilot_native_evidence.py +818 -0
  401. package/tools/deepxiv_fetch.py +213 -0
  402. package/tools/evidence_check.py +212 -0
  403. package/tools/exa_search.py +425 -0
  404. package/tools/experiment_queue/README.md +118 -0
  405. package/tools/experiment_queue/build_manifest.py +44 -0
  406. package/tools/experiment_queue/queue_manager.py +44 -0
  407. package/tools/extract_paper_style.py +560 -0
  408. package/tools/figure_renderer.py +69 -0
  409. package/tools/forensics_gate.py +669 -0
  410. package/tools/generate_codex_claude_review_overrides.py +299 -0
  411. package/tools/idea_discovery_gate.py +256 -0
  412. package/tools/install_aris.ps1 +1372 -0
  413. package/tools/install_aris.sh +1370 -0
  414. package/tools/install_aris_codex.sh +1023 -0
  415. package/tools/install_aris_copilot.sh +1052 -0
  416. package/tools/iteration_log.py +143 -0
  417. package/tools/lint_skills_helpers.sh +84 -0
  418. package/tools/meta_opt/check_ready.sh +80 -0
  419. package/tools/meta_opt/log_event.sh +91 -0
  420. package/tools/meta_opt/trigger_eval.py +280 -0
  421. package/tools/meta_opt/trigger_evals.sample.json +28 -0
  422. package/tools/openalex_fetch.py +326 -0
  423. package/tools/overleaf_audit.sh +104 -0
  424. package/tools/overleaf_setup.sh +150 -0
  425. package/tools/paper_illustration_image2.py +62 -0
  426. package/tools/provenance.py +294 -0
  427. package/tools/research_wiki.py +1720 -0
  428. package/tools/review_gate.py +502 -0
  429. package/tools/run_state.py +399 -0
  430. package/tools/save_trace.sh +477 -0
  431. package/tools/semantic_scholar_fetch.py +438 -0
  432. package/tools/skill-groups.tsv +116 -0
  433. package/tools/skill_picker.py +238 -0
  434. package/tools/smart_update.ps1 +521 -0
  435. package/tools/smart_update.sh +591 -0
  436. package/tools/smart_update_codex.sh +419 -0
  437. package/tools/smart_update_copilot.sh +605 -0
  438. package/tools/threat_scan.py +222 -0
  439. package/tools/verify_paper_audits.sh +487 -0
  440. package/tools/verify_papers.py +613 -0
  441. package/tools/verify_wiki_coverage.sh +176 -0
  442. package/tools/watchdog.py +485 -0
@@ -0,0 +1,363 @@
1
+ ---
2
+ name: idea-discovery-robot
3
+ description: "Workflow 1 adaptation for robotics and embodied AI. Orchestrates robotics-aware literature survey, idea generation, novelty check, and critical review to go from a broad robotics direction to benchmark-grounded, simulation-first ideas. Use when user says \"robotics idea discovery\", \"机器人找idea\", \"embodied AI idea\", \"机器人方向探索\", \"sim2real 选题\", or wants ideas for manipulation, locomotion, navigation, drones, humanoids, or general robot learning."
4
+ argument-hint: "[robotics-direction]"
5
+ allowed-tools: Bash(*), Read, Write, Edit, Grep, Glob, WebSearch, WebFetch, Skill, mcp__codex__codex, mcp__codex__codex-reply
6
+ ---
7
+
8
+ # Robotics Idea Discovery Pipeline
9
+
10
+ Orchestrate a robotics-specific idea discovery workflow for: **$ARGUMENTS**
11
+
12
+ ## Overview
13
+
14
+ This skill chains four sub-skills into a single automated pipeline:
15
+
16
+ ```
17
+ /research-lit → /idea-creator (robotics framing) → /novelty-check → /research-review
18
+ (survey) (filter + pilot plan) (verify novel) (critical feedback)
19
+ ```
20
+
21
+ But every phase must be grounded in robotics-specific constraints:
22
+ - **Embodiment**: arm, mobile manipulator, drone, humanoid, quadruped, autonomous car, etc.
23
+ - **Task family**: grasping, insertion, locomotion, navigation, manipulation, rearrangement, multi-step planning
24
+ - **Observation + action interface**: RGB/RGB-D/tactile/language; torque/velocity/waypoints/end-effector actions
25
+ - **Simulator / benchmark availability**: simulation-first by default
26
+ - **Real robot constraints**: hardware availability, reset cost, safety, operator time
27
+ - **Evaluation quality**: success rate plus failure cases, safety violations, intervention count, latency, sample efficiency
28
+ - **Sim2real story**: whether the idea can stay in sim, needs offline logs, or truly requires hardware
29
+
30
+ The goal is not to produce flashy demos. The goal is to produce ideas that are:
31
+ - benchmarkable
32
+ - falsifiable
33
+ - feasible with available robotics infrastructure
34
+ - interesting even if the answer is negative
35
+
36
+ ## Constants
37
+
38
+ - **MAX_PILOT_IDEAS = 3** — Validate at most 3 top ideas deeply
39
+ - **PILOT_MODE = `sim-first`** — Prefer simulation or offline-log pilots before any hardware execution
40
+ - **REAL_ROBOT_PILOTS = `explicit approval only`** — Never assume physical robot access or approval
41
+ - **AUTO_PROCEED = true** — If user does not respond at checkpoints, proceed with the best sim-first option
42
+ - **REVIEWER_MODEL = `gpt-5.6-sol`** — External reviewer model via Codex MCP
43
+ - **TARGET_VENUES = CoRL, RSS, ICRA, IROS, RA-L** — Default novelty and reviewer framing
44
+
45
+ > Override inline, e.g. `/idea-discovery-robot "bimanual manipulation" — only sim ideas, no real robot` or `/idea-discovery-robot "drone navigation" — focus on CoRL/RSS, 2 pilot ideas max`
46
+
47
+ ## Execution Rule
48
+
49
+ Follow the phases in order. Do **not** stop after a checkpoint unless:
50
+ - the user explicitly says to stop, or
51
+ - the user asks to change scope and re-run an earlier phase
52
+
53
+ If `AUTO_PROCEED=true` and the user does not respond, continue immediately to the next phase using the strongest **sim-first, benchmark-grounded** option.
54
+
55
+ ## Phase 0: Frame the Robotics Problem
56
+
57
+ Before generating ideas, extract or infer this **Robotics Problem Frame** from `$ARGUMENTS` and local project context:
58
+
59
+ - **Embodiment**
60
+ - **Task family**
61
+ - **Environment type**: tabletop, warehouse, home, outdoor, aerial, driving, legged terrain
62
+ - **Observation modalities**
63
+ - **Action interface / controller abstraction**
64
+ - **Learning regime**: RL, imitation, behavior cloning, world model, planning, VLA/VLM, classical robotics, hybrid
65
+ - **Available assets**: simulator, benchmark suite, teleop data, offline logs, existing codebase, real hardware
66
+ - **Compute budget**
67
+ - **Safety constraints**
68
+ - **Desired contribution type**: method, benchmark, diagnosis, systems, sim2real, data curation
69
+
70
+ If some fields are missing, make explicit assumptions and default to:
71
+ - **simulation-first**
72
+ - **public benchmark preferred**
73
+ - **no real robot execution**
74
+
75
+ Write this frame into working notes before moving on. Every later decision should reference it.
76
+
77
+ ## Phase 1: Robotics Literature Survey
78
+
79
+ Invoke:
80
+
81
+ ```
82
+ /research-lit "$ARGUMENTS — focus venues: CoRL, RSS, ICRA, IROS, RA-L, TRO, Science Robotics"
83
+ ```
84
+
85
+ Then reorganize the findings using a robotics lens instead of a generic ML lens.
86
+
87
+ ### Build a Robotics Landscape Matrix
88
+
89
+ For each relevant paper, classify:
90
+
91
+ | Axis | Examples |
92
+ |------|----------|
93
+ | Embodiment | single-arm, mobile manipulator, humanoid, drone, quadruped |
94
+ | Task | pick-place, insertion, navigation, locomotion, long-horizon rearrangement |
95
+ | Learning setup | RL, BC, IL, offline RL, world model, planning, diffusion policy |
96
+ | Observation | RGB, RGB-D, proprioception, tactile, language |
97
+ | Action abstraction | torque, joint velocity, end-effector delta pose, waypoint planner |
98
+ | Eval regime | pure sim, sim+real, real-only, offline benchmark |
99
+ | Benchmark | ManiSkill, RLBench, Isaac Lab, Habitat, Meta-World, CALVIN, LIBERO, custom |
100
+ | Metrics | success rate, collision rate, intervention count, path length, latency, energy |
101
+ | Main bottleneck | sample inefficiency, brittleness, reset cost, perception drift, sim2real gap |
102
+
103
+ ### Search Priorities
104
+
105
+ When refining the survey, prioritize:
106
+ - recent work from **CoRL, RSS, ICRA, IROS, RA-L**
107
+ - recent arXiv papers from the last 6-12 months
108
+ - benchmark papers and follow-up reproductions
109
+ - negative-result or diagnosis papers if they reveal system bottlenecks
110
+
111
+ ### What to Look For
112
+
113
+ Do not stop at "who got the best success rate." Explicitly identify:
114
+ - recurring failure modes papers do not fix
115
+ - benchmarks that are saturated or misleading
116
+ - places where embodiment changes invalidate prior conclusions
117
+ - methods that only work with privileged observations
118
+ - ideas whose reported gains come from reset engineering, reward shaping, or hidden infrastructure
119
+ - task families where evaluation quality is weak even if performance numbers look high
120
+
121
+ **Checkpoint:** Present the landscape to the user in robotics terms:
122
+
123
+ ```
124
+ 🤖 Robotics survey complete. I grouped the field by embodiment, benchmark, action interface, and sim2real setup.
125
+
126
+ Main gaps:
127
+ 1. [...]
128
+ 2. [...]
129
+ 3. [...]
130
+
131
+ Should I generate ideas under this framing, or should I narrow to a specific robot / benchmark / modality?
132
+ ```
133
+
134
+ - **User approves** (or no response + AUTO_PROCEED=true) → proceed to Phase 2 with the best robotics frame.
135
+ - **User requests changes** (e.g. narrower embodiment, different benchmark family, no sim2real, no hardware) → refine the robotics frame, re-run Phase 1, and present again.
136
+
137
+ ## Phase 2: Robotics-Specific Idea Generation and Filtering
138
+
139
+ Generate ideas only after the robotics frame is explicit.
140
+
141
+ Invoke the existing idea generator, but pass the **Robotics Problem Frame** and landscape matrix into the prompt so it does not produce generic ML ideas:
142
+
143
+ ```
144
+ /idea-creator "$ARGUMENTS — robotics frame: [paste Robotics Problem Frame] — focus venues: CoRL, RSS, ICRA, IROS, RA-L — benchmark-specific ideas only — sim-first pilots — no real-robot execution without explicit approval — require failure metrics and baseline clarity"
145
+ ```
146
+
147
+ Then rewrite and filter the output using the robotics-specific rules below.
148
+
149
+ Each candidate idea must include:
150
+ - **One-sentence summary**
151
+ - **Target embodiment**
152
+ - **Target benchmark / simulator / dataset**
153
+ - **Core bottleneck being addressed**
154
+ - **Minimum sim-first pilot**
155
+ - **Mandatory metrics**
156
+ - **Expected failure mode if the idea does not work**
157
+ - **Whether the idea truly needs real hardware**
158
+
159
+ ### Good Robotics Idea Patterns
160
+
161
+ Prefer ideas that:
162
+ - expose a real bottleneck in perception-action coupling
163
+ - improve robustness under embodiment or environment shift
164
+ - reduce operator time, reset cost, or demonstration cost
165
+ - strengthen sim2real transfer with measurable mechanisms
166
+ - improve recovery, retry behavior, or failure detection
167
+ - create a better benchmark, diagnostic, or evaluation protocol
168
+ - test an assumption the community repeats but rarely measures
169
+
170
+ ### Weak Robotics Idea Patterns
171
+
172
+ Downrank ideas that are mostly:
173
+ - "apply a foundation model / VLM / diffusion model to robot X" with no new bottleneck analysis
174
+ - demo-driven but not benchmarkable
175
+ - dependent on inaccessible hardware, custom sensors, or massive private datasets
176
+ - impossible to evaluate without a months-long infrastructure build
177
+ - only interesting if everything works perfectly
178
+
179
+ ### Filtering Rules
180
+
181
+ For each idea, reject or heavily downrank if:
182
+ - no concrete simulator or benchmark is available
183
+ - no credible baseline exists
184
+ - no measurable metric beyond "looks better"
185
+ - real robot execution is required but hardware access is unclear
186
+ - the setup depends on privileged observations that make the claim weak
187
+ - the expected contribution disappears if evaluation is made fair
188
+
189
+ **Checkpoint:** Present the ranked robotics ideas before novelty checking:
190
+
191
+ ```
192
+ 💡 Robotics ideas generated. Top candidates:
193
+
194
+ 1. [Idea 1] — Embodiment: [...] — Benchmark: [...] — Pilot: sim/offline — Risk: LOW/MEDIUM/HIGH
195
+ 2. [Idea 2] — Embodiment: [...] — Benchmark: [...] — Pilot: sim/offline — Risk: LOW/MEDIUM/HIGH
196
+ 3. [Idea 3] — requires hardware / weak benchmark / high risk
197
+
198
+ Should I carry the top sim-first ideas into novelty checking and external review?
199
+ (If no response, I'll continue with the strongest benchmark-grounded ideas.)
200
+ ```
201
+
202
+ - **User picks ideas** (or no response + AUTO_PROCEED=true) → proceed to Phase 3 with the top sim-first ideas, then continue to Phase 4 and Phase 5.
203
+ - **User wants different constraints** → update the robotics frame and re-run Phase 2.
204
+ - **User wants narrower scope** → go back to Phase 1 with a tighter embodiment / task / benchmark focus.
205
+
206
+ ## Phase 3: Feasibility and Pilot Design
207
+
208
+ For the top ideas, design a **minimal validation package**.
209
+
210
+ If the repository already contains a usable simulator, benchmark harness, or offline dataset pipeline, you may validate the top 1-3 ideas there. If not, do **not** force execution. Produce a concrete pilot plan instead.
211
+
212
+ By default, pilots should be one of:
213
+ - **simulation pilot**
214
+ - **offline log / dataset pilot**
215
+ - **analysis-only pilot** using existing benchmark outputs
216
+
217
+ Only propose a real-robot pilot if the user explicitly wants that.
218
+
219
+ For each surviving idea, specify:
220
+
221
+ ```markdown
222
+ - Embodiment:
223
+ - Benchmark / simulator:
224
+ - Baselines:
225
+ - Pilot type: sim / offline / real
226
+ - Compute estimate:
227
+ - Human/operator time:
228
+ - Success metrics:
229
+ - Failure metrics:
230
+ - Safety concerns:
231
+ - What result would count as positive signal:
232
+ - What negative result would still be publishable:
233
+ ```
234
+
235
+ ### Real Robot Rule
236
+
237
+ **Never auto-proceed to physical robot testing.** If an idea needs hardware:
238
+ - mark it as `needs physical validation`
239
+ - design the sim or offline precursor first
240
+ - ask for explicit user confirmation before any real-robot step
241
+
242
+ If no cheap sim/offline pilot exists, keep the idea in the report but label it **high execution risk**.
243
+
244
+ After Phase 3, continue to Phase 4 even if you only produced a pilot plan rather than running a pilot. Lack of immediate execution is not a reason to stop the workflow.
245
+
246
+ ## Phase 4: Deep Novelty Verification
247
+
248
+ For each top idea, run:
249
+
250
+ ```
251
+ /novelty-check "[idea description with embodiment + task family + benchmark + sensor stack + controller/policy class + sim2real angle + target venues: CoRL/RSS/ICRA/IROS/RA-L]"
252
+ ```
253
+
254
+ Robotics novelty checks must include:
255
+ - embodiment
256
+ - task family
257
+ - benchmark / simulator
258
+ - sensor stack
259
+ - controller / policy type
260
+ - sim2real or safety angle if relevant
261
+
262
+ Be especially skeptical of ideas that are just:
263
+ - old method + new benchmark
264
+ - VLA/VLM + standard manipulation benchmark
265
+ - sim2real claim without new transfer mechanism
266
+
267
+ If the method is not novel but the **finding** or **evaluation protocol** is, say that explicitly.
268
+
269
+ ## Phase 5: External Robotics Review
270
+
271
+ Invoke:
272
+
273
+ ```
274
+ /research-review "[top idea with robotics framing, embodiment, benchmark, baselines, pilot plan, evaluation metrics, and sim2real/hardware risks — review as CoRL/RSS/ICRA reviewer]"
275
+ ```
276
+
277
+ Frame the reviewer as a senior **CoRL / RSS / ICRA** reviewer. Ask them to focus on:
278
+ - whether the contribution is really new for robotics, not just ML
279
+ - the minimum benchmark package needed for credibility
280
+ - whether the sim2real story is justified
281
+ - missing baselines or failure analyses
282
+ - whether the idea survives realistic infrastructure constraints
283
+
284
+ Update the report with the reviewer's minimum viable evidence package.
285
+
286
+ ## Phase 6: Final Report
287
+
288
+ Write or update `idea-stage/IDEA_REPORT.md` with a robotics-specific structure so it stays compatible with downstream workflows.
289
+
290
+ ```markdown
291
+ # Robotics Idea Discovery Report
292
+
293
+ **Direction**: $ARGUMENTS
294
+ **Date**: [today]
295
+ **Pipeline**: research-lit → idea-creator (robotics framing) → novelty-check → research-review
296
+
297
+ ## Robotics Problem Frame
298
+ - Embodiment:
299
+ - Task family:
300
+ - Observation / action interface:
301
+ - Available assets:
302
+ - Constraints:
303
+
304
+ ## Landscape Matrix
305
+ [grouped by embodiment, benchmark, and bottleneck]
306
+
307
+ ## Ranked Ideas
308
+
309
+ ### Idea 1: [title] — RECOMMENDED
310
+ - Embodiment:
311
+ - Benchmark / simulator:
312
+ - Bottleneck addressed:
313
+ - Pilot type: sim / offline / real
314
+ - Positive signal:
315
+ - Novelty:
316
+ - Reviewer score:
317
+ - Hardware risk:
318
+ - Next step:
319
+
320
+ ## Eliminated Ideas
321
+ - [idea] — killed because benchmark unclear / hardware inaccessible / novelty weak / no fair evaluation
322
+
323
+ ## Evidence Package for the Top Idea
324
+ - Required baselines:
325
+ - Required metrics:
326
+ - Required failure cases:
327
+ - Whether real robot evidence is mandatory:
328
+
329
+ ## Next Steps
330
+ - [ ] Implement sim-first pilot
331
+ - [ ] Run /novelty-check on the final idea wording
332
+ - [ ] Only after approval: consider hardware validation
333
+ ```
334
+
335
+ ## Key Rules
336
+
337
+ - **Simulation first.** Hardware is never the default.
338
+ - **Benchmark specificity is mandatory.** No benchmark, no serious idea.
339
+ - **Evaluation must include failures.** Success rate alone is not enough.
340
+ - **Embodiment matters.** Do not assume a result on one robot transfers to another.
341
+ - **Avoid foundation-model theater.** Novel terminology is not novelty.
342
+ - **Infrastructure realism matters.** Operator time, reset burden, and safety count as research constraints.
343
+ - **If the contribution is mainly diagnostic or evaluative, say so.** That can still be publishable.
344
+
345
+ ## Composing with Later Work
346
+
347
+ After this workflow identifies a strong robotics idea:
348
+
349
+ ```
350
+ /idea-discovery-robot "direction" ← you are here
351
+ implement sim-first pilot
352
+ /run-experiment ← if infrastructure exists
353
+ /auto-review-loop "top robotics idea"
354
+ ```
355
+
356
+ If no simulator or benchmark is available yet, stop at the report and ask the user to choose whether to build infrastructure or pivot to a more executable idea.
357
+
358
+ ## Output Protocols
359
+
360
+ > Follow these shared protocols for all output files:
361
+ > - **[Output Versioning Protocol](../shared-references/output-versioning.md)** — write timestamped file first, then copy to fixed name
362
+ > - **[Output Manifest Protocol](../shared-references/output-manifest.md)** — log every output to MANIFEST.md
363
+ > - **[Output Language Protocol](../shared-references/output-language.md)** — respect the project's language setting
@@ -0,0 +1,284 @@
1
+ ---
2
+ name: integrity-forensics
3
+ description: "Run the Anti-Autoresearch integrity-forensics sweep (span-anchored evidence ledger → GPT auditors propose findings → a rules-only reporter that lists every proposal with what the auditor said about it) against a paper via a SHA-pinned thin launcher — then convert the verdict into a typed policy gate (BLOCK/WARN/NO_NEW_BLOCKER) and an append-only obligations ledger. Use when user says \"integrity forensics\", \"forensic audit this paper\", \"投稿前自查诚信\", \"审这篇论文的诚信\", or says \"anti-autoresearch\" when the upstream repo's own skills are not installed. Also invoked by /paper-writing (submission self-forensics, default ON), /peer-review (forensic appendix), /resubmit-pipeline."
4
+ argument-hint: "[paper-dir | pdf | arxiv-id]"
5
+ allowed-tools: Bash(*), Read, Write, Grep, Glob, mcp__codex__codex
6
+ ---
7
+
8
+ # Integrity Forensics — thin launcher for Anti-Autoresearch
9
+
10
+ Audit target: **$ARGUMENTS**
11
+
12
+ > **What this is.** ARIS generates papers; [Anti-Autoresearch](https://github.com/wanshuiyin/Anti-Autoresearch)
13
+ > is its outward-pointed dual — reviewer-side integrity forensics (46 patterns
14
+ > across 8 families, deterministic GRIM/GRIMMER/statcheck core, span-anchored
15
+ > claims, a rules-only reporter that summarizes rather than adjudicates). This skill is a
16
+ > **thin launcher**: it pins an upstream commit, validates the pin with the
17
+ > upstream eval gate, delegates execution unchanged, and post-processes the
18
+ > verdict into ARIS's policy vocabulary. It vendors nothing and forks nothing.
19
+
20
+ > 🔁 **Cadence fence** (`shared-references/external-cadence.md`): this skill is
21
+ > verdict-bearing decision support. Do not wrap it in `/loop` / `/schedule` —
22
+ > and NEVER as "iterate edits until it stops flagging" (see The One Forbidden
23
+ > Loop below).
24
+
25
+ ## Constants
26
+
27
+ - **ANTI_AR_REPO = `https://github.com/wanshuiyin/Anti-Autoresearch.git`**
28
+ - **ANTI_AR_COMMIT = `b47af6f983b38347b6d2110379e266400597cf66`** — the SHA-pin.
29
+ The launcher NEVER tracks upstream HEAD; bumping this constant is a reviewed
30
+ change (see Pin-bump checklist).
31
+ - **CLONE_DIR = `~/.aris/anti-autoresearch`** — the pinned working copy. Host-neutral
32
+ on purpose: ARIS also runs on DeepSeek Harness, Codex CLI, Cursor, Trae,
33
+ Antigravity and Copilot CLI, where `~/.claude/` would name an installation the
34
+ user does not have. An older clone at `~/.claude/anti-autoresearch` is unused;
35
+ move it and its `.aris_eval_ok_*` receipt only to keep an offline
36
+ deterministic-only run working, otherwise delete it whenever convenient.
37
+ - **NO REVIEWER KNOBS.** This launcher exposes no reviewer model/effort
38
+ parameters and never maps ARIS `— effort:` onto upstream settings. The
39
+ pinned upstream runs exactly what it pins (`gpt-5.6-sol` + `xhigh`, its own
40
+ design decision). Overriding upstream review policy from a launcher would
41
+ create a second, unauditable configuration surface.
42
+ - **GATE_HELPER = `forensics_gate.py`** — resolved via the canonical chain
43
+ (`shared-references/integration-contract.md` §2): `.aris/tools/` →
44
+ `tools/` → `$ARIS_REPO/tools/` → `$ARIS_REPO/tools/` via `~/.aris/repo`.
45
+ Failure policy A (required): if it cannot be resolved at
46
+ `assurance: submission`, STOP — never improvise the gate.
47
+
48
+ ## Step 0 — Bootstrap the pin (idempotent)
49
+
50
+ ```bash
51
+ CLONE_DIR="$HOME/.aris/anti-autoresearch"
52
+ ANTI_AR_COMMIT="b47af6f983b38347b6d2110379e266400597cf66"
53
+
54
+ mkdir -p "$HOME/.aris"
55
+ if [ ! -d "$CLONE_DIR/.git" ]; then
56
+ git clone --no-checkout https://github.com/wanshuiyin/Anti-Autoresearch.git "$CLONE_DIR"
57
+ fi
58
+ # fetch ONLY if the pin isn't already present — a cached, validated pin works offline
59
+ git -C "$CLONE_DIR" cat-file -e "$ANTI_AR_COMMIT^{commit}" 2>/dev/null \
60
+ || git -C "$CLONE_DIR" fetch -q origin
61
+ git -C "$CLONE_DIR" checkout -qf "$ANTI_AR_COMMIT" || {
62
+ echo "FATAL: cannot checkout pinned commit $ANTI_AR_COMMIT"; exit 1; }
63
+ # Force a PRISTINE tree at the pin — local tampering with the clone (edited
64
+ # adjudicator, injected module, even one hidden inside a NESTED git repo,
65
+ # which single-f clean skips) must not survive bootstrap and run under the
66
+ # official pin's name. Every step is checked; then the tree is verified.
67
+ git -C "$CLONE_DIR" reset --hard -q "$ANTI_AR_COMMIT" || {
68
+ echo "FATAL: reset to pin failed"; exit 1; }
69
+ git -C "$CLONE_DIR" clean -ffdxq || {
70
+ echo "FATAL: clean failed"; exit 1; }
71
+ [ -z "$(git -C "$CLONE_DIR" status --porcelain)" ] || {
72
+ echo "FATAL: clone is not pristine after reset+clean — refusing to run"; exit 1; }
73
+
74
+ # One-time-per-pin validation: the upstream eval gate (8 injected-defect
75
+ # classes, 100% recall + zero clean false positives) must PASS before this
76
+ # pin is allowed to produce a verdict. NEVER skip; NEVER proceed on failure.
77
+ # The marker lives OUTSIDE the clone: a marker inside a tamperable tree proves
78
+ # nothing (and `git clean` above would erase it, forcing re-eval every run).
79
+ MARKER="${CLONE_DIR}.aris_eval_ok_${ANTI_AR_COMMIT}"
80
+ if [ ! -f "$MARKER" ]; then
81
+ ( cd "$CLONE_DIR" && python3 eval/run_eval.py ) || {
82
+ echo "FATAL: upstream eval gate FAILED at pin $ANTI_AR_COMMIT — refusing to"
83
+ echo " use an unvalidated forensics pin for verdicts."; exit 1; }
84
+ touch "$MARKER"
85
+ fi
86
+ echo "anti-autoresearch pinned at $ANTI_AR_COMMIT (eval gate: validated)"
87
+ ```
88
+
89
+ ## Step 1 — Delegate: run the upstream sweep, unchanged
90
+
91
+ Open and follow **`$CLONE_DIR/workflows/anti-autoresearch/SKILL.md`** end to
92
+ end on the target. Two wrapper rules — the ONLY things this launcher adds:
93
+
94
+ 1. **cwd.** Upstream skills self-locate via `git rev-parse --show-toplevel`.
95
+ Run every upstream bash block with `cd "$CLONE_DIR"` first — ALWAYS the cd,
96
+ never just an exported `ROOT` (upstream blocks re-derive ROOT themselves
97
+ and would overwrite it) — and refer to the paper by **absolute path**,
98
+ otherwise upstream resolves ROOT to the ARIS repo and finds the wrong
99
+ Python spine.
100
+ 2. **Codex calls carry `approval-policy: never` + `sandbox: read-only`**
101
+ (session hygiene; upstream already specifies fresh-thread-per-dimension,
102
+ serial execution, and its own model pins — do not alter them).
103
+
104
+ Everything else — the evidence ledger, coverage.json state machine, the nine
105
+ auditor dimensions, the refutation pass, the deterministic summary — is
106
+ upstream's contract. **Never rewrite, soften, or re-map its outputs**
107
+ (`report.json` + `REPORT.md`, verdict ∈ CLEAN_GIVEN_EVIDENCE / SOFT_FLAGS /
108
+ HARD_FLAGS / REVIEW_UNAVAILABLE). The observability level (L0/L1/L2) is
109
+ whatever upstream derives from the artifacts present — do not promise L2.
110
+
111
+ ## Step 2 — Typed gate + obligations (ARIS-side post-processing)
112
+
113
+ ```bash
114
+ # Resolve $GATE_HELPER via the canonical chain (integration-contract §2), then
115
+ # ONE atomic call (update + gate in a single locked transaction — the gate only
116
+ # ever speaks for the report the ledger has folded, sha-bound):
117
+ python3 "$GATE_HELPER" evaluate --report "$PAPER_DIR/report.json" --paper-dir "$PAPER_DIR" \
118
+ --anti-ar-commit "$ANTI_AR_COMMIT" --executor-model "<this pipeline's executor>"
119
+ # exit 0 = WARN / NO_NEW_BLOCKER · exit 1 = BLOCK
120
+ ```
121
+
122
+ The gate translates the verdict into policy WITHOUT re-labeling it:
123
+
124
+ | upstream verdict | policy |
125
+ |---|---|
126
+ | `HARD_FLAGS` | **BLOCK** — an auditor proposed something critical and it is on the table for you to read; never "the machine found fraud" |
127
+ | `REVIEW_UNAVAILABLE` | **BLOCK** — an incomplete sweep cannot wave a paper through |
128
+ | `SOFT_FLAGS` | **WARN** — human disposition. Read the never-ran list too: the upstream verdict folds incompleteness in only when it would otherwise be clean, so a WARN can sit on top of a sweep where verdict-bearing dimensions never ran. `evaluate` and `fresh` both print those dimensions |
129
+ | `CLEAN_GIVEN_EVIDENCE` | **NO_NEW_BLOCKER** — *never* called PASS or accepted: it means "no flag found in the evidence at hand", not an acquittal |
130
+ | anything else | **BLOCK** (fail closed) |
131
+
132
+ plus: any OPEN critical obligation → BLOCK; any OPEN obligation → at least
133
+ WARN; a closed-without-receipt or unknown-status ledger entry → BLOCK (a
134
+ hand-edited `"status": "RESOLVED"` does not open the gate).
135
+
136
+ `gate.json` also records a `paper_fingerprint` (sha over the paper's compile
137
+ inputs AND deliverables — `.tex`/`.bib`/`.sty`/`.cls`/figures/PDF). The
138
+ downstream preflight is ONE command:
139
+ `python3 "$GATE_HELPER" fresh --paper-dir "$PAPER_DIR" --anti-ar-commit "$ANTI_AR_COMMIT"`
140
+ — exit 0 ⟺ the gate was produced at the CURRENT pin ∧ a gate
141
+ exists ∧ nothing in the paper changed after it ∧ the gate matches the current
142
+ obligations ledger ∧ the decision — **re-computed from the sha-verified
143
+ archived report (`last_report.json`) + the live ledger, never read from the
144
+ gate's stored token** — is pass-capable (`WARN` / `NO_NEW_BLOCKER`). Anything
145
+ else — missing gate, post-gate edit or recompile, unbound ledger or archive,
146
+ recompute mismatch, `BLOCK`, unknown token — exits 1: re-run the sweep +
147
+ `evaluate`. Every ledger mutation (`update`/`resolve`/`waive`) deletes the
148
+ standing `gate.json`, so an interrupted run can never leave a stale pass; and
149
+ `evaluate` refuses a report OLDER than any paper file (a stale report cannot
150
+ be folded onto text it never audited). Run `evaluate` immediately after the
151
+ sweep, before touching any paper file.
152
+
153
+ The gate artifact also records honest provenance: upstream's auditors are
154
+ GPT-family, so for a **Claude executor** the findings carry `cross-family`
155
+ proposal provenance; for a **Codex executor** they are `same-family`. Either
156
+ way this gate only raises flags — it has no acceptance to grant, so the
157
+ distinction is informational, not a loophole.
158
+
159
+ ## Step 3 — Fix what it found (obligations, not a polish loop)
160
+
161
+ Every OPEN obligation gets DISPOSITIONED — fixed, or explicitly waived. Upstream now
162
+ reports every proposal an auditor made rather than deciding which ones do not count, so
163
+ expect more obligations than a pre-2026-08 sweep opened, and expect some of them to be
164
+ proposals you disagree with. **`waive` is a first-class, expected outcome** — "a model
165
+ proposed this and I, the human, judge it wrong" is a normal disposition here, not a last
166
+ resort. Weigh each one against the report's columns: `Anchored`, `Observability`,
167
+ `FP-risk`, `Surface`, `Ext-check`.
168
+
169
+ For the ones that are real, use the right door:
170
+
171
+ | Finding family | Repair route |
172
+ |---|---|
173
+ | A — numeric self-consistency | recompute from the RESULT FILES (`/paper-claim-audit` evidence chain); fix the number, not the sentence |
174
+ | D — experiment integrity | back to `/experiment-audit` / rerun |
175
+ | E — citations | `/citation-audit` KEEP/FIX/REPLACE machinery |
176
+ | G — proof & derivation | `/proof-checker`'s fix loop |
177
+ | B / C / H — scope, baselines, eval design | science-level: feed the finding to `/auto-review-loop` as reviewer INPUT, or to the human |
178
+ | AIS / advisory (zero-weight) | optional context for `/auto-paper-improvement-loop`; never gates |
179
+
180
+ Close each obligation explicitly — the receipt is typed and hashed:
181
+
182
+ ```bash
183
+ python3 "$GATE_HELPER" resolve --paper-dir "$PAPER_DIR" --obligation-id <id> \
184
+ --fix-type corrected-from-results|claim-narrowed|claim-withdrawn|citation-replaced \
185
+ --evidence <path-to-the-ground-truth-that-backs-the-fix> \
186
+ --verified-by "human:<name>" | "checker:<tool>" | "cross-family-review:<thread-id>"
187
+ # or, with HUMAN sign-off only:
188
+ python3 "$GATE_HELPER" waive --paper-dir "$PAPER_DIR" --obligation-id <id> \
189
+ --approver "human:<name>" --reason "<why this stands as-is>"
190
+ ```
191
+
192
+ Rules the ledger enforces mechanically (`tests/test_forensics_gate.py`):
193
+ - **append-only** — re-running the sweep can open obligations, never close them;
194
+ - a finding that *disappears* from a later report stays OPEN and gains
195
+ `UNRESOLVED_DISAPPEARANCE` — rewording the span is not a fix;
196
+ - `claim-withdrawn` is an honest fix (deleting an unsupported claim is a
197
+ legitimate resolution — with the deletion diff as evidence);
198
+ - a **waiver is not a resolution**: human-approved, permanently recorded,
199
+ original finding snapshot immutable;
200
+ - the executor's `fix_type` label is a receipt, not a verdict — closure of a
201
+ critical needs a family checker, a fresh cross-family review, or a human
202
+ (`--verified-by` requires TYPED provenance and is recorded; naming a human
203
+ who did not approve is a false record with a permanent paper trail);
204
+ - receipts are **re-verified, not remembered**: on every later gate the
205
+ evidence file must still exist and still hash to what was recorded at
206
+ closure time — editing the evidence after closing re-opens the BLOCK;
207
+ - `resolve`/`waive` (like `update`) **invalidate the standing `gate.json`** —
208
+ finish Step 3 by re-running the sweep + `evaluate`, so the gate that
209
+ downstream preflights read reflects the post-fix state.
210
+
211
+ ### The One Forbidden Loop
212
+
213
+ **Never run "edit → re-sweep → repeat until CLEAN".** That objective function
214
+ teaches the editor to defeat the detector — deleting an anchored span kills a
215
+ flag faster than fixing the number, and the result is a paper laundered
216
+ against its own audit. The re-run after fixes exists to confirm the
217
+ DISCREPANCY is gone (and to catch new ones); the obligations ledger — not the
218
+ verdict — decides whether the gate opens.
219
+
220
+ ## Trust boundary (what is computed vs what is protocol)
221
+
222
+ - **Computed** (the gate enforces these mechanically): verdict→policy mapping,
223
+ append-only ledger lifecycle, sha bindings (report ↔ ledger ↔ archive),
224
+ receipt re-hashing, the paper fingerprint, pin/version match, and the
225
+ recomputed decision (`fresh` never trusts a stored token).
226
+ - **Protocol** (instruction-graded, deliberately): that the sweep actually ran
227
+ at the pinned clone against this paper. The gate raises the bar —
228
+ structural floor (a report must name its adjudicator and carry a coverage
229
+ map), stale-report mtime guard — and that is where it stops. There is no
230
+ cryptographic binding between the report and the paper, deliberately: this is
231
+ a research-workflow gate, not a provenance system, and the honest statement is
232
+ that a determined executor can hand it a stale report. Likewise `human:` / `checker:` / `cross-family-review:` labels are
233
+ accountability, not authentication: a false label is an explicit,
234
+ permanent false record.
235
+ - **Out of scope**: a party rewriting the `.aris/` artifacts consistently with
236
+ shell access has owner power (they could delete the directory outright).
237
+ The gate defends against the sloppy or corner-cutting executor and against
238
+ honest crashes/races/resumes — not against the machine's owner.
239
+
240
+ ## Pin-bump checklist (maintainers)
241
+
242
+ 1. Set the new `ANTI_AR_COMMIT`; delete no markers (the eval gate re-runs
243
+ automatically for the new SHA).
244
+ 2. Diff upstream's `schemas/report.schema.json` + verdict vocabulary against
245
+ the gate's policy table; extend `tools/forensics_gate.py` BEFORE bumping if
246
+ they moved.
247
+ 3. Old findings/obligations stay valid (fingerprints are span/hash-based, not
248
+ id-based) — but findings produced by an older adjudicator must be
249
+ **re-audited, not re-adjudicated** (upstream's own migration rule).
250
+ 4. Tell users when a bump changes how much they must disposition. `fresh`
251
+ rejects every stored `gate.json` at the old pin with `PIN_MISMATCH`, so a
252
+ bump already forces a re-sweep for everyone — bundle upstream changes behind
253
+ ONE bump rather than two, or the re-sweep cost is paid twice.
254
+
255
+ > **2026-08 bump (`98a75fc`) — expect more open obligations.** Upstream moved
256
+ > from adjudicating proposals to reporting them: findings its FP-risk,
257
+ > observability, surface and needs-external-check gates used to demote to `info`
258
+ > now arrive above info, so they open obligations. Nothing got worse in the
259
+ > paper; more of what the auditors said is now visible. Waiving a proposal you
260
+ > judge wrong is the expected disposition, and the report's per-finding columns
261
+ > (`Anchored`, `Observability`, `FP-risk`, `Surface`, `Ext-check`) are what you
262
+ > weigh. Upstream also deleted its report self-binding hashes in the same window
263
+ > — nothing here ever consumed them.
264
+
265
+ ## Codex-native note (mirror)
266
+
267
+ Upstream ships no Codex-native pack; its auditor skills are Claude-Code
268
+ contracts. A Codex-native session may run upstream's **deterministic-only
269
+ mode** (numeric core + adjudicator with an all-`review_unavailable` coverage
270
+ map — honestly scoped: it can flag, it can never say CLEAN). The full
271
+ nine-dimension sweep requires a host that can execute upstream's Claude-Code
272
+ contracts unchanged — Claude Code and the `dsh-aris` bundle on DeepSeek Harness
273
+ are the known ones. Translating upstream's
274
+ reviewer calls into `spawn_agent` on the fly is REWRITING an upstream
275
+ contract — forbidden.
276
+
277
+ ## Review tracing
278
+
279
+ Upstream saves its own per-dimension traces under the paper's
280
+ `.aris/traces/`. The launcher adds only the `.aris/forensics/` artifacts:
281
+ `gate.json` (pins `anti_ar_commit` + report/ledger hashes + the paper-text
282
+ fingerprint), `obligations.json` (the append-only ledger), and
283
+ `last_report.json` (the sha-verified archive of the folded report that
284
+ `fresh` recomputes from).