dsh-aris-panel 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (442) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +98 -0
  3. package/README_CN.md +87 -0
  4. package/dsh/checkout.patch.yml +38 -0
  5. package/dsh/client.js +634 -0
  6. package/dsh/cordis.patch.yml +44 -0
  7. package/dsh/index.mjs +76 -0
  8. package/dsh/run-status.mjs +182 -0
  9. package/dsh/scope-limits.mjs +50 -0
  10. package/dsh/workbench.mjs +291 -0
  11. package/mcp-servers/claude-review/README.md +93 -0
  12. package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
  13. package/mcp-servers/claude-review/server.py +718 -0
  14. package/mcp-servers/codex-image2/README.md +65 -0
  15. package/mcp-servers/codex-image2/server.py +893 -0
  16. package/mcp-servers/feishu-bridge/requirements.txt +1 -0
  17. package/mcp-servers/feishu-bridge/server.py +240 -0
  18. package/mcp-servers/gemini-review/README.md +171 -0
  19. package/mcp-servers/gemini-review/server.py +1856 -0
  20. package/mcp-servers/llm-chat/requirements.txt +1 -0
  21. package/mcp-servers/llm-chat/server.py +664 -0
  22. package/mcp-servers/manual-review/README.md +133 -0
  23. package/mcp-servers/manual-review/server.py +910 -0
  24. package/mcp-servers/manual-review/ui.html +279 -0
  25. package/mcp-servers/minimax-chat/requirements.txt +1 -0
  26. package/mcp-servers/minimax-chat/server.py +381 -0
  27. package/package.json +51 -0
  28. package/skills/ablation-planner/SKILL.md +123 -0
  29. package/skills/alphaxiv/SKILL.md +196 -0
  30. package/skills/analyze-results/SKILL.md +46 -0
  31. package/skills/arxiv/SKILL.md +248 -0
  32. package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
  33. package/skills/auto-review-loop/SKILL.md +1137 -0
  34. package/skills/auto-review-loop-llm/SKILL.md +259 -0
  35. package/skills/auto-review-loop-minimax/SKILL.md +302 -0
  36. package/skills/citation-audit/SKILL.md +502 -0
  37. package/skills/claims-drafting/SKILL.md +227 -0
  38. package/skills/comm-lit-review/SKILL.md +297 -0
  39. package/skills/deepxiv/SKILL.md +263 -0
  40. package/skills/dse-loop/SKILL.md +296 -0
  41. package/skills/embodiment-description/SKILL.md +129 -0
  42. package/skills/exa-search/SKILL.md +205 -0
  43. package/skills/experiment-audit/SKILL.md +311 -0
  44. package/skills/experiment-bridge/SKILL.md +376 -0
  45. package/skills/experiment-plan/SKILL.md +249 -0
  46. package/skills/experiment-queue/SKILL.md +431 -0
  47. package/skills/experiment-queue/scripts/build_manifest.py +142 -0
  48. package/skills/experiment-queue/scripts/queue_manager.py +433 -0
  49. package/skills/feishu-notify/SKILL.md +156 -0
  50. package/skills/figure-description/SKILL.md +138 -0
  51. package/skills/figure-spec/SKILL.md +262 -0
  52. package/skills/figure-spec/scripts/figure_renderer.py +799 -0
  53. package/skills/formula-derivation/SKILL.md +280 -0
  54. package/skills/gemini-search/SKILL.md +231 -0
  55. package/skills/grant-proposal/SKILL.md +698 -0
  56. package/skills/idea-creator/SKILL.md +542 -0
  57. package/skills/idea-discovery/SKILL.md +521 -0
  58. package/skills/idea-discovery-robot/SKILL.md +363 -0
  59. package/skills/integrity-forensics/SKILL.md +284 -0
  60. package/skills/interview-cheatsheet/SKILL.md +245 -0
  61. package/skills/invention-structuring/SKILL.md +188 -0
  62. package/skills/jurisdiction-format/SKILL.md +192 -0
  63. package/skills/kill-argument/SKILL.md +437 -0
  64. package/skills/mermaid-diagram/SKILL.md +419 -0
  65. package/skills/meta-apply/SKILL.md +141 -0
  66. package/skills/meta-optimize/SKILL.md +437 -0
  67. package/skills/monitor-experiment/SKILL.md +140 -0
  68. package/skills/novelty-check/SKILL.md +101 -0
  69. package/skills/openalex/SKILL.md +237 -0
  70. package/skills/overleaf-sync/SKILL.md +220 -0
  71. package/skills/paper-claim-audit/SKILL.md +348 -0
  72. package/skills/paper-compile/SKILL.md +266 -0
  73. package/skills/paper-figure/SKILL.md +312 -0
  74. package/skills/paper-illustration/SKILL.md +736 -0
  75. package/skills/paper-illustration-image2/SKILL.md +391 -0
  76. package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  77. package/skills/paper-plan/SKILL.md +386 -0
  78. package/skills/paper-poster/SKILL.md +19 -0
  79. package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
  80. package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
  81. package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
  82. package/skills/paper-poster-html/NOTICE.md +57 -0
  83. package/skills/paper-poster-html/SKILL.md +323 -0
  84. package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
  85. package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
  86. package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
  87. package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
  88. package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
  89. package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
  90. package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
  91. package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
  92. package/skills/paper-poster-html/scripts/asset_check.py +897 -0
  93. package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
  94. package/skills/paper-poster-html/scripts/poster_check.py +251 -0
  95. package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
  96. package/skills/paper-poster-html/scripts/render_preview.py +217 -0
  97. package/skills/paper-poster-html/scripts/run_gates.py +556 -0
  98. package/skills/paper-poster-html/scripts/style_check.py +1324 -0
  99. package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
  100. package/skills/paper-poster-html/templates/README.md +170 -0
  101. package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
  102. package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
  103. package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
  104. package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
  105. package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
  106. package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
  107. package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
  108. package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
  109. package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
  110. package/skills/paper-slides/SKILL.md +635 -0
  111. package/skills/paper-talk/SKILL.md +381 -0
  112. package/skills/paper-write/SKILL.md +604 -0
  113. package/skills/paper-write/templates/IEEEtran.bst +2409 -0
  114. package/skills/paper-write/templates/IEEEtran.cls +6347 -0
  115. package/skills/paper-write/templates/iclr2026.tex +84 -0
  116. package/skills/paper-write/templates/icml2025.tex +87 -0
  117. package/skills/paper-write/templates/ieee_conference.tex +89 -0
  118. package/skills/paper-write/templates/ieee_journal.tex +93 -0
  119. package/skills/paper-write/templates/math_commands.tex +48 -0
  120. package/skills/paper-write/templates/neurips2025.tex +80 -0
  121. package/skills/paper-writing/SKILL.md +916 -0
  122. package/skills/patent-novelty-check/SKILL.md +153 -0
  123. package/skills/patent-pipeline/SKILL.md +344 -0
  124. package/skills/patent-review/SKILL.md +203 -0
  125. package/skills/pixel-art/SKILL.md +137 -0
  126. package/skills/prior-art-search/SKILL.md +146 -0
  127. package/skills/proof-checker/SKILL.md +866 -0
  128. package/skills/proof-orchestrator/NOTICE.md +24 -0
  129. package/skills/proof-orchestrator/SKILL.md +254 -0
  130. package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
  131. package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
  132. package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
  133. package/skills/proof-orchestrator/references/notation-audit.md +135 -0
  134. package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  135. package/skills/proof-orchestrator/references/stress-tests.md +38 -0
  136. package/skills/proof-writer/SKILL.md +223 -0
  137. package/skills/qzcli/SKILL.md +324 -0
  138. package/skills/rebuttal/SKILL.md +376 -0
  139. package/skills/render-html/SKILL.md +316 -0
  140. package/skills/render-html/scripts/render_html.py +1006 -0
  141. package/skills/render-html/scripts/templates/academic.html +703 -0
  142. package/skills/render-html/scripts/templates/dashboard.html +333 -0
  143. package/skills/research-lit/SKILL.md +756 -0
  144. package/skills/research-pipeline/SKILL.md +384 -0
  145. package/skills/research-refine/SKILL.md +770 -0
  146. package/skills/research-refine-pipeline/SKILL.md +186 -0
  147. package/skills/research-review/SKILL.md +198 -0
  148. package/skills/research-wiki/SKILL.md +461 -0
  149. package/skills/resubmit-pipeline/SKILL.md +447 -0
  150. package/skills/result-to-claim/SKILL.md +311 -0
  151. package/skills/run-experiment/SKILL.md +313 -0
  152. package/skills/semantic-scholar/SKILL.md +236 -0
  153. package/skills/serverless-modal/SKILL.md +335 -0
  154. package/skills/shared-references/acceptance-gate.md +324 -0
  155. package/skills/shared-references/assurance-contract.md +248 -0
  156. package/skills/shared-references/capture-antipatterns.md +78 -0
  157. package/skills/shared-references/citation-discipline.md +583 -0
  158. package/skills/shared-references/compute-env-contract.md +163 -0
  159. package/skills/shared-references/effort-contract.md +183 -0
  160. package/skills/shared-references/evidence-precheck.md +65 -0
  161. package/skills/shared-references/experiment-integrity.md +49 -0
  162. package/skills/shared-references/external-cadence.md +326 -0
  163. package/skills/shared-references/fan-out-pattern.md +366 -0
  164. package/skills/shared-references/injection-hygiene.md +127 -0
  165. package/skills/shared-references/integration-contract.md +461 -0
  166. package/skills/shared-references/output-composition.md +93 -0
  167. package/skills/shared-references/output-language.md +45 -0
  168. package/skills/shared-references/output-manifest.md +49 -0
  169. package/skills/shared-references/output-versioning.md +111 -0
  170. package/skills/shared-references/patent-format-cn.md +199 -0
  171. package/skills/shared-references/patent-format-ep.md +173 -0
  172. package/skills/shared-references/patent-format-us.md +161 -0
  173. package/skills/shared-references/patent-writing-principles.md +197 -0
  174. package/skills/shared-references/prior-art-databases.md +141 -0
  175. package/skills/shared-references/resumable-runs.md +109 -0
  176. package/skills/shared-references/review-scope-limits.md +81 -0
  177. package/skills/shared-references/review-tracing.md +391 -0
  178. package/skills/shared-references/reviewer-independence.md +79 -0
  179. package/skills/shared-references/reviewer-routing.md +852 -0
  180. package/skills/shared-references/skill-governance.md +104 -0
  181. package/skills/shared-references/taste-calibration.md +85 -0
  182. package/skills/shared-references/venue-checklists.md +114 -0
  183. package/skills/shared-references/wiki-helper-resolution.md +134 -0
  184. package/skills/shared-references/writing-principles.md +525 -0
  185. package/skills/skills-codex/README.md +102 -0
  186. package/skills/skills-codex/README_CN.md +100 -0
  187. package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
  188. package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
  189. package/skills/skills-codex/analyze-results/SKILL.md +45 -0
  190. package/skills/skills-codex/arxiv/SKILL.md +210 -0
  191. package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
  192. package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
  193. package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
  194. package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
  195. package/skills/skills-codex/citation-audit/SKILL.md +504 -0
  196. package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
  197. package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
  198. package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
  199. package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
  200. package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
  201. package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
  202. package/skills/skills-codex/deepxiv/SKILL.md +142 -0
  203. package/skills/skills-codex/dse-loop/SKILL.md +285 -0
  204. package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
  205. package/skills/skills-codex/exa-search/SKILL.md +192 -0
  206. package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
  207. package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
  208. package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
  209. package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
  210. package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
  211. package/skills/skills-codex/figure-description/SKILL.md +138 -0
  212. package/skills/skills-codex/figure-spec/SKILL.md +252 -0
  213. package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
  214. package/skills/skills-codex/gemini-search/SKILL.md +205 -0
  215. package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
  216. package/skills/skills-codex/idea-creator/SKILL.md +405 -0
  217. package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
  218. package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
  219. package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
  220. package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
  221. package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
  222. package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
  223. package/skills/skills-codex/kill-argument/SKILL.md +403 -0
  224. package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
  225. package/skills/skills-codex/meta-apply/SKILL.md +154 -0
  226. package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
  227. package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
  228. package/skills/skills-codex/novelty-check/SKILL.md +89 -0
  229. package/skills/skills-codex/openalex/SKILL.md +228 -0
  230. package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
  231. package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
  232. package/skills/skills-codex/paper-compile/SKILL.md +253 -0
  233. package/skills/skills-codex/paper-figure/SKILL.md +311 -0
  234. package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
  235. package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
  236. package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  237. package/skills/skills-codex/paper-plan/SKILL.md +278 -0
  238. package/skills/skills-codex/paper-poster/SKILL.md +19 -0
  239. package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
  240. package/skills/skills-codex/paper-slides/SKILL.md +571 -0
  241. package/skills/skills-codex/paper-talk/SKILL.md +381 -0
  242. package/skills/skills-codex/paper-write/SKILL.md +411 -0
  243. package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
  244. package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
  245. package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
  246. package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
  247. package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
  248. package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
  249. package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
  250. package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
  251. package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
  252. package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
  253. package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
  254. package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
  255. package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
  256. package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
  257. package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
  258. package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
  259. package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
  260. package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
  261. package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
  262. package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
  263. package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
  264. package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
  265. package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
  266. package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
  267. package/skills/skills-codex/paper-writing/SKILL.md +731 -0
  268. package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
  269. package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
  270. package/skills/skills-codex/patent-review/SKILL.md +202 -0
  271. package/skills/skills-codex/pixel-art/SKILL.md +139 -0
  272. package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
  273. package/skills/skills-codex/proof-checker/SKILL.md +554 -0
  274. package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
  275. package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
  276. package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
  277. package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
  278. package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
  279. package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  280. package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
  281. package/skills/skills-codex/proof-writer/SKILL.md +222 -0
  282. package/skills/skills-codex/qzcli/SKILL.md +324 -0
  283. package/skills/skills-codex/rebuttal/SKILL.md +305 -0
  284. package/skills/skills-codex/render-html/SKILL.md +305 -0
  285. package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
  286. package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
  287. package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
  288. package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
  289. package/skills/skills-codex/research-lit/SKILL.md +464 -0
  290. package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
  291. package/skills/skills-codex/research-refine/SKILL.md +721 -0
  292. package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
  293. package/skills/skills-codex/research-review/SKILL.md +135 -0
  294. package/skills/skills-codex/research-wiki/SKILL.md +421 -0
  295. package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
  296. package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
  297. package/skills/skills-codex/run-experiment/SKILL.md +236 -0
  298. package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
  299. package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
  300. package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
  301. package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
  302. package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
  303. package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
  304. package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
  305. package/skills/skills-codex/shared-references/effort-contract.md +143 -0
  306. package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
  307. package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
  308. package/skills/skills-codex/shared-references/external-cadence.md +334 -0
  309. package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
  310. package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
  311. package/skills/skills-codex/shared-references/integration-contract.md +372 -0
  312. package/skills/skills-codex/shared-references/output-composition.md +98 -0
  313. package/skills/skills-codex/shared-references/output-language.md +45 -0
  314. package/skills/skills-codex/shared-references/output-manifest.md +40 -0
  315. package/skills/skills-codex/shared-references/output-versioning.md +111 -0
  316. package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
  317. package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
  318. package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
  319. package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
  320. package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
  321. package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
  322. package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
  323. package/skills/skills-codex/shared-references/review-tracing.md +144 -0
  324. package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
  325. package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
  326. package/skills/skills-codex/shared-references/skill-governance.md +119 -0
  327. package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
  328. package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
  329. package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
  330. package/skills/skills-codex/shared-references/writing-principles.md +525 -0
  331. package/skills/skills-codex/slides-polish/SKILL.md +563 -0
  332. package/skills/skills-codex/specification-writing/SKILL.md +211 -0
  333. package/skills/skills-codex/system-profile/SKILL.md +103 -0
  334. package/skills/skills-codex/training-check/SKILL.md +83 -0
  335. package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
  336. package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
  337. package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
  338. package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
  339. package/skills/skills-codex-claude-review/README.md +79 -0
  340. package/skills/skills-codex-claude-review/README_CN.md +78 -0
  341. package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
  342. package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
  343. package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
  344. package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
  345. package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
  346. package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
  347. package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
  348. package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
  349. package/skills/skills-codex-gemini-review/README.md +176 -0
  350. package/skills/skills-codex-gemini-review/README_CN.md +175 -0
  351. package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
  352. package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
  353. package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
  354. package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
  355. package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
  356. package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
  357. package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
  358. package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
  359. package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
  360. package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
  361. package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
  362. package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
  363. package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
  364. package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
  365. package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
  366. package/skills/slides-polish/SKILL.md +565 -0
  367. package/skills/specification-writing/SKILL.md +211 -0
  368. package/skills/system-profile/SKILL.md +103 -0
  369. package/skills/training-check/SKILL.md +132 -0
  370. package/skills/vast-gpu/SKILL.md +394 -0
  371. package/skills/web-debug-search/SKILL.md +334 -0
  372. package/skills/wiki-enrich/SKILL.md +257 -0
  373. package/skills/writing-systems-papers/SKILL.md +184 -0
  374. package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
  375. package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
  376. package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
  377. package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
  378. package/templates/FINDINGS_TEMPLATE.md +52 -0
  379. package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
  380. package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
  381. package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
  382. package/templates/MANIFEST_TEMPLATE.md +7 -0
  383. package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
  384. package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
  385. package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
  386. package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
  387. package/templates/README.md +57 -0
  388. package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
  389. package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
  390. package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
  391. package/templates/claude-hooks/corpus_write_guard.json +16 -0
  392. package/templates/claude-hooks/corpus_write_guard.py +85 -0
  393. package/templates/claude-hooks/meta_logging.json +74 -0
  394. package/templates/gitignore-trace.txt +3 -0
  395. package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
  396. package/tools/arxiv_fetch.py +311 -0
  397. package/tools/capture_filter.py +126 -0
  398. package/tools/check_skills_inventory.py +273 -0
  399. package/tools/convert_skills_to_llm_chat.py +282 -0
  400. package/tools/copilot_native_evidence.py +818 -0
  401. package/tools/deepxiv_fetch.py +213 -0
  402. package/tools/evidence_check.py +212 -0
  403. package/tools/exa_search.py +425 -0
  404. package/tools/experiment_queue/README.md +118 -0
  405. package/tools/experiment_queue/build_manifest.py +44 -0
  406. package/tools/experiment_queue/queue_manager.py +44 -0
  407. package/tools/extract_paper_style.py +560 -0
  408. package/tools/figure_renderer.py +69 -0
  409. package/tools/forensics_gate.py +669 -0
  410. package/tools/generate_codex_claude_review_overrides.py +299 -0
  411. package/tools/idea_discovery_gate.py +256 -0
  412. package/tools/install_aris.ps1 +1372 -0
  413. package/tools/install_aris.sh +1370 -0
  414. package/tools/install_aris_codex.sh +1023 -0
  415. package/tools/install_aris_copilot.sh +1052 -0
  416. package/tools/iteration_log.py +143 -0
  417. package/tools/lint_skills_helpers.sh +84 -0
  418. package/tools/meta_opt/check_ready.sh +80 -0
  419. package/tools/meta_opt/log_event.sh +91 -0
  420. package/tools/meta_opt/trigger_eval.py +280 -0
  421. package/tools/meta_opt/trigger_evals.sample.json +28 -0
  422. package/tools/openalex_fetch.py +326 -0
  423. package/tools/overleaf_audit.sh +104 -0
  424. package/tools/overleaf_setup.sh +150 -0
  425. package/tools/paper_illustration_image2.py +62 -0
  426. package/tools/provenance.py +294 -0
  427. package/tools/research_wiki.py +1720 -0
  428. package/tools/review_gate.py +502 -0
  429. package/tools/run_state.py +399 -0
  430. package/tools/save_trace.sh +477 -0
  431. package/tools/semantic_scholar_fetch.py +438 -0
  432. package/tools/skill-groups.tsv +116 -0
  433. package/tools/skill_picker.py +238 -0
  434. package/tools/smart_update.ps1 +521 -0
  435. package/tools/smart_update.sh +591 -0
  436. package/tools/smart_update_codex.sh +419 -0
  437. package/tools/smart_update_copilot.sh +605 -0
  438. package/tools/threat_scan.py +222 -0
  439. package/tools/verify_paper_audits.sh +487 -0
  440. package/tools/verify_papers.py +613 -0
  441. package/tools/verify_wiki_coverage.sh +176 -0
  442. package/tools/watchdog.py +485 -0
@@ -0,0 +1,205 @@
1
+ ---
2
+ name: exa-search
3
+ description: AI-powered web search via Exa with content extraction. Use when user says "exa search", "web search with content", "find similar pages", or needs broad web results beyond academic databases (arXiv, Semantic Scholar).
4
+ argument-hint: "[search-query-or-url]"
5
+ allowed-tools: Bash(*), Read, Write
6
+ ---
7
+
8
+ # Exa AI-Powered Web Search
9
+
10
+ Search query: $ARGUMENTS
11
+
12
+ ## Role & Positioning
13
+
14
+ Exa is the **broad web search** source with built-in content extraction:
15
+
16
+ | Skill | Best for |
17
+ |------|----------|
18
+ | `/arxiv` | Direct preprint search and PDF download |
19
+ | `/semantic-scholar` | Published venue papers (IEEE, ACM, Springer), citation counts |
20
+ | `/deepxiv` | Layered reading: search, brief, section map, section reads |
21
+ | `/exa-search` | Broad web search: blogs, docs, news, companies, research papers โ€” with content extraction |
22
+
23
+ Use Exa when you need results beyond academic databases, or when you want content (highlights, full text, summaries) extracted alongside search results.
24
+
25
+ ## Constants
26
+
27
+ - **EXA_FETCHER** โ€” canonical name `exa_search.py`, resolved per
28
+ [`shared-references/integration-contract.md`](../shared-references/integration-contract.md) ยง2
29
+ (Policy D1 โ€” standalone `/exa-search` has no documented fallback,
30
+ so unresolved helper terminates with an explicit error).
31
+ - **MAX_RESULTS = 10** โ€” Default number of results to return.
32
+
33
+ > Overrides (append to arguments):
34
+ > - `/exa-search "RAG pipelines" โ€” max: 5` โ€” top 5 results
35
+ > - `/exa-search "diffusion models" โ€” category: research paper` โ€” research papers only
36
+ > - `/exa-search "startup funding" โ€” category: news, start date: 2025-01-01` โ€” recent news
37
+ > - `/exa-search "transformer" โ€” content: text, max chars: 8000` โ€” full text mode
38
+ > - `/exa-search "transformer" โ€” content: summary` โ€” LLM-generated summaries
39
+ > - `/exa-search "transformer" โ€” domains: arxiv.org,huggingface.co` โ€” domain filter
40
+ > - `/exa-search "https://arxiv.org/abs/2301.07041" โ€” similar` โ€” find similar pages
41
+
42
+ ## Setup
43
+
44
+ Exa requires the `exa-py` SDK and an API key:
45
+
46
+ ```bash
47
+ pip install exa-py
48
+ ```
49
+
50
+ Set your API key:
51
+ ```bash
52
+ export EXA_API_KEY=your-key-here
53
+ ```
54
+
55
+ Get a key from [exa.ai](https://exa.ai).
56
+
57
+ ## Workflow
58
+
59
+ ### Step 1: Parse Arguments
60
+
61
+ Parse `$ARGUMENTS` for:
62
+ - **query**: The search query (required) or a URL (for `find-similar` mode)
63
+ - **similar**: If present, use `find-similar` mode instead of search
64
+ - **max**: Override MAX_RESULTS
65
+ - **category**: `research paper`, `news`, `company`, `personal site`, `financial report`, `people`
66
+ - **content**: `highlights` (default), `text`, `summary`, `none`
67
+ - **max chars**: Max characters for content extraction
68
+ - **type**: Search type โ€” `auto` (default), `neural`, `fast`, `instant`
69
+ - **domains**: Comma-separated include domains
70
+ - **exclude domains**: Comma-separated exclude domains
71
+ - **include text**: Phrase that must appear in results
72
+ - **exclude text**: Phrase to exclude from results
73
+ - **start date**: ISO 8601 date โ€” only results after this
74
+ - **end date**: ISO 8601 date โ€” only results before this
75
+ - **location**: Two-letter ISO country code
76
+
77
+ ### Step 2: Locate Script
78
+
79
+ Resolve `$EXA_FETCHER` via the canonical strict-safe chain (see
80
+ [`shared-references/integration-contract.md`](../shared-references/integration-contract.md) ยง2).
81
+ Policy D1 cascade: there is no native inline fallback for Exa
82
+ (retrieval requires the `exa-py` SDK + API key, which lives in the
83
+ fetcher), so unresolved helper means the SKILL cannot produce its
84
+ primary output โ€” fail with explicit remediation.
85
+
86
+ ```bash
87
+ cd "$(git rev-parse --show-toplevel 2>/dev/null || pwd)" || exit 1
88
+ if [ -z "${ARIS_REPO:-}" ] && [ -f .aris/installed-skills.txt ]; then
89
+ ARIS_REPO=$(awk -F'\t' '$1=="repo_root"{print $2; exit}' .aris/installed-skills.txt 2>/dev/null) || true
90
+ fi
91
+ if [ -z "${ARIS_REPO:-}" ] && [ -f "$HOME/.aris/repo" ]; then
92
+ ARIS_REPO=$(cat "$HOME/.aris/repo" 2>/dev/null) || true
93
+ fi
94
+ EXA_FETCHER=".aris/tools/exa_search.py"
95
+ [ -f "$EXA_FETCHER" ] || EXA_FETCHER="tools/exa_search.py"
96
+ [ -f "$EXA_FETCHER" ] || { [ -n "${ARIS_REPO:-}" ] && EXA_FETCHER="$ARIS_REPO/tools/exa_search.py"; }
97
+ [ -f "$EXA_FETCHER" ] || {
98
+ echo "ERROR: exa_search.py not resolved at .aris/tools/, tools/, \$ARIS_REPO/tools/, or via ~/.aris/repo." >&2
99
+ echo " Fix: rerun bash tools/install_aris.sh or smart_update.sh (refreshes ~/.aris/repo), export ARIS_REPO, or copy the helper to tools/." >&2
100
+ echo " Also ensure 'exa-py' is installed: pip install exa-py" >&2
101
+ exit 1
102
+ }
103
+ ```
104
+
105
+ ### Step 3: Execute Search
106
+
107
+ **Standard search:**
108
+ ```bash
109
+ python3 "$EXA_FETCHER" search "QUERY" --max 10 --content highlights
110
+ ```
111
+
112
+ **With filters:**
113
+ ```bash
114
+ python3 "$EXA_FETCHER" search "QUERY" --max 10 \
115
+ --category "research paper" \
116
+ --start-date 2025-01-01 \
117
+ --content text --max-chars 8000
118
+ ```
119
+
120
+ **Find similar pages:**
121
+ ```bash
122
+ python3 "$EXA_FETCHER" find-similar "URL" --max 5 --content highlights
123
+ ```
124
+
125
+ **Get content for known URLs:**
126
+ ```bash
127
+ python3 "$EXA_FETCHER" get-contents "URL1" "URL2" --content text
128
+ ```
129
+
130
+ ### Step 4: Present Results
131
+
132
+ Format results as a structured table:
133
+
134
+ ```
135
+ | # | Title | Authors | Venue/Publisher | URL | Date | Key Content |
136
+ |---|-------|---------|-----------------|-----|------|-------------|
137
+ ```
138
+
139
+ For each result:
140
+ - Show title and URL
141
+ - Show published date if available
142
+ - Show highlights, text excerpt, or summary depending on content mode
143
+ - Flag particularly relevant results
144
+ - **For `category: "research paper"` hits only** โ€” also record authors
145
+ (from Exa's `author`/`authors` fields, or fallback: parse from the
146
+ result snippet) and venue/publisher (from `publisher`, `source`, or
147
+ the domain hosting the paper). These are needed by Step 6's wiki
148
+ hook; if either is unavailable for a given hit, skip wiki ingest
149
+ for that one hit and log a note.
150
+
151
+ ### Step 5: Offer Follow-up
152
+
153
+ After presenting results, suggest:
154
+ - **Deepen**: "I can fetch full text for any of these results"
155
+ - **Find similar**: "I can find pages similar to any result"
156
+ - **Narrow**: "I can re-search with domain/date/text filters"
157
+
158
+ ### Step 6: Update Research Wiki (if active, research-paper results only)
159
+
160
+ **Required when `research-wiki/` exists AND the search returned
161
+ results of `category: "research paper"`**; skip silently otherwise.
162
+ General web results (blog posts, docs, news) are **not** ingested โ€”
163
+ the wiki is for papers only.
164
+
165
+ When the predicates hold, resolve `$WIKI_SCRIPT` per the canonical
166
+ chain at
167
+ [`shared-references/wiki-helper-resolution.md`](../shared-references/wiki-helper-resolution.md)
168
+ (Variant B โ€” warn-and-skip). For each research paper hit, try to
169
+ recover an arXiv ID from the URL (`arxiv.org/abs/<id>`); if present,
170
+ use `--arxiv-id`. Otherwise fall back to manual metadata:
171
+
172
+ ```bash
173
+ if [ -d research-wiki/ ] and query category was "research paper":
174
+ cd "$(git rev-parse --show-toplevel 2>/dev/null || pwd)" || exit 1
175
+ ARIS_REPO="${ARIS_REPO:-$(awk -F'\t' '$1=="repo_root"{print $2; exit}' .aris/installed-skills.txt 2>/dev/null)}"
176
+ if [ -z "${ARIS_REPO:-}" ] && [ -f "$HOME/.aris/repo" ]; then
177
+ ARIS_REPO=$(cat "$HOME/.aris/repo" 2>/dev/null) || true
178
+ fi
179
+ WIKI_SCRIPT=".aris/tools/research_wiki.py"
180
+ [ -f "$WIKI_SCRIPT" ] || WIKI_SCRIPT="tools/research_wiki.py"
181
+ [ -f "$WIKI_SCRIPT" ] || { [ -n "${ARIS_REPO:-}" ] && WIKI_SCRIPT="$ARIS_REPO/tools/research_wiki.py"; }
182
+ [ -f "$WIKI_SCRIPT" ] || {
183
+ echo "WARN: research_wiki.py not found; exa-search results delivered, wiki ingest skipped. Fix: bash tools/install_aris.sh or smart_update.sh (refreshes ~/.aris/repo), export ARIS_REPO, or cp <ARIS-repo>/tools/research_wiki.py tools/." >&2
184
+ WIKI_SCRIPT=""
185
+ }
186
+ [ -n "$WIKI_SCRIPT" ] && for each research-paper hit in results:
187
+ if URL matches arxiv.org/abs/<id>:
188
+ python3 "$WIKI_SCRIPT" ingest_paper research-wiki/ \
189
+ --arxiv-id "<id>"
190
+ else:
191
+ python3 "$WIKI_SCRIPT" ingest_paper research-wiki/ \
192
+ --title "<title>" --authors "<authors joined by , >" \
193
+ --year <year> --venue "<venue or publisher>"
194
+ ```
195
+
196
+ The helper handles slug / dedup / page / index / log โ€” **do not
197
+ handwrite `papers/<slug>.md`**. See
198
+ [`shared-references/integration-contract.md`](../shared-references/integration-contract.md).
199
+
200
+ ## Key Rules
201
+ - Always check that `EXA_API_KEY` is set before searching
202
+ - Default to `highlights` content mode for a good balance of speed and context
203
+ - Use `category: "research paper"` when the user is clearly looking for academic content
204
+ - Use `text` content mode when the user needs full page content
205
+ - Combine with `/arxiv` or `/semantic-scholar` for comprehensive literature coverage
@@ -0,0 +1,311 @@
1
+ ---
2
+ name: experiment-audit
3
+ description: "Audit experiment integrity before claiming results. Uses cross-model review (external reviewer backend) to check for fake ground truth, score normalization fraud, phantom results, and insufficient scope. Use when user says \"ๅฎก่ฎกๅฎž้ชŒ\", \"check experiment integrity\", \"audit results\", \"ๅฎž้ชŒ่ฏšๅฎžๅบฆ\", or after experiments complete before writing claims."
4
+ argument-hint: "[experiment-dir-or-results-path]"
5
+ allowed-tools: Bash(*), Read, Write, Edit, Grep, Glob, mcp__codex__codex, mcp__codex__codex-reply, mcp__manual_review__review, mcp__manual_review__review_reply
6
+ ---
7
+
8
+ # Experiment Audit: Cross-Model Integrity Verification
9
+
10
+ > ๐Ÿ”’ **Do not wrap this skill in `/loop`, `/schedule`, or `CronCreate`.** It is
11
+ > verdict-bearing โ€” it judges experiment integrity. Re-running that verdict on a
12
+ > timer adds no new signal, and a loop that accepts its own output to decide
13
+ > when to stop crosses into self-acquittal (`acceptance-gate.md`). Schedule the
14
+ > *external wait that precedes it* โ€” experiments done โ†’ then audit **once**. See
15
+ > [`shared-references/external-cadence.md`](../shared-references/external-cadence.md).
16
+
17
+ Audit experiment integrity for: **$ARGUMENTS**
18
+
19
+ ## Why This Exists
20
+
21
+ LLM agents can produce fraudulent experimental results through:
22
+ 1. **Fake ground truth** โ€” creating synthetic "reference" from model outputs, then reporting high agreement as performance
23
+ 2. **Score normalization** โ€” dividing metrics by the model's own max to get 0.99+
24
+ 3. **Phantom results** โ€” claiming numbers from files that don't exist or functions never called
25
+ 4. **Insufficient scope** โ€” reporting 2-scene pilots as "comprehensive evaluation"
26
+
27
+ These are NOT intentional deception โ€” they are failure modes of optimizing agents that lack integrity constraints. This skill adds that constraint.
28
+
29
+ ## Core Principle
30
+
31
+ **The executor collects file paths. The external reviewer backend reads code and judges integrity. The executor does NOT participate in integrity judgment.**
32
+
33
+ This follows `shared-references/reviewer-independence.md` and `shared-references/experiment-integrity.md`.
34
+
35
+ ## Constants
36
+
37
+ - **REVIEWER_BACKEND = `codex`** โ€” Default: Codex MCP (ultra). Override with `โ€” reviewer: oracle-pro` for Oracle MCP, or `โ€” reviewer: manual` for Manual Review MCP. If manual-review MCP is unavailable, stop and print the install command; do not fall back to Codex. See `shared-references/reviewer-routing.md`.
38
+
39
+ ## Reviewer Calling Convention
40
+
41
+ When calling the reviewer, branch on REVIEWER_BACKEND:
42
+
43
+ **If REVIEWER_BACKEND = `codex`:**
44
+ Use `mcp__codex__codex` for new review threads.
45
+ Use `mcp__codex__codex-reply` for follow-up rounds (reuse threadId).
46
+
47
+ **If REVIEWER_BACKEND = `manual`:**
48
+ Use `mcp__manual_review__review` for new review threads with:
49
+ prompt: [exact same prompt that would go to Codex]
50
+ config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}
51
+ Save the returned `threadId`.
52
+ Use `mcp__manual_review__review_reply` for follow-up rounds with:
53
+ threadId: [saved manual-review threadId]
54
+ prompt: [follow-up prompt]
55
+ config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}
56
+
57
+ Prompt fidelity: the manual prompt must be exactly the same text that Codex would receive.
58
+ Review tracing applies equally to both backends.
59
+
60
+ ## Workflow
61
+
62
+ ### Step 1: Collect Artifacts (Executor โ€” Claude)
63
+
64
+ Locate and list these files WITHOUT reading or summarizing their content:
65
+
66
+ ```
67
+ Scan project directory for:
68
+ 1. Evaluation scripts: *eval*.py, *metric*.py, *test*.py, *benchmark*.py
69
+ 2. Result files: *.json, *.csv in results/, outputs/, logs/
70
+ 3. Ground truth paths: look in eval scripts for data loading (dataset paths, GT references)
71
+ 4. Experiment tracker: EXPERIMENT_TRACKER.md, EXPERIMENT_LOG.md
72
+ 5. Paper claims: NARRATIVE_REPORT.md, paper/sections/*.tex, PAPER_PLAN.md
73
+ 6. Config files: *.yaml, *.toml, *.json configs with metric definitions
74
+ ```
75
+ A verdict-bearing manual response MUST begin with
76
+ `Reviewer-Model: <exact-model-id>` โ€” pass the model THIS session is actually
77
+ running as in `executor_model`. Missing, unknown, or same-family identity
78
+ cannot acquit; emit `REVIEW_UNAVAILABLE` rather than guessing. If the executor
79
+ model cannot be named, manual review's cross-family claim is unprovable โ€” say
80
+ so in the report instead of asserting it.
81
+
82
+
83
+ **DO NOT summarize, interpret, or explain any file content.** Only collect paths.
84
+
85
+ ### Step 2: Send to Reviewer
86
+
87
+ Based on the selected reviewer backend (see Reviewer Calling Convention), pass ONLY file paths and the audit checklist to the reviewer. The reviewer reads everything directly.
88
+
89
+ For `codex`, call `mcp__codex__codex` with:
90
+ - `model: gpt-5.6-sol`
91
+ - `config: {"model_reasoning_effort": "ultra"}`
92
+ - `sandbox: read-only`
93
+ - `cwd: [project directory]`
94
+ - `prompt: [the exact full prompt below]`
95
+
96
+ For `manual`, call `mcp__manual_review__review` with:
97
+ - `config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}`
98
+ - `prompt: [the exact full prompt below]`
99
+
100
+ Manual review cannot use Codex-only `model`, `sandbox`, or `cwd`; include the same file paths in the prompt so the user can inspect them.
101
+
102
+ Use this exact prompt for both backends:
103
+
104
+ ```
105
+ You are an experiment integrity auditor. Start from the assumption that the
106
+ evaluation is compromised somewhere โ€” your job is to find where. Be
107
+ adversarial. Trust nothing the author tells you โ€” verify everything
108
+ yourself. Read ALL files listed below and check for the following fraud
109
+ patterns.
110
+
111
+ Files to read:
112
+ - Evaluation scripts: [list paths]
113
+ - Result files: [list paths]
114
+ - Experiment tracker: [list paths]
115
+ - Paper claims: [list paths]
116
+ - Config files: [list paths]
117
+
118
+ ## Audit Checklist
119
+
120
+ ### A. Ground Truth Provenance
121
+ For each evaluation script:
122
+ 1. Where does "ground truth" / "reference" / "target" come from?
123
+ 2. Is it loaded from the DATASET, or generated/derived from MODEL OUTPUTS?
124
+ 3. If derived: is it explicitly labeled as proxy evaluation?
125
+ 4. Are official eval scripts used when available for this benchmark?
126
+ FAIL if: GT is derived from model outputs without explicit proxy labeling.
127
+
128
+ ### B. Score Normalization
129
+ For each metric computation:
130
+ 1. Is any metric divided by max/min/mean of the model's OWN output?
131
+ 2. Are raw scores reported alongside any normalized scores?
132
+ 3. Are any scores suspiciously close to 1.0 or 100%?
133
+ FAIL if: Normalization denominator comes from prediction statistics.
134
+
135
+ ### C. Result File Existence
136
+ For each claim in the paper/narrative:
137
+ 1. Does the referenced result file actually exist?
138
+ 2. Does the claimed metric key exist in that file?
139
+ 3. Does the claimed NUMBER match what's in the file?
140
+ 4. Is the experiment tracker status DONE (not TODO/IN_PROGRESS)?
141
+ FAIL if: Claimed results reference nonexistent files or mismatched numbers.
142
+
143
+ ### D. Dead Code Detection
144
+ For each metric function defined in eval scripts:
145
+ 1. Is it actually CALLED in any evaluation pipeline?
146
+ 2. Does its output appear in any result file?
147
+ WARN if: Metric functions exist but are never called.
148
+
149
+ ### E. Scope Assessment
150
+ 1. How many scenes/datasets/configurations were actually tested?
151
+ 2. How many seeds/runs per configuration?
152
+ 3. Does the paper use words like "comprehensive", "extensive", "robust"?
153
+ 4. Is the actual scope sufficient for those claims?
154
+ WARN if: Scope language exceeds actual evidence.
155
+
156
+ ### F. Evaluation Type Classification
157
+ Classify each evaluation as:
158
+ - real_gt: uses dataset-provided ground truth
159
+ - synthetic_proxy: uses model-generated reference
160
+ - self_supervised_proxy: no GT by design
161
+ - simulation_only: simulated environment
162
+ - human_eval: human judges
163
+
164
+ ## Output Format
165
+
166
+ For each check (A-F), report:
167
+ - Status: PASS | WARN | FAIL
168
+ - Evidence: exact file:line references
169
+ - Details: what specifically was found
170
+
171
+ Overall verdict: PASS | WARN | FAIL
172
+
173
+ Be thorough. Read every eval script line by line.
174
+ ```
175
+
176
+ ### Step 3: Parse and Write Report (Executor โ€” Claude)
177
+
178
+ Parse the reviewer's response and write `EXPERIMENT_AUDIT.md`:
179
+
180
+ ```markdown
181
+ # Experiment Audit Report
182
+
183
+ **Date**: [today]
184
+ **Auditor**: External reviewer backend, ultra reasoning (cross-model, read-only)
185
+ **Project**: [project name]
186
+
187
+ ## Overall Verdict: [PASS | WARN | FAIL]
188
+
189
+ ## Integrity Status: [pass | warn | fail]
190
+
191
+ ## Checks
192
+
193
+ ### A. Ground Truth Provenance: [PASS|WARN|FAIL]
194
+ [details + file:line evidence]
195
+
196
+ ### B. Score Normalization: [PASS|WARN|FAIL]
197
+ [details]
198
+
199
+ ### C. Result File Existence: [PASS|WARN|FAIL]
200
+ [details]
201
+
202
+ ### D. Dead Code Detection: [PASS|WARN|FAIL]
203
+ [details]
204
+
205
+ ### E. Scope Assessment: [PASS|WARN|FAIL]
206
+ [details]
207
+
208
+ ### F. Evaluation Type: [real_gt | synthetic_proxy | ...]
209
+ [classification + evidence]
210
+
211
+ ## Action Items
212
+ - [specific fixes if WARN or FAIL]
213
+
214
+ ## Claim Impact
215
+ - Claim 1: [supported | needs qualifier | unsupported]
216
+ - Claim 2: ...
217
+ ```
218
+
219
+ Also write `EXPERIMENT_AUDIT.json` for machine consumption:
220
+
221
+ ```json
222
+ {
223
+ "date": "2026-04-10",
224
+ "auditor": "external-reviewer-ultra",
225
+ "overall_verdict": "warn",
226
+ "integrity_status": "warn",
227
+ "checks": {
228
+ "gt_provenance": {"status": "pass", "details": "..."},
229
+ "score_normalization": {"status": "warn", "details": "..."},
230
+ "result_existence": {"status": "pass", "details": "..."},
231
+ "dead_code": {"status": "pass", "details": "..."},
232
+ "scope": {"status": "warn", "details": "..."},
233
+ "eval_type": "real_gt"
234
+ },
235
+ "claims": [
236
+ {"id": "C1", "impact": "supported"},
237
+ {"id": "C2", "impact": "needs_qualifier"}
238
+ ]
239
+ }
240
+ ```
241
+
242
+ ### Step 4: Print Summary
243
+
244
+ ```
245
+ ๐Ÿ”ฌ Experiment Audit Complete
246
+
247
+ GT Provenance: โœ… PASS โ€” real dataset GT used
248
+ Score Normalization: โš ๏ธ WARN โ€” boundary metric uses self-reference
249
+ Result Existence: โœ… PASS โ€” all files exist, numbers match
250
+ Dead Code: โœ… PASS โ€” all metric functions called
251
+ Scope: โš ๏ธ WARN โ€” 2 scenes, paper says "comprehensive"
252
+
253
+ Overall: โš ๏ธ WARN
254
+
255
+ See EXPERIMENT_AUDIT.md for details.
256
+ ```
257
+
258
+ ## Integration with Other Skills
259
+
260
+ ### Automatic in /research-pipeline (advisory, never blocks)
261
+
262
+ When integrated into the pipeline, this skill runs automatically after `/experiment-bridge` and before `/auto-review-loop`:
263
+
264
+ ```
265
+ /experiment-bridge โ†’ results ready
266
+ โ†“
267
+ /experiment-audit (automatic, advisory)
268
+ โ”œโ”€โ”€ PASS โ†’ continue normally
269
+ โ”œโ”€โ”€ WARN โ†’ print โš ๏ธ warning, continue, tag claims as [INTEGRITY: WARN]
270
+ โ””โ”€โ”€ FAIL โ†’ print ๐Ÿ”ด alert, continue, tag claims as [INTEGRITY CONCERN]
271
+ โ†“
272
+ /auto-review-loop โ†’ proceeds with integrity tags visible to reviewer
273
+ ```
274
+
275
+ **Never blocks the pipeline.** Even on FAIL, the pipeline continues โ€” but claims carry visible integrity tags.
276
+
277
+ ### Read by /result-to-claim (if exists)
278
+
279
+ ```
280
+ if EXPERIMENT_AUDIT.json exists:
281
+ read integrity_status
282
+ attach to verdict: {claim_supported: "yes", integrity_status: "warn"}
283
+ if integrity_status == "fail":
284
+ downgrade verdict display: "yes [INTEGRITY CONCERN]"
285
+ else:
286
+ verdict as normal, integrity_status = "unavailable"
287
+ mark as "provisional โ€” no integrity audit"
288
+ ```
289
+
290
+ ### Read by /paper-write (if exists)
291
+
292
+ ```
293
+ if EXPERIMENT_AUDIT.json exists AND integrity_status == "fail":
294
+ add footnote to affected claims: "Note: integrity audit flagged concerns with this evaluation"
295
+ ```
296
+
297
+ ## Key Rules
298
+
299
+ - **Reviewer independence**: executor collects paths, reviewer judges. Period.
300
+ - **Never block**: warn loudly, never halt the pipeline.
301
+ - **File-as-switch**: no EXPERIMENT_AUDIT.md = skill was never run = zero impact on existing behavior.
302
+ - **Cross-model**: the reviewer MUST be a different model family from the executor.
303
+ - **Honest about limits**: the audit catches common patterns, not all possible fraud. It is a safety net, not a guarantee.
304
+
305
+ ## Acknowledgements
306
+
307
+ Motivated by community-reported integrity issues (#57, #131) where executor agents created fake ground truth and self-normalized scores.
308
+
309
+ ## Review Tracing
310
+
311
+ After each reviewer call (`mcp__codex__codex`, `mcp__codex__codex-reply`, `mcp__manual_review__review`, or `mcp__manual_review__review_reply`), save the trace following `shared-references/review-tracing.md` (Policy C โ€” forensic; never silently skip). Use `save_trace.sh` (resolved per the chain in `shared-references/integration-contract.md` ยง2) or write files directly to `.aris/traces/<skill>/<date>_run<NN>/`. Respect the `--- trace:` parameter (default: `full`).