dsh-aris-panel 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (442) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +98 -0
  3. package/README_CN.md +87 -0
  4. package/dsh/checkout.patch.yml +38 -0
  5. package/dsh/client.js +634 -0
  6. package/dsh/cordis.patch.yml +44 -0
  7. package/dsh/index.mjs +76 -0
  8. package/dsh/run-status.mjs +182 -0
  9. package/dsh/scope-limits.mjs +50 -0
  10. package/dsh/workbench.mjs +291 -0
  11. package/mcp-servers/claude-review/README.md +93 -0
  12. package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
  13. package/mcp-servers/claude-review/server.py +718 -0
  14. package/mcp-servers/codex-image2/README.md +65 -0
  15. package/mcp-servers/codex-image2/server.py +893 -0
  16. package/mcp-servers/feishu-bridge/requirements.txt +1 -0
  17. package/mcp-servers/feishu-bridge/server.py +240 -0
  18. package/mcp-servers/gemini-review/README.md +171 -0
  19. package/mcp-servers/gemini-review/server.py +1856 -0
  20. package/mcp-servers/llm-chat/requirements.txt +1 -0
  21. package/mcp-servers/llm-chat/server.py +664 -0
  22. package/mcp-servers/manual-review/README.md +133 -0
  23. package/mcp-servers/manual-review/server.py +910 -0
  24. package/mcp-servers/manual-review/ui.html +279 -0
  25. package/mcp-servers/minimax-chat/requirements.txt +1 -0
  26. package/mcp-servers/minimax-chat/server.py +381 -0
  27. package/package.json +51 -0
  28. package/skills/ablation-planner/SKILL.md +123 -0
  29. package/skills/alphaxiv/SKILL.md +196 -0
  30. package/skills/analyze-results/SKILL.md +46 -0
  31. package/skills/arxiv/SKILL.md +248 -0
  32. package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
  33. package/skills/auto-review-loop/SKILL.md +1137 -0
  34. package/skills/auto-review-loop-llm/SKILL.md +259 -0
  35. package/skills/auto-review-loop-minimax/SKILL.md +302 -0
  36. package/skills/citation-audit/SKILL.md +502 -0
  37. package/skills/claims-drafting/SKILL.md +227 -0
  38. package/skills/comm-lit-review/SKILL.md +297 -0
  39. package/skills/deepxiv/SKILL.md +263 -0
  40. package/skills/dse-loop/SKILL.md +296 -0
  41. package/skills/embodiment-description/SKILL.md +129 -0
  42. package/skills/exa-search/SKILL.md +205 -0
  43. package/skills/experiment-audit/SKILL.md +311 -0
  44. package/skills/experiment-bridge/SKILL.md +376 -0
  45. package/skills/experiment-plan/SKILL.md +249 -0
  46. package/skills/experiment-queue/SKILL.md +431 -0
  47. package/skills/experiment-queue/scripts/build_manifest.py +142 -0
  48. package/skills/experiment-queue/scripts/queue_manager.py +433 -0
  49. package/skills/feishu-notify/SKILL.md +156 -0
  50. package/skills/figure-description/SKILL.md +138 -0
  51. package/skills/figure-spec/SKILL.md +262 -0
  52. package/skills/figure-spec/scripts/figure_renderer.py +799 -0
  53. package/skills/formula-derivation/SKILL.md +280 -0
  54. package/skills/gemini-search/SKILL.md +231 -0
  55. package/skills/grant-proposal/SKILL.md +698 -0
  56. package/skills/idea-creator/SKILL.md +542 -0
  57. package/skills/idea-discovery/SKILL.md +521 -0
  58. package/skills/idea-discovery-robot/SKILL.md +363 -0
  59. package/skills/integrity-forensics/SKILL.md +284 -0
  60. package/skills/interview-cheatsheet/SKILL.md +245 -0
  61. package/skills/invention-structuring/SKILL.md +188 -0
  62. package/skills/jurisdiction-format/SKILL.md +192 -0
  63. package/skills/kill-argument/SKILL.md +437 -0
  64. package/skills/mermaid-diagram/SKILL.md +419 -0
  65. package/skills/meta-apply/SKILL.md +141 -0
  66. package/skills/meta-optimize/SKILL.md +437 -0
  67. package/skills/monitor-experiment/SKILL.md +140 -0
  68. package/skills/novelty-check/SKILL.md +101 -0
  69. package/skills/openalex/SKILL.md +237 -0
  70. package/skills/overleaf-sync/SKILL.md +220 -0
  71. package/skills/paper-claim-audit/SKILL.md +348 -0
  72. package/skills/paper-compile/SKILL.md +266 -0
  73. package/skills/paper-figure/SKILL.md +312 -0
  74. package/skills/paper-illustration/SKILL.md +736 -0
  75. package/skills/paper-illustration-image2/SKILL.md +391 -0
  76. package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  77. package/skills/paper-plan/SKILL.md +386 -0
  78. package/skills/paper-poster/SKILL.md +19 -0
  79. package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
  80. package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
  81. package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
  82. package/skills/paper-poster-html/NOTICE.md +57 -0
  83. package/skills/paper-poster-html/SKILL.md +323 -0
  84. package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
  85. package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
  86. package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
  87. package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
  88. package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
  89. package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
  90. package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
  91. package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
  92. package/skills/paper-poster-html/scripts/asset_check.py +897 -0
  93. package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
  94. package/skills/paper-poster-html/scripts/poster_check.py +251 -0
  95. package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
  96. package/skills/paper-poster-html/scripts/render_preview.py +217 -0
  97. package/skills/paper-poster-html/scripts/run_gates.py +556 -0
  98. package/skills/paper-poster-html/scripts/style_check.py +1324 -0
  99. package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
  100. package/skills/paper-poster-html/templates/README.md +170 -0
  101. package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
  102. package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
  103. package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
  104. package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
  105. package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
  106. package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
  107. package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
  108. package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
  109. package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
  110. package/skills/paper-slides/SKILL.md +635 -0
  111. package/skills/paper-talk/SKILL.md +381 -0
  112. package/skills/paper-write/SKILL.md +604 -0
  113. package/skills/paper-write/templates/IEEEtran.bst +2409 -0
  114. package/skills/paper-write/templates/IEEEtran.cls +6347 -0
  115. package/skills/paper-write/templates/iclr2026.tex +84 -0
  116. package/skills/paper-write/templates/icml2025.tex +87 -0
  117. package/skills/paper-write/templates/ieee_conference.tex +89 -0
  118. package/skills/paper-write/templates/ieee_journal.tex +93 -0
  119. package/skills/paper-write/templates/math_commands.tex +48 -0
  120. package/skills/paper-write/templates/neurips2025.tex +80 -0
  121. package/skills/paper-writing/SKILL.md +916 -0
  122. package/skills/patent-novelty-check/SKILL.md +153 -0
  123. package/skills/patent-pipeline/SKILL.md +344 -0
  124. package/skills/patent-review/SKILL.md +203 -0
  125. package/skills/pixel-art/SKILL.md +137 -0
  126. package/skills/prior-art-search/SKILL.md +146 -0
  127. package/skills/proof-checker/SKILL.md +866 -0
  128. package/skills/proof-orchestrator/NOTICE.md +24 -0
  129. package/skills/proof-orchestrator/SKILL.md +254 -0
  130. package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
  131. package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
  132. package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
  133. package/skills/proof-orchestrator/references/notation-audit.md +135 -0
  134. package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  135. package/skills/proof-orchestrator/references/stress-tests.md +38 -0
  136. package/skills/proof-writer/SKILL.md +223 -0
  137. package/skills/qzcli/SKILL.md +324 -0
  138. package/skills/rebuttal/SKILL.md +376 -0
  139. package/skills/render-html/SKILL.md +316 -0
  140. package/skills/render-html/scripts/render_html.py +1006 -0
  141. package/skills/render-html/scripts/templates/academic.html +703 -0
  142. package/skills/render-html/scripts/templates/dashboard.html +333 -0
  143. package/skills/research-lit/SKILL.md +756 -0
  144. package/skills/research-pipeline/SKILL.md +384 -0
  145. package/skills/research-refine/SKILL.md +770 -0
  146. package/skills/research-refine-pipeline/SKILL.md +186 -0
  147. package/skills/research-review/SKILL.md +198 -0
  148. package/skills/research-wiki/SKILL.md +461 -0
  149. package/skills/resubmit-pipeline/SKILL.md +447 -0
  150. package/skills/result-to-claim/SKILL.md +311 -0
  151. package/skills/run-experiment/SKILL.md +313 -0
  152. package/skills/semantic-scholar/SKILL.md +236 -0
  153. package/skills/serverless-modal/SKILL.md +335 -0
  154. package/skills/shared-references/acceptance-gate.md +324 -0
  155. package/skills/shared-references/assurance-contract.md +248 -0
  156. package/skills/shared-references/capture-antipatterns.md +78 -0
  157. package/skills/shared-references/citation-discipline.md +583 -0
  158. package/skills/shared-references/compute-env-contract.md +163 -0
  159. package/skills/shared-references/effort-contract.md +183 -0
  160. package/skills/shared-references/evidence-precheck.md +65 -0
  161. package/skills/shared-references/experiment-integrity.md +49 -0
  162. package/skills/shared-references/external-cadence.md +326 -0
  163. package/skills/shared-references/fan-out-pattern.md +366 -0
  164. package/skills/shared-references/injection-hygiene.md +127 -0
  165. package/skills/shared-references/integration-contract.md +461 -0
  166. package/skills/shared-references/output-composition.md +93 -0
  167. package/skills/shared-references/output-language.md +45 -0
  168. package/skills/shared-references/output-manifest.md +49 -0
  169. package/skills/shared-references/output-versioning.md +111 -0
  170. package/skills/shared-references/patent-format-cn.md +199 -0
  171. package/skills/shared-references/patent-format-ep.md +173 -0
  172. package/skills/shared-references/patent-format-us.md +161 -0
  173. package/skills/shared-references/patent-writing-principles.md +197 -0
  174. package/skills/shared-references/prior-art-databases.md +141 -0
  175. package/skills/shared-references/resumable-runs.md +109 -0
  176. package/skills/shared-references/review-scope-limits.md +81 -0
  177. package/skills/shared-references/review-tracing.md +391 -0
  178. package/skills/shared-references/reviewer-independence.md +79 -0
  179. package/skills/shared-references/reviewer-routing.md +852 -0
  180. package/skills/shared-references/skill-governance.md +104 -0
  181. package/skills/shared-references/taste-calibration.md +85 -0
  182. package/skills/shared-references/venue-checklists.md +114 -0
  183. package/skills/shared-references/wiki-helper-resolution.md +134 -0
  184. package/skills/shared-references/writing-principles.md +525 -0
  185. package/skills/skills-codex/README.md +102 -0
  186. package/skills/skills-codex/README_CN.md +100 -0
  187. package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
  188. package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
  189. package/skills/skills-codex/analyze-results/SKILL.md +45 -0
  190. package/skills/skills-codex/arxiv/SKILL.md +210 -0
  191. package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
  192. package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
  193. package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
  194. package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
  195. package/skills/skills-codex/citation-audit/SKILL.md +504 -0
  196. package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
  197. package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
  198. package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
  199. package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
  200. package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
  201. package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
  202. package/skills/skills-codex/deepxiv/SKILL.md +142 -0
  203. package/skills/skills-codex/dse-loop/SKILL.md +285 -0
  204. package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
  205. package/skills/skills-codex/exa-search/SKILL.md +192 -0
  206. package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
  207. package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
  208. package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
  209. package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
  210. package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
  211. package/skills/skills-codex/figure-description/SKILL.md +138 -0
  212. package/skills/skills-codex/figure-spec/SKILL.md +252 -0
  213. package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
  214. package/skills/skills-codex/gemini-search/SKILL.md +205 -0
  215. package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
  216. package/skills/skills-codex/idea-creator/SKILL.md +405 -0
  217. package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
  218. package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
  219. package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
  220. package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
  221. package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
  222. package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
  223. package/skills/skills-codex/kill-argument/SKILL.md +403 -0
  224. package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
  225. package/skills/skills-codex/meta-apply/SKILL.md +154 -0
  226. package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
  227. package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
  228. package/skills/skills-codex/novelty-check/SKILL.md +89 -0
  229. package/skills/skills-codex/openalex/SKILL.md +228 -0
  230. package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
  231. package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
  232. package/skills/skills-codex/paper-compile/SKILL.md +253 -0
  233. package/skills/skills-codex/paper-figure/SKILL.md +311 -0
  234. package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
  235. package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
  236. package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
  237. package/skills/skills-codex/paper-plan/SKILL.md +278 -0
  238. package/skills/skills-codex/paper-poster/SKILL.md +19 -0
  239. package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
  240. package/skills/skills-codex/paper-slides/SKILL.md +571 -0
  241. package/skills/skills-codex/paper-talk/SKILL.md +381 -0
  242. package/skills/skills-codex/paper-write/SKILL.md +411 -0
  243. package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
  244. package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
  245. package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
  246. package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
  247. package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
  248. package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
  249. package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
  250. package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
  251. package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
  252. package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
  253. package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
  254. package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
  255. package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
  256. package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
  257. package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
  258. package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
  259. package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
  260. package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
  261. package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
  262. package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
  263. package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
  264. package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
  265. package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
  266. package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
  267. package/skills/skills-codex/paper-writing/SKILL.md +731 -0
  268. package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
  269. package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
  270. package/skills/skills-codex/patent-review/SKILL.md +202 -0
  271. package/skills/skills-codex/pixel-art/SKILL.md +139 -0
  272. package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
  273. package/skills/skills-codex/proof-checker/SKILL.md +554 -0
  274. package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
  275. package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
  276. package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
  277. package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
  278. package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
  279. package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
  280. package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
  281. package/skills/skills-codex/proof-writer/SKILL.md +222 -0
  282. package/skills/skills-codex/qzcli/SKILL.md +324 -0
  283. package/skills/skills-codex/rebuttal/SKILL.md +305 -0
  284. package/skills/skills-codex/render-html/SKILL.md +305 -0
  285. package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
  286. package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
  287. package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
  288. package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
  289. package/skills/skills-codex/research-lit/SKILL.md +464 -0
  290. package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
  291. package/skills/skills-codex/research-refine/SKILL.md +721 -0
  292. package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
  293. package/skills/skills-codex/research-review/SKILL.md +135 -0
  294. package/skills/skills-codex/research-wiki/SKILL.md +421 -0
  295. package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
  296. package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
  297. package/skills/skills-codex/run-experiment/SKILL.md +236 -0
  298. package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
  299. package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
  300. package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
  301. package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
  302. package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
  303. package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
  304. package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
  305. package/skills/skills-codex/shared-references/effort-contract.md +143 -0
  306. package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
  307. package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
  308. package/skills/skills-codex/shared-references/external-cadence.md +334 -0
  309. package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
  310. package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
  311. package/skills/skills-codex/shared-references/integration-contract.md +372 -0
  312. package/skills/skills-codex/shared-references/output-composition.md +98 -0
  313. package/skills/skills-codex/shared-references/output-language.md +45 -0
  314. package/skills/skills-codex/shared-references/output-manifest.md +40 -0
  315. package/skills/skills-codex/shared-references/output-versioning.md +111 -0
  316. package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
  317. package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
  318. package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
  319. package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
  320. package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
  321. package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
  322. package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
  323. package/skills/skills-codex/shared-references/review-tracing.md +144 -0
  324. package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
  325. package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
  326. package/skills/skills-codex/shared-references/skill-governance.md +119 -0
  327. package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
  328. package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
  329. package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
  330. package/skills/skills-codex/shared-references/writing-principles.md +525 -0
  331. package/skills/skills-codex/slides-polish/SKILL.md +563 -0
  332. package/skills/skills-codex/specification-writing/SKILL.md +211 -0
  333. package/skills/skills-codex/system-profile/SKILL.md +103 -0
  334. package/skills/skills-codex/training-check/SKILL.md +83 -0
  335. package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
  336. package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
  337. package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
  338. package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
  339. package/skills/skills-codex-claude-review/README.md +79 -0
  340. package/skills/skills-codex-claude-review/README_CN.md +78 -0
  341. package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
  342. package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
  343. package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
  344. package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
  345. package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
  346. package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
  347. package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
  348. package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
  349. package/skills/skills-codex-gemini-review/README.md +176 -0
  350. package/skills/skills-codex-gemini-review/README_CN.md +175 -0
  351. package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
  352. package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
  353. package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
  354. package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
  355. package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
  356. package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
  357. package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
  358. package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
  359. package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
  360. package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
  361. package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
  362. package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
  363. package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
  364. package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
  365. package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
  366. package/skills/slides-polish/SKILL.md +565 -0
  367. package/skills/specification-writing/SKILL.md +211 -0
  368. package/skills/system-profile/SKILL.md +103 -0
  369. package/skills/training-check/SKILL.md +132 -0
  370. package/skills/vast-gpu/SKILL.md +394 -0
  371. package/skills/web-debug-search/SKILL.md +334 -0
  372. package/skills/wiki-enrich/SKILL.md +257 -0
  373. package/skills/writing-systems-papers/SKILL.md +184 -0
  374. package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
  375. package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
  376. package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
  377. package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
  378. package/templates/FINDINGS_TEMPLATE.md +52 -0
  379. package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
  380. package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
  381. package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
  382. package/templates/MANIFEST_TEMPLATE.md +7 -0
  383. package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
  384. package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
  385. package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
  386. package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
  387. package/templates/README.md +57 -0
  388. package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
  389. package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
  390. package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
  391. package/templates/claude-hooks/corpus_write_guard.json +16 -0
  392. package/templates/claude-hooks/corpus_write_guard.py +85 -0
  393. package/templates/claude-hooks/meta_logging.json +74 -0
  394. package/templates/gitignore-trace.txt +3 -0
  395. package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
  396. package/tools/arxiv_fetch.py +311 -0
  397. package/tools/capture_filter.py +126 -0
  398. package/tools/check_skills_inventory.py +273 -0
  399. package/tools/convert_skills_to_llm_chat.py +282 -0
  400. package/tools/copilot_native_evidence.py +818 -0
  401. package/tools/deepxiv_fetch.py +213 -0
  402. package/tools/evidence_check.py +212 -0
  403. package/tools/exa_search.py +425 -0
  404. package/tools/experiment_queue/README.md +118 -0
  405. package/tools/experiment_queue/build_manifest.py +44 -0
  406. package/tools/experiment_queue/queue_manager.py +44 -0
  407. package/tools/extract_paper_style.py +560 -0
  408. package/tools/figure_renderer.py +69 -0
  409. package/tools/forensics_gate.py +669 -0
  410. package/tools/generate_codex_claude_review_overrides.py +299 -0
  411. package/tools/idea_discovery_gate.py +256 -0
  412. package/tools/install_aris.ps1 +1372 -0
  413. package/tools/install_aris.sh +1370 -0
  414. package/tools/install_aris_codex.sh +1023 -0
  415. package/tools/install_aris_copilot.sh +1052 -0
  416. package/tools/iteration_log.py +143 -0
  417. package/tools/lint_skills_helpers.sh +84 -0
  418. package/tools/meta_opt/check_ready.sh +80 -0
  419. package/tools/meta_opt/log_event.sh +91 -0
  420. package/tools/meta_opt/trigger_eval.py +280 -0
  421. package/tools/meta_opt/trigger_evals.sample.json +28 -0
  422. package/tools/openalex_fetch.py +326 -0
  423. package/tools/overleaf_audit.sh +104 -0
  424. package/tools/overleaf_setup.sh +150 -0
  425. package/tools/paper_illustration_image2.py +62 -0
  426. package/tools/provenance.py +294 -0
  427. package/tools/research_wiki.py +1720 -0
  428. package/tools/review_gate.py +502 -0
  429. package/tools/run_state.py +399 -0
  430. package/tools/save_trace.sh +477 -0
  431. package/tools/semantic_scholar_fetch.py +438 -0
  432. package/tools/skill-groups.tsv +116 -0
  433. package/tools/skill_picker.py +238 -0
  434. package/tools/smart_update.ps1 +521 -0
  435. package/tools/smart_update.sh +591 -0
  436. package/tools/smart_update_codex.sh +419 -0
  437. package/tools/smart_update_copilot.sh +605 -0
  438. package/tools/threat_scan.py +222 -0
  439. package/tools/verify_paper_audits.sh +487 -0
  440. package/tools/verify_papers.py +613 -0
  441. package/tools/verify_wiki_coverage.sh +176 -0
  442. package/tools/watchdog.py +485 -0
@@ -0,0 +1,1137 @@
1
+ ---
2
+ name: auto-review-loop
3
+ description: Autonomous multi-round research review loop. In Copilot CLI it defaults to the native complementary rubber-duck subagent with host-event model evidence; elsewhere it uses Codex, while explicit external reviewer overrides remain available. Implements fixes and re-reviews until a policy-approved positive assessment or max rounds is reached.
4
+ argument-hint: "[topic-or-scope]"
5
+ allowed-tools: Bash(*), Read, Grep, Glob, Write, Edit, Skill, Task, mcp__codex__codex, mcp__codex__codex-reply, mcp__manual_review__review, mcp__manual_review__review_reply
6
+ ---
7
+
8
+ # Auto Review Loop: Autonomous Research Improvement
9
+
10
+ > 🔒 **Do not wrap this skill in `/loop`, `/schedule`, or `CronCreate`.** It
11
+ > already loops internally (review → fix → re-review) and the reviewer carries
12
+ > round-to-round memory in one `threadId` (`codex-reply`). An external timer
13
+ > re-enters from the top each tick — fresh `threadId`, reviewer memory reset —
14
+ > firing the verdict on wall-clock time instead of on artifact change: zero new
15
+ > signal, full token cost. If you want to schedule something, schedule the
16
+ > *external wait that precedes it* (experiments done → then run this once). See
17
+ > [`shared-references/external-cadence.md`](../shared-references/external-cadence.md).
18
+
19
+ Autonomously iterate: review → implement fixes → re-review, until an independent reviewer gives a policy-approved positive assessment or MAX_ROUNDS is reached.
20
+
21
+ ## Context: $ARGUMENTS
22
+
23
+ ## Constants
24
+
25
+ - MAX_ROUNDS = 4
26
+ - POSITIVE_THRESHOLD: score >= 6/10 **AND** verdict ∈ {"ready", "almost"} — **both** must hold. This matches the operative Phase-E STOP CONDITION exactly; the verdict vocabulary is {"ready", "almost", "not ready"} (a high score with a "not ready" verdict does NOT stop the loop). Earlier wording here used `or` and a stale verdict set ("accept"/"sufficient"/"ready for submission") — that was an internal inconsistency; the `AND` form is authoritative.
27
+ - REVIEW_DOC: `review-stage/AUTO_REVIEW.md` (cumulative log) *(fall back to `./AUTO_REVIEW.md` for legacy projects)*
28
+ - REVIEWER_MODEL = `gpt-5.6-sol` — Default model for the Codex backend. Must be an OpenAI model (e.g., `gpt-5.6-sol`, `o3`, `gpt-4o`). Manual backend uses a model the user chooses — it must be a recognized model from a different family (OpenAI, Anthropic, Google, DeepSeek, Moonshot/Kimi, Qwen).
29
+ - **REVIEWER_BACKEND** — With no reviewer directive, start as `auto`; Step -1 runs exactly one two-call native marker/challenge probe for the first review. A bound Copilot CLI root session uses `copilot-native` (built-in complementary `rubber-duck` subagent); an unbound/non-Copilot host keeps the existing `codex` default. Explicit `— reviewer: codex`, `oracle-pro`, `agy`, or `manual` bypasses the probe and selects that external backend. Explicit `— reviewer: copilot` retains the compatibility `copilot --agent` drive mode and its later Codex/manual finalizer. The native path gets both actual model IDs from host session events; it never needs `COPILOT_CLI` or caller-provided `--executor-model`. See `shared-references/reviewer-routing.md`.
30
+ - **OUTPUT_DIR = `review-stage/`** — All review-stage outputs go here. Create the directory if it doesn't exist.
31
+ - **HUMAN_CHECKPOINT = false** — When `true`, pause after each round's review (Phase B) and present the score + weaknesses to the user. Wait for user input before proceeding to Phase C. The user can: approve the suggested fixes, provide custom modification instructions, skip specific fixes, or stop the loop early. When `false` (default), the loop runs fully autonomously.
32
+ - **COMPACT = false** — When `true`, (1) read `EXPERIMENT_LOG.md` and `findings.md` instead of parsing full logs on session recovery, (2) append key findings to `findings.md` after each round.
33
+ - **REVIEWER_DIFFICULTY = medium** — Controls how adversarial the reviewer is. Three levels:
34
+ - `medium` (default): Current behavior — MCP-based review, the executor controls what context the reviewer sees.
35
+ - `hard`: Adds **Reviewer Memory** (the reviewer tracks its own suspicions across rounds) + **Debate Protocol** (the executor can rebut, the reviewer rules).
36
+ - `nightmare`: Everything in `hard` + **Codex exec reviewer reads the repo directly** via `codex exec` (the executor cannot filter what the reviewer sees) + **Adversarial Verification** (the reviewer independently checks if code matches claims).
37
+ - **RENDER_HTML = true** — When `true` (default), auto-render `review-stage/AUTO_REVIEW.md` to HTML on loop termination via `/render-html`. Uses `--no-review` (the loop itself IS the cross-model review; the HTML is a structural conversion). Set `false` to skip, or pass `— render html: false`.
38
+
39
+ > ⚠️ **Nightmare + Manual incompatibility**: If `REVIEWER_BACKEND = manual` and `REVIEWER_DIFFICULTY = nightmare`, STOP with:
40
+ > "difficulty: nightmare requires Codex CLI / codex exec and is not compatible with --reviewer: manual. Use difficulty: hard, or switch reviewer to codex."
41
+
42
+ > 💡 Override: `/auto-review-loop "topic" — compact: true, human checkpoint: true, difficulty: hard`
43
+
44
+ ## Reviewer Calling Convention
45
+
46
+ When calling the reviewer, branch on REVIEWER_BACKEND:
47
+
48
+ **If no `--reviewer:` directive was supplied:**
49
+ Set REVIEWER_BACKEND to `auto`. At Step -1 of the first round, resolve
50
+ `copilot_native_evidence.py` using the canonical four-layer helper chain.
51
+ Generate a fresh binding `<run_id>_r<round>_review_<8-random-hex>` and invoke
52
+ `marker`, wait, then invoke `challenge` as **two distinct root Bash calls**.
53
+ Put the literal binding and concrete resolved helper path in both calls;
54
+ Copilot Bash calls do not share variables. If the challenge binds, set
55
+ REVIEWER_BACKEND to `copilot-native` and use that same challenge for the
56
+ first review. Do not issue a second activation challenge in Phase A. If it
57
+ exits 3 because no current Copilot root session is bound, use `codex`.
58
+ Explicit reviewer directives bypass this probe. If the helper is missing,
59
+ native acceptance is unavailable; use Codex only if that external backend
60
+ is positively available, otherwise emit `REVIEW_UNAVAILABLE`.
61
+
62
+ **If REVIEWER_BACKEND = `copilot-native`:**
63
+ Read the challenge nonce and host-reported executor model. Invoke the host's
64
+ native `task` tool with `agent_type: rubber-duck`; do not start a subprocess
65
+ and do not specify a reviewer model. The prompt contains the exact standalone
66
+ `ARIS_REVIEW_NONCE=<nonce>` line, artifact/diff paths, the output contract,
67
+ and (round 2+) `review-stage/REVIEWER_MEMORY.md`. It contains no executor
68
+ summary or fix narrative. After the task completes, invoke
69
+ `copilot_native_evidence.py verify` to create the evidence and raw-response
70
+ artifacts. The verifier must observe one successful linked rubber-duck
71
+ lifecycle and known, different host-reported model families.
72
+
73
+ Pass the evidence to both `review_gate.py --native-evidence` and
74
+ `save_trace.sh --backend copilot-native --native-evidence`. A qualifying
75
+ native positive may stop directly; no external finalizer is needed. A native
76
+ negative continues with a fresh marker/challenge/subagent next round. Every
77
+ verdict-bearing native call—including a hard-mode rebuttal ruling—gets one
78
+ unique `<run_id, round, purpose>` artifact set and exactly one challenge.
79
+ Missing, same/unknown-family, malformed, stale, or mismatched evidence is
80
+ never a verdict. If native complementary dispatch is unavailable, fall back
81
+ only to a positively available opposite-family backend: Anthropic/Google
82
+ executor → Codex; OpenAI executor → manual with a reported non-OpenAI model.
83
+ Otherwise emit `REVIEW_UNAVAILABLE`. Full protocol:
84
+ `shared-references/reviewer-routing.md`.
85
+
86
+ **If REVIEWER_BACKEND = `copilot`:**
87
+ **Require `--executor-model`:** if not provided → emit `REVIEW_UNAVAILABLE`.
88
+ **Determine executor family** from `--executor-model` (see reviewer-routing.md).
89
+ **Router picks opposite-family profile:**
90
+ - executor_family=openai → profile="aris-reviewer-claude" (anthropic)
91
+ - executor_family=anthropic → profile="aris-reviewer-openai" (openai)
92
+ - executor_family=google → profile="aris-reviewer-openai" (openai, default cross)
93
+ - executor_family=unknown → `REVIEW_UNAVAILABLE` (fail closed).
94
+ **Verify the profile file** exists at `.github/agents/<profile>.agent.md`.
95
+ If missing → `REVIEW_UNAVAILABLE`.
96
+ **Read its `model:` field** into `REVIEWER_MODEL`, derive `reviewer_family`
97
+ from that model string, and verify it differs from `executor_family`. Pass
98
+ the same value through subprocess `--model`; never trust a caller-supplied
99
+ family label or profile-only pinning under an Auto session.
100
+ **Identity assurance:** `--executor-model` is caller-declared routing input,
101
+ not runtime attestation. Record `executor_model_source: caller-declared`, the
102
+ derived `family_relation`, and `independence_verified: unverified`. A pair of
103
+ different model strings must never be promoted to independently verified.
104
+ **Capability gate:** `copilot --help` must advertise `--model`, `--effort`,
105
+ and `--allow-tool`; otherwise emit `REVIEW_UNAVAILABLE`.
106
+ **Use the `copilot --agent` subprocess** (documented Copilot CLI form)
107
+ with the selected profile, `--model "$REVIEWER_MODEL"`, `--effort xhigh`,
108
+ and `--allow-tool=read` for each review call.
109
+ **Multi-round:** each round is a fresh `copilot --agent` call with the same
110
+ profile; reviewer memory is carried via `review-stage/REVIEWER_MEMORY.md` artifact.
111
+ If `copilot` CLI is unavailable → `REVIEW_UNAVAILABLE` for that drive round;
112
+ do not silently substitute another transport. A later positive Copilot
113
+ verdict still requires the separately documented Codex/manual finalizer.
114
+ See `shared-references/reviewer-routing.md` for the full copilot contract.
115
+
116
+ **If REVIEWER_BACKEND = `codex`:**
117
+ Use `mcp__codex__codex` for new review threads.
118
+ Use `mcp__codex__codex-reply` for follow-up rounds (reuse threadId).
119
+
120
+ **If REVIEWER_BACKEND = `manual`:**
121
+ Use `mcp__manual_review__review` for new review threads with:
122
+ prompt: [exact same prompt that would go to Codex]
123
+ config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}
124
+ Save the returned `threadId`.
125
+ Use `mcp__manual_review__review_reply` for follow-up rounds with:
126
+ threadId: [saved manual-review threadId]
127
+ prompt: [follow-up prompt]
128
+ config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}
129
+ A verdict-bearing manual response MUST begin with
130
+ `Reviewer-Model: <exact-model-id>`. Derive `reviewer_family` from that model
131
+ identity. Missing, unknown, or same-family identity cannot acquit; for a
132
+ mandatory escalation, emit `REVIEW_UNAVAILABLE` rather than guessing.
133
+
134
+ Prompt fidelity: the manual review task must be exactly the same text that Codex would receive; the transport may add only the required `Reviewer-Model:` response-format instruction.
135
+ Review tracing applies to every backend. Native traces are populated from the
136
+ revalidated host-event artifact rather than caller model declarations.
137
+
138
+ ## State Persistence (Compact Recovery)
139
+
140
+ Long-running loops may hit the context window limit, triggering automatic compaction. To survive this, persist state to `review-stage/REVIEW_STATE.json` after each round:
141
+
142
+ ```json
143
+ {
144
+ "run_id": "run_20260713_a1b2c3d4",
145
+ "round": 2,
146
+ "threadId": null,
147
+ "reviewer_profile": "rubber-duck",
148
+ "reviewer_backend": "copilot-native",
149
+ "executor_model": "claude-sonnet-4.6",
150
+ "executor_model_source": "host-session-event",
151
+ "executor_family": "anthropic",
152
+ "requested_reviewer_model": null,
153
+ "reported_reviewer_model": "gpt-5.5",
154
+ "reviewer_model_source": "host-session-event",
155
+ "reviewer_family": "openai",
156
+ "family_relation": "different",
157
+ "identity_assurance": "host_event_verified",
158
+ "independence_verified": true,
159
+ "native_evidence_id": "cne_0123456789abcdef0123456789abcdef",
160
+ "native_evidence_path": "review-stage/COPILOT_NATIVE_run_20260713_a1b2c3d4_ROUND_2_REVIEW.evidence.json",
161
+ "requires_external_acquittal": false,
162
+ "status": "in_progress",
163
+ "difficulty": "medium",
164
+ "last_score": 5.0,
165
+ "last_verdict": "not ready",
166
+ "pending_experiments": ["screen_name_1"],
167
+ "timestamp": "2026-03-13T21:00:00"
168
+ }
169
+ ```
170
+
171
+ - **`run_id`** — Globally unique per invocation. Generated on fresh start as `run_<YYYYMMDD>_<8-char-hex>` (e.g., `run_20260713_a1b2c3d4`). Preserved across round writes. On resume, read from state file unchanged. This binds all round state, reviewer-memory appends, and acquittal receipts to one run so a stale completed state from a previous invocation cannot leak into the current run's acquittal check.
172
+
173
+ When REVIEWER_BACKEND = `copilot-native`, save the evidence ID/path and the
174
+ host-event executor/reviewer models, derived families, and sources. Each round
175
+ is a fresh rubber-duck subagent and therefore gets a fresh evidence artifact;
176
+ there is no persistent child handle. When REVIEWER_BACKEND = compatibility
177
+ `copilot`, retain `reviewer_profile`, requested model, caller-declared executor
178
+ model, `independence_verified: "unverified"`, and the external-finalizer
179
+ obligation. For `codex` save its MCP `threadId`; for `manual` save `threadId`
180
+ and the reported reviewer identity. On resume, use `reviewer_backend` to select
181
+ the continuation mechanism and preserve `requires_external_acquittal`.
182
+
183
+ **Write this file at the end of every Phase E** (after documenting the round). Overwrite each time — only the latest round's state matters. The `run_id` field MUST persist unchanged across overwrites within the same run.
184
+
185
+ **On completion** (positive assessment or max rounds), set `"status": "completed"` so future invocations don't accidentally resume a finished loop.
186
+
187
+ ### Append-Only External-Finalizer Receipt
188
+
189
+ Whenever a Copilot path hands the verdict to an external backend—after a
190
+ positive compatibility-drive review or after a pre-verdict native dispatch
191
+ failure—maintain an **append-only** finalizer log at
192
+ `review-stage/ACQUITTAL_LOG.jsonl`. Each line records the Codex/manual reviewer
193
+ that completed that run. A successful native rubber-duck round never needs or
194
+ writes this receipt; its evidence sidecar is the acceptance record. The
195
+ historical filename is retained for compatibility:
196
+
197
+ ```jsonl
198
+ {"run_id":"run_20260713_a1b2c3d4","round":3,"backend":"codex","effort":"xhigh","verdict":"ready","score":7.5,"executor_model":"claude-sonnet-4-5","executor_model_source":"caller-declared","executor_family":"anthropic","reviewer_model":"gpt-5.6-sol","reviewer_model_source":"requested","reviewer_family":"openai","family_relation":"different","identity_assurance":"caller_declared","independence_verified":"unverified","trace_id":"auto-review-loop/2026-07-13_run03","timestamp":"2026-07-13T14:22:00Z"}
199
+ ```
200
+
201
+ **Rules (non-negotiable):**
202
+
203
+ | Rule | Detail |
204
+ |------|--------|
205
+ | **Append-only** | Never delete, never truncate, never overwrite lines. Only `>>`. |
206
+ | **Who writes** | Only a `codex` or `manual` round at `xhigh` effort when `round_requires_external_acquittal` was `true`. A Copilot review/dispatch never writes a finalizer line itself. |
207
+ | **When to write** | At the end of Phase E, after the policy-approved finalizer returns score >= 6 AND verdict ∈ {"ready", "almost"}. A normal default-Codex run does not need this sidecar. |
208
+ | **`run_id` binding** | Every line carries the current `run_id` and round so the Copilot → finalizer transition is auditable. |
209
+ | **Trace linkage** | `trace_id` MUST reference the real trace artifact in `.aris/traces/`; source and family fields in the receipt must exactly match that trace. |
210
+ | **Identity honesty** | Re-derive `family_relation` from the model strings, but preserve their sources. With the current caller-declared executor identity, write `identity_assurance: "caller_declared"` and `independence_verified: "unverified"`; never promote different strings to independent attestation. |
211
+ | **No overwrite** | `REVIEW_STATE.json` is overwritten each round (only latest state). `ACQUITTAL_LOG.jsonl` is NEVER overwritten — it is the permanent, cumulative record. |
212
+
213
+ **Why this exists:** `REVIEW_STATE.json` is overwritten each round. The log
214
+ preserves evidence that a compatibility drive verdict or failed native attempt
215
+ did not terminate by itself. A successful `copilot-native` verdict instead
216
+ uses its host-event evidence sidecar.
217
+
218
+ ## Output Protocols
219
+
220
+ > Follow these shared protocols for all output files:
221
+ > - **[Output Versioning Protocol](../shared-references/output-versioning.md)** — write timestamped file first, then copy to fixed name
222
+ > - **[Output Manifest Protocol](../shared-references/output-manifest.md)** — log every output to MANIFEST.md
223
+ > - **[Output Language Protocol](../shared-references/output-language.md)** — respect the project's language setting
224
+
225
+ ## Workflow
226
+
227
+ ### Initialization
228
+
229
+ 1. **Check for `review-stage/REVIEW_STATE.json`** *(fall back to `./REVIEW_STATE.json` if not found — legacy path)*:
230
+ - If neither path exists: **fresh start** (normal case, identical to behavior before this feature existed)
231
+ - **Generate `run_id`**: `run_<YYYYMMDD>_<8-char-hex>` (e.g., `run_20260713_a1b2c3d4`). Use `date +%Y%m%d` and 8 random hex characters. This run_id persists across all round writes and binds acquittal receipts to this invocation.
232
+ - If it exists AND `status` is `"completed"`: **fresh start** (previous loop finished normally — but its `ACQUITTAL_LOG.jsonl` entries are retained as an audit trail with their own `run_id`, and are NOT valid for the current run's stop gate)
233
+ - **Generate a new `run_id`** for this invocation.
234
+ - If it exists AND `status` is `"in_progress"` AND `timestamp` is older than 24 hours: **fresh start** (stale state from a killed/abandoned run — delete the file and start over)
235
+ - **Generate a new `run_id`** for this invocation.
236
+ - If it exists AND `status` is `"in_progress"` AND `timestamp` is within 24 hours: **resume**
237
+ - Read the state file to recover `run_id`, `round`, `threadId` (or evidence/profile fields for Copilot backends), `reviewer_backend`, `last_score`, `pending_experiments`
238
+ - **Legacy backward compat**: if `reviewer_backend` is absent from the state file, default to `codex` (pre-copilot-era states did not record this field). If `requires_external_acquittal` is absent, default it to `false`; a legacy default-Codex run must not inherit the stricter Copilot-finalizer state. If `run_id` is absent from the state file (pre-run_id era), generate a new `run_id` and log: "No run_id in legacy state file; assigned run_<...> for this resume."
239
+ - Read `review-stage/AUTO_REVIEW.md` to restore full context of prior rounds *(fall back to `./AUTO_REVIEW.md`)*
240
+ - If `pending_experiments` is non-empty, check if they have completed (e.g., check screen sessions)
241
+ - Resume from the next round (round = saved round + 1)
242
+ - Use `reviewer_backend` to determine continuation: `codex-reply` for codex; a fresh marker/challenge/rubber-duck/evidence cycle for `copilot-native`; a fresh `copilot --agent` subprocess with the saved profile/model for compatibility `copilot`; `manual_review_reply` for manual
243
+ - Log: "Recovered from context compaction. Resuming at Round N."
244
+ 2. Read project narrative documents, memory files, and any prior review documents. **When `COMPACT = true` and compact files exist**: read `findings.md` + `EXPERIMENT_LOG.md` instead of full `review-stage/AUTO_REVIEW.md` and raw logs — saves context window.
245
+ 3. Read recent experiment results (check output directories, logs)
246
+ 4. Identify current weaknesses and open TODOs from prior reviews
247
+ 5. Initialize round counter = 1 (unless recovered from state file)
248
+ 6. Create/update `review-stage/AUTO_REVIEW.md` with header and timestamp
249
+ 7. If this is a fresh run with no explicit reviewer directive, initialize
250
+ REVIEWER_BACKEND to `auto`. Step -1 of Round 1 performs activation and uses
251
+ that same challenge for the review. Explicit reviewer directives initialize
252
+ their selected backend and bypass activation. Do not use environment
253
+ heuristics.
254
+
255
+ ### Loop (repeat up to MAX_ROUNDS)
256
+
257
+ **Step -1 — Resolve the automatic backend and prepare one native challenge:**
258
+
259
+ - If REVIEWER_BACKEND is `auto`, resolve the native helper and run one root
260
+ `marker` call followed by one root `challenge` call. Use binding
261
+ `<run_id>_r<round>_review_<8-random-hex>` and output
262
+ `review-stage/COPILOT_NATIVE_<run_id>_ROUND_<round>_REVIEW.challenge.json`.
263
+ A bound challenge sets REVIEWER_BACKEND to `copilot-native` and
264
+ NATIVE_CHALLENGE to that path. Exit 3/unbound sets REVIEWER_BACKEND to
265
+ `codex`. Any other failure follows the fail-closed capability rules.
266
+ - If REVIEWER_BACKEND is already `copilot-native` (a later round or a resumed
267
+ run), create one fresh marker/challenge pair with the same run-scoped naming
268
+ pattern and set NATIVE_CHALLENGE. An unbound or invalid challenge cannot be
269
+ treated as a verdict or silently relabeled.
270
+ - Explicit external or compatibility backends do nothing in this step.
271
+
272
+ The challenge created here is the challenge consumed by Phase A. **Do not run
273
+ another marker/challenge for the same review.** Run-scoped filenames are
274
+ append-only audit identities; never pass `--replace` to reuse evidence from an
275
+ older invocation.
276
+
277
+ **Step 0 — Snapshot current-round state:** After Step -1 resolves `auto`, set `round_backend = <current REVIEWER_BACKEND>` and `round_requires_external_acquittal = <current requires_external_acquittal, default false>`. These variables label the backend and obligation that actually governed the CURRENT round. If compatibility-drive escalation occurs later in Phase B.5.1 (`copilot` → codex/manual), the snapshots retain their pre-escalation values while the forward-looking state is updated for the NEXT round. A native dispatch failure is different because no review occurred: replace both snapshots with the external fallback values before that reviewer call, as specified in Phase A. A successful native call never sets the finalizer obligation. Phase E uses only the resulting snapshots when documenting or writing a finalizer receipt.
278
+
279
+ #### Phase A: Review
280
+
281
+ **Route by REVIEWER_BACKEND and REVIEWER_DIFFICULTY.**
282
+
283
+ If REVIEWER_BACKEND = `copilot-native`, execute one fresh native cycle:
284
+
285
+ 1. Use NATIVE_CHALLENGE prepared by Step -1. It must be the run-scoped
286
+ `..._ROUND_<round>_REVIEW.challenge.json` artifact created in this round.
287
+ Do not issue a second marker/challenge here.
288
+ 2. Read the returned nonce. Call the host Task tool with
289
+ `agent_type: rubber-duck`, a fresh name, and a prompt whose first line is
290
+ exactly `ARIS_REVIEW_NONCE=<nonce>`. Supply paths to claims, methods/code,
291
+ raw results, diff/current inputs, and reviewer memory—not an executor
292
+ summary. Require exactly one `Score: X/10` and `Verdict: ready | almost |
293
+ not ready` plus ranked weaknesses/minimum fixes/memory update. Do not pass a
294
+ model override: Copilot's complementary strategy selects it.
295
+ 3. After Task completes, run `python3 "<resolved-helper>" verify --challenge
296
+ "$NATIVE_CHALLENGE" --output
297
+ "review-stage/COPILOT_NATIVE_<run_id>_ROUND_<round>_REVIEW.evidence.json"
298
+ --response-output
299
+ "review-stage/COPILOT_NATIVE_<run_id>_ROUND_<round>_REVIEW.response.md"` in a new root Bash
300
+ call. Use only the extracted response artifact for Phase B.
301
+ 4. Exit 10 (same/unknown family), incomplete lifecycle, invalid response, or
302
+ unavailable complementary model is not a review. Apply the opposite-family
303
+ fallback table in `reviewer-routing.md`; if none is positively available,
304
+ emit `REVIEW_UNAVAILABLE`. Never emulate rubber-duck using a slash prompt,
305
+ `copilot --agent rubber-duck`, or generic subagent. Trace a pre-evidence
306
+ dispatch failure as `--backend copilot-native --status error` without
307
+ evidence, then trace the actual fallback separately; this error trace has no
308
+ authority at the stop gate. Before fallback, run
309
+ `copilot_native_evidence.py validate-challenge --challenge
310
+ "$NATIVE_CHALLENGE"` and take EXECUTOR_MODEL only from that output. Then:
311
+ - Anthropic/Google executor + usable Codex → set both REVIEWER_BACKEND and
312
+ `round_backend` to `codex` before the external call.
313
+ - OpenAI executor + manual reviewer reporting a known non-OpenAI model → set
314
+ both values to `manual` before the external call.
315
+ - Set `round_requires_external_acquittal=true` for either fallback, clear
316
+ NATIVE_EVIDENCE, and pass the validated executor model plus the fallback's
317
+ resolved reviewer model to `review_gate.py`. This deliberately uses the
318
+ stricter external-finalizer branch, which re-derives and enforces different
319
+ families. If the external call does not return a usable review, emit
320
+ `REVIEW_UNAVAILABLE`.
321
+
322
+ If REVIEWER_BACKEND = `copilot`, enforce opposite-family routing from the declared executor identity FIRST:
323
+ - Require `--executor-model <model>` parameter. If missing → `REVIEW_UNAVAILABLE`. Stop.
324
+ - Derive `executor_family` from `executor_model`:
325
+ - Model names containing `gpt`, `o1`, `o3`, `o4`, `chatgpt` → `openai`
326
+ - Model names containing `claude`, `sonnet`, `opus`, `haiku` → `anthropic`
327
+ - Model names containing `gemini` → `google`
328
+ - Anything else → `unknown`
329
+ - If `executor_family` is `unknown` → `REVIEW_UNAVAILABLE` (fail closed). Stop.
330
+ - Treat this as route selection, not attestation: persist
331
+ `executor_model_source: caller-declared`; a derived `family_relation:
332
+ different` remains `independence_verified: unverified` unless a future
333
+ stable runtime signal independently proves the parent executor model.
334
+ - Router picks opposite-family profile:
335
+ - `openai` → `"aris-reviewer-claude"` (anthropic, forced cross-family)
336
+ - `anthropic` → `"aris-reviewer-openai"` (openai, forced cross-family)
337
+ - `google` → `"aris-reviewer-openai"` (openai default)
338
+ - Verify the profile file exists at `.github/agents/<profile>.agent.md`.
339
+ If missing → `REVIEW_UNAVAILABLE`. Stop.
340
+ - Read the profile's first frontmatter `model:` value, derive its family, and
341
+ verify it is known and differs from `executor_family`. If not, fail closed.
342
+ - Verify `copilot --help` exposes `--model`, `--effort`, and `--allow-tool`.
343
+ Older/unpinned CLIs are `REVIEW_UNAVAILABLE`.
344
+ - Adapt the Codex MCP calls below to use the **`copilot --agent`** subprocess
345
+ (documented Copilot CLI form):
346
+ - Replace `mcp__codex__codex` with `copilot --agent "<profile>" --model "<parsed-model>" --effort xhigh --allow-tool=read --prompt "..."`
347
+ - Each round is a fresh `copilot --agent` call with the same profile +
348
+ `review-stage/REVIEWER_MEMORY.md` artifact carrying round-to-round state.
349
+ - The prompt text and Review Tracing are identical to the Codex path.
350
+ - If `copilot` CLI is unavailable → `REVIEW_UNAVAILABLE` (no MCP fallback).
351
+ - If `REVIEWER_DIFFICULTY = nightmare`, skip Copilot (nightmare requires Codex
352
+ exec): emit `REVIEW_UNAVAILABLE`.
353
+ See `shared-references/reviewer-routing.md`.
354
+
355
+ **If REVIEWER_BACKEND ∈ {codex, manual}:** use the backend-specific MCP call per the
356
+ Reviewer Calling Convention above. The prompt text is the same regardless of backend.
357
+
358
+ ##### Medium (default) — MCP Review
359
+
360
+ Send comprehensive context to the independent reviewer using the selected backend.
361
+
362
+ *For codex backend:*
363
+
364
+ ```
365
+ mcp__codex__codex:
366
+ model: gpt-5.6-sol
367
+ config: {"model_reasoning_effort": "xhigh"}
368
+ prompt: |
369
+ [Round N/MAX_ROUNDS of autonomous review loop]
370
+
371
+ Review the work directly from its artifacts — executor notes are not
372
+ evidence, so read the files yourself rather than trusting my framing:
373
+ - Claims / paper draft: <path>
374
+ - Methods / code under review: <path(s)>
375
+ - Raw results (verbatim files, not a summary): <path(s)>
376
+ - Changed since last round: <changed-file paths> — read the diff, not my description
377
+
378
+ Please act as a senior ML reviewer (NeurIPS/ICML level). Start from the
379
+ assumption that the work is broken somewhere — your job is to find where.
380
+ Be adversarial. Trust nothing the author tells you — verify everything
381
+ yourself.
382
+
383
+ 1. Score this work 1-10 for a top venue
384
+ 2. List remaining critical weaknesses (ranked by severity)
385
+ 3. For each weakness, specify the MINIMUM fix (experiment, analysis, or reframing)
386
+ 4. State clearly: is this READY for submission? Yes/No/Almost
387
+
388
+ Be brutally honest. If, after genuinely trying to break it, the work holds
389
+ up and is ready, say so clearly.
390
+
391
+ === SCOPE LIMITS (these bound what you PROPOSE, never what you look for) ===
392
+ Report anything that is actually wrong here — including a rare-looking case, if
393
+ this repo actually produces it. Then keep the fix in scope:
394
+ 1. This is a RESEARCH-WORKFLOW tool, not a security paper. Verification is
395
+ welcome; over-defense is not. Assume a cooperating operator on their own
396
+ machine — a malicious local user is NOT in the threat model.
397
+ 2. Do NOT propose SHA / hash / content-fingerprint / digest-binding schemes.
398
+ Reporting a real defect in hashing code that already exists is fine.
399
+ 3. NO speculative machinery: do not add feature flags, migration frameworks,
400
+ compat layers, wrappers, pins, or similar mechanisms unless evidence shows
401
+ a current repo defect they fix or an explicit existing invariant they must
402
+ preserve. "Load-bearing", "compatibility", and "not scaffolding" are labels,
403
+ not evidence. Point to the failing path/artifact or invariant, and check the
404
+ proposal's factual premises, such as whether a named package version exists.
405
+ 4. NO corner-case obsession: exotic encodings, symlink races, RTL text and
406
+ millisecond races are out of scope unless you can show the case arises here.
407
+ 5. Where a rubric or checklist is genuinely needed, do not over-mechanize
408
+ judgement. A clear sentence a human reads beats a scored table nobody
409
+ maintains.
410
+ Exception: code that runs remote commands, starts a network service, or installs
411
+ an MCP server runs on the user's machine with their credentials — trust-boundary
412
+ findings there are in scope and the default is strict.
413
+ Say plainly when something is correct. Do not manufacture findings.
414
+ ```
415
+
416
+ *For manual backend:* use `mcp__manual_review__review` with the `prompt` text above and `config: {"model_reasoning_effort": "xhigh", "executor_model": "<actual executor model>", "require_reviewer_model": true}`. Save the returned `threadId`.
417
+
418
+ If this is round 2+, use `mcp__codex__codex-reply` (codex) or `mcp__manual_review__review_reply` (manual) with the saved threadId.
419
+
420
+ ##### Hard — MCP Review + Reviewer Memory
421
+
422
+ Same as medium, but **prepend Reviewer Memory** to the prompt. Use the selected backend.
423
+
424
+ *For codex backend:*
425
+
426
+ ```
427
+ mcp__codex__codex:
428
+ model: gpt-5.6-sol
429
+ config: {"model_reasoning_effort": "xhigh"}
430
+ prompt: |
431
+ [Round N/MAX_ROUNDS of autonomous review loop]
432
+
433
+ ## Your Reviewer Memory (persistent across rounds)
434
+ [Paste full contents of review-stage/REVIEWER_MEMORY.md here]
435
+
436
+ IMPORTANT: You have memory from prior rounds. Check whether your
437
+ previous suspicions were genuinely addressed or merely sidestepped.
438
+ The author (the executor model) controls what context you see — be skeptical
439
+ of convenient omissions.
440
+
441
+ Review directly from the artifacts (paths below) — read the files yourself:
442
+ - Claims / methods / code: <path(s)>
443
+ - Raw results: <path(s)>
444
+ - Changed since last round: <changed-file paths> (read the raw diff)
445
+
446
+ Please act as a senior ML reviewer (NeurIPS/ICML level).
447
+ 1. Score this work 1-10 for a top venue
448
+ 2. List remaining critical weaknesses (ranked by severity)
449
+ 3. For each weakness, specify the MINIMUM fix
450
+ 4. State clearly: is this READY for submission? Yes/No/Almost
451
+ 5. **Memory update**: List any new suspicions, unresolved concerns,
452
+ or patterns you want to track in future rounds.
453
+
454
+ Be brutally honest. Actively look for things the author might be hiding.
455
+
456
+ === SCOPE LIMITS (these bound what you PROPOSE, never what you look for) ===
457
+ Report anything that is actually wrong here — including a rare-looking case, if
458
+ this repo actually produces it. Then keep the fix in scope:
459
+ 1. This is a RESEARCH-WORKFLOW tool, not a security paper. Verification is
460
+ welcome; over-defense is not. Assume a cooperating operator on their own
461
+ machine — a malicious local user is NOT in the threat model.
462
+ 2. Do NOT propose SHA / hash / content-fingerprint / digest-binding schemes.
463
+ Reporting a real defect in hashing code that already exists is fine.
464
+ 3. NO speculative machinery: do not add feature flags, migration frameworks,
465
+ compat layers, wrappers, pins, or similar mechanisms unless evidence shows
466
+ a current repo defect they fix or an explicit existing invariant they must
467
+ preserve. "Load-bearing", "compatibility", and "not scaffolding" are labels,
468
+ not evidence. Point to the failing path/artifact or invariant, and check the
469
+ proposal's factual premises, such as whether a named package version exists.
470
+ 4. NO corner-case obsession: exotic encodings, symlink races, RTL text and
471
+ millisecond races are out of scope unless you can show the case arises here.
472
+ 5. Where a rubric or checklist is genuinely needed, do not over-mechanize
473
+ judgement. A clear sentence a human reads beats a scored table nobody
474
+ maintains.
475
+ Exception: code that runs remote commands, starts a network service, or installs
476
+ an MCP server runs on the user's machine with their credentials — trust-boundary
477
+ findings there are in scope and the default is strict.
478
+ Say plainly when something is correct. Do not manufacture findings.
479
+ ```
480
+
481
+ ##### Nightmare — Codex Exec (GPT reads repo directly)
482
+
483
+ **Do NOT use MCP.** Instead, let GPT access the repo autonomously via `codex exec`:
484
+
485
+ ```bash
486
+ codex exec "$(cat <<'PROMPT'
487
+ You are an adversarial senior ML reviewer (NeurIPS/ICML level).
488
+ This is Round N/MAX_ROUNDS of an autonomous review loop.
489
+
490
+ ## Your Reviewer Memory (persistent across rounds)
491
+ [Paste full contents of review-stage/REVIEWER_MEMORY.md]
492
+
493
+ ## Instructions
494
+ You have FULL READ ACCESS to this repository. The author (the executor model) does NOT
495
+ control what you see — explore freely. Your job is to find problems the
496
+ author might hide or downplay.
497
+
498
+ DO THE FOLLOWING:
499
+ 1. Read the experiment code, results files (JSON/CSV), and logs YOURSELF
500
+ 2. Verify that reported numbers match what's actually in the output files
501
+ 3. Check if evaluation metrics are computed correctly (ground truth, not model output)
502
+ 4. Look for cherry-picked results, missing ablations, or suspicious hyperparameter choices
503
+ 5. Read NARRATIVE_REPORT.md or review-stage/AUTO_REVIEW.md for the author's claims — then verify each against code
504
+
505
+ OUTPUT FORMAT:
506
+ - Score: X/10
507
+ - Verdict: ready / almost / not ready
508
+ - Verified claims: [which claims you independently confirmed]
509
+ - Unverified/false claims: [which claims don't match the code or results]
510
+ - Weaknesses (ranked): [with MINIMUM fix for each]
511
+ - Memory update: [new suspicions and patterns to track next round]
512
+
513
+ Be adversarial. Trust nothing the author tells you — verify everything yourself.
514
+
515
+ === SCOPE LIMITS (these bound what you PROPOSE, never what you look for) ===
516
+ Report anything that is actually wrong here — including a rare-looking case, if
517
+ this repo actually produces it. Then keep the fix in scope:
518
+ 1. This is a RESEARCH-WORKFLOW tool, not a security paper. Verification is
519
+ welcome; over-defense is not. Assume a cooperating operator on their own
520
+ machine — a malicious local user is NOT in the threat model.
521
+ 2. Do NOT propose SHA / hash / content-fingerprint / digest-binding schemes.
522
+ Reporting a real defect in hashing code that already exists is fine.
523
+ 3. NO speculative machinery: do not add feature flags, migration frameworks,
524
+ compat layers, wrappers, pins, or similar mechanisms unless evidence shows
525
+ a current repo defect they fix or an explicit existing invariant they must
526
+ preserve. "Load-bearing", "compatibility", and "not scaffolding" are labels,
527
+ not evidence. Point to the failing path/artifact or invariant, and check the
528
+ proposal's factual premises, such as whether a named package version exists.
529
+ 4. NO corner-case obsession: exotic encodings, symlink races, RTL text and
530
+ millisecond races are out of scope unless you can show the case arises here.
531
+ 5. Where a rubric or checklist is genuinely needed, do not over-mechanize
532
+ judgement. A clear sentence a human reads beats a scored table nobody
533
+ maintains.
534
+ Exception: code that runs remote commands, starts a network service, or installs
535
+ an MCP server runs on the user's machine with their credentials — trust-boundary
536
+ findings there are in scope and the default is strict.
537
+ Say plainly when something is correct. Do not manufacture findings.
538
+ PROMPT
539
+ )" --skip-git-repo-check 2>&1
540
+ ```
541
+
542
+ **Key difference**: In nightmare mode, GPT independently reads code, result files, and logs. Claude cannot filter or curate what GPT sees. This is the closest analog to a real hostile reviewer who reads your actual paper + supplementary materials.
543
+
544
+ #### Phase B: Parse Assessment
545
+
546
+ **CRITICAL: Save the FULL raw response** from the reviewer verbatim (store in a variable for Phase E). For `copilot-native`, this must be the response artifact extracted by the evidence helper, never text copied by the executor. Do NOT discard or summarize — the raw text is the primary record.
547
+
548
+ Then extract structured fields:
549
+ - **Score** (numeric 1-10)
550
+ - **Verdict** ("ready" / "almost" / "not ready")
551
+ - **Action items** (ranked list of fixes)
552
+
553
+ #### Phase B.5: Reviewer Memory Update
554
+
555
+ After parsing the assessment, append to the canonical memory artifact at `review-stage/REVIEWER_MEMORY.md`. Both Copilot backends depend on this file for round-to-round continuity (each native subagent or compatibility subprocess is fresh), so the update runs regardless of `REVIEWER_DIFFICULTY`. No project-root fallback is permitted; create `review-stage/` before the first append:
556
+
557
+ ```markdown
558
+ # Reviewer Memory
559
+
560
+ ## Round 1 — Score: X/10
561
+
562
+ ### Raw Reviewer Response (verbatim)
563
+ [Paste the COMPLETE raw reviewer response here — never summarized or curated by the executor.]
564
+
565
+ ### Memory Update
566
+ [Reviewer's own memory update section, if provided — verbatim.]
567
+ - **Suspicion**: [what the reviewer flagged]
568
+ - **Unresolved**: [concerns not yet addressed]
569
+ - **Patterns**: [recurring issues the reviewer noticed]
570
+
571
+ ---
572
+
573
+ ## Round 2 — Score: X/10
574
+
575
+ ### Raw Reviewer Response (verbatim)
576
+ [Paste the COMPLETE raw reviewer response here.]
577
+
578
+ ### Memory Update
579
+ - **Previous suspicions addressed?**: [yes/no for each, with reviewer's judgment]
580
+ - **New suspicions**: [...]
581
+ - **Unresolved**: [carried forward + new]
582
+
583
+ ---
584
+ ```
585
+
586
+ **Rules**:
587
+ - **Append-only — never delete, never truncate.** The file is a reviewer-owned audit trail. The executor must never summarize, curate, or edit prior rounds' content. Append the reviewer's full raw response for this round verbatim, then append a memory update section if the reviewer provided one.
588
+ - Each round's append must be the reviewer's own words — if the reviewer's response includes a "Memory update" section, copy it verbatim as a `## Round N — Memory Update` subsection after the raw response.
589
+ - This file is passed back to the reviewer in the next round's Phase A — it is the reviewer's persistent memory.
590
+ - **Record the file's SHA-256 hash before each reviewer call** and pass it to `save_trace.sh` via `--memory-hash`. Hash the memory as supplied to the call (pre-call artifact), not the post-append version, so the trace proves which memory was in play for that invocation.
591
+ - **If the score REGRESSES round-to-round**, don't just write a new memory line:
592
+ diff the two rounds' raw `.response.md` files in `.aris/traces/` first and find
593
+ the exact criterion that flipped (see `shared-references/review-tracing.md`
594
+ § *Debugging With Traces*). The memory file is a summary; the trace is evidence.
595
+
596
+ #### Phase B.5.1: Stop-Evaluation Gate
597
+
598
+ **STOP CONDITION — branch by `round_backend` (the backend that actually ran this round), never by the forward-looking `REVIEWER_BACKEND`. Use the executable transition table in `review_gate.py`; resolve it through the canonical helper chain in `shared-references/integration-contract.md` §2. Its JSON `decision` and `next_backend` fields are authoritative. If the helper cannot be resolved or executed, emit `REVIEW_UNAVAILABLE`; do not improvise a transition.**
599
+
600
+ Invoke the gate once per completed round. Pass model strings only—the helper
601
+ derives families internally and does not accept caller-supplied family labels.
602
+ Backend availability must be positively established from the current host's
603
+ tool configuration; both finalizers default to unavailable:
604
+
605
+ ```bash
606
+ cd "$(git rev-parse --show-toplevel 2>/dev/null || pwd)" || exit 1
607
+ if [ -z "${ARIS_REPO:-}" ] && [ -f .aris/installed-skills.txt ]; then
608
+ ARIS_REPO=$(awk -F'\t' '$1=="repo_root"{print $2; exit}' .aris/installed-skills.txt 2>/dev/null) || true
609
+ fi
610
+ if [ -z "${ARIS_REPO:-}" ] && [ -f "$HOME/.aris/repo" ]; then
611
+ ARIS_REPO=$(cat "$HOME/.aris/repo" 2>/dev/null) || true
612
+ fi
613
+ REVIEW_GATE=".aris/tools/review_gate.py"
614
+ [ -f "$REVIEW_GATE" ] || REVIEW_GATE="tools/review_gate.py"
615
+ [ -f "$REVIEW_GATE" ] || { [ -n "${ARIS_REPO:-}" ] && REVIEW_GATE="$ARIS_REPO/tools/review_gate.py"; }
616
+ [ -f "$REVIEW_GATE" ] || REVIEW_GATE=""
617
+ [ -n "$REVIEW_GATE" ] || { echo "REVIEW_UNAVAILABLE: review_gate.py not resolved" >&2; exit 1; }
618
+
619
+ GATE_REVIEWER_MODEL="${REPORTED_REVIEWER_MODEL:-${REQUESTED_REVIEWER_MODEL:-${REVIEWER_MODEL:-}}}"
620
+ GATE_ARGS=(
621
+ --round-backend "$round_backend"
622
+ --score "$SCORE"
623
+ --verdict "$VERDICT"
624
+ --executor-model "${EXECUTOR_MODEL:-}"
625
+ --reviewer-model "$GATE_REVIEWER_MODEL"
626
+ )
627
+ if [[ "$round_backend" == "copilot-native" ]]; then
628
+ [[ -n "${NATIVE_EVIDENCE:-}" ]] || {
629
+ echo "REVIEW_UNAVAILABLE: native round has no evidence artifact" >&2
630
+ exit 1
631
+ }
632
+ GATE_ARGS+=(--native-evidence "$NATIVE_EVIDENCE")
633
+ fi
634
+ if [[ "$round_requires_external_acquittal" == "true" ]]; then
635
+ GATE_ARGS+=(--requires-external-acquittal)
636
+ fi
637
+ if [[ "${CODEX_AVAILABLE:-false}" == "true" ]]; then
638
+ GATE_ARGS+=(--codex-available)
639
+ fi
640
+ if [[ "${MANUAL_AVAILABLE:-false}" == "true" ]]; then
641
+ GATE_ARGS+=(--manual-available)
642
+ fi
643
+ if [[ "${MANUAL_IDENTITY_REPORTED:-false}" == "true" ]]; then
644
+ GATE_ARGS+=(--manual-identity-reported)
645
+ fi
646
+ GATE_JSON=$(python3 "$REVIEW_GATE" "${GATE_ARGS[@]}") || {
647
+ echo "REVIEW_UNAVAILABLE: review gate execution failed" >&2
648
+ exit 1
649
+ }
650
+ ```
651
+
652
+ Parse `GATE_JSON` as JSON; never infer a transition from the helper's prose
653
+ `reason`. A `review_unavailable` decision is terminal. Copy `next_backend` and
654
+ `requires_external_acquittal` into forward-looking state for `escalate` or
655
+ `continue`; only `stop` enters the successful termination path.
656
+
657
+ - **Default Codex compatibility (`round_backend = codex`, `round_requires_external_acquittal = false`):** score >= 6 AND verdict ∈ {"ready", "almost"} stops exactly as it did before this Copilot integration. Executor identity is advisory trace metadata and may be absent; do not turn a valid default-Codex positive verdict into `REVIEW_UNAVAILABLE`. This path does not write an external-finalizer receipt.
658
+ - **Existing Oracle/Agy routes:** when explicitly selected outside a Copilot-finalizer state, their qualifying positive verdicts retain the same pre-Copilot stop behavior. They are not valid substitutes once `requires_external_acquittal=true`; that state permits only Codex/manual.
659
+ - **Explicit manual backend:** a positive verdict still requires the response's exact `Reviewer-Model:` header. Missing identity is `REVIEW_UNAVAILABLE`.
660
+ - **Native Copilot round (`round_backend = copilot-native`):** the evidence
661
+ artifact is mandatory and revalidated by the gate. Its response-derived
662
+ Score/Verdict must equal the CLI fields. A qualifying positive returns
663
+ `decision: stop` with `identity_assurance: host_event_verified`; a negative
664
+ returns `continue` on `copilot-native`. No external-finalizer state is set.
665
+ - **Compatibility Copilot drive round (`round_backend = copilot`):** this path never stops the loop. A negative verdict continues on compatibility Copilot. A positive verdict returns `decision: escalate`, sets `requires_external_acquittal: true`, and chooses the next backend from the caller-declared executor family:
666
+ - `anthropic` or `google` → Codex when available, otherwise manual;
667
+ - `openai` → manual only (Codex would be same-family);
668
+ - `unknown` or no policy-approved finalizer → `REVIEW_UNAVAILABLE`.
669
+ - **External-finalizer round (`round_requires_external_acquittal = true`):** Codex/manual may stop on a qualifying positive verdict only when the model strings derive to known, different families; manual also requires its reported model header. This is fail-closed route consistency, not independent executor attestation. Record `identity_assurance: caller_declared` and `independence_verified: "unverified"`. A negative finalizer verdict continues on the same finalizer backend with the obligation still true.
670
+
671
+ On compatibility Copilot escalation, update the forward-looking `reviewer_backend` and `requires_external_acquittal` in `REVIEW_STATE.json`; keep `round_backend` and `round_requires_external_acquittal` unchanged for Phase E. Once a finalizer returns a qualifying positive verdict, set the forward flag to false and stop. `ACQUITTAL_LOG.jsonl` is an append-only audit receipt, never an input that lets a later compatibility Copilot verdict stop the loop. Native evidence is evaluated directly and never consults that log.
672
+
673
+ This evaluation runs AFTER Phase B.5 so the terminal-round memory is always appended to `review-stage/REVIEWER_MEMORY.md` before exit.
674
+
675
+ #### Phase B.6: Debate Protocol (hard + nightmare only)
676
+
677
+ **Skip entirely if `REVIEWER_DIFFICULTY = medium`.**
678
+
679
+ After parsing the review, the executor gets a chance to **rebut**:
680
+
681
+ **Step 1 — Executor Rebuttal:**
682
+
683
+ For each weakness the reviewer identified, the executor writes a structured response:
684
+
685
+ ```markdown
686
+ ### Rebuttal to Weakness #1: [title]
687
+ - **Accept / Partially Accept / Reject**
688
+ - **Argument**: [why this criticism is invalid, already addressed, or based on a misunderstanding]
689
+ - **Evidence**: [point to specific code, results, or prior round fixes]
690
+ ```
691
+
692
+ Rules for the executor's rebuttal:
693
+ - Must be honest — do NOT fabricate evidence or misrepresent results
694
+ - Can point out factual errors in the review (reviewer misread code, wrong metric, etc.)
695
+ - Can argue a weakness is out of scope or would require unreasonable effort
696
+ - Maximum 3 rebuttals per round (pick the most impactful to contest)
697
+
698
+ **Step 2 — Reviewer Rules on Rebuttal:**
699
+
700
+ Send the executor's rebuttal back to the reviewer for a ruling:
701
+
702
+ *Hard mode — use the selected backend for the rebuttal step:*
703
+
704
+ *For copilot-native:* run a fresh marker/challenge and invoke a fresh native
705
+ `rubber-duck` Task. Give it paths to `review-stage/REVIEWER_MEMORY.md`, the raw
706
+ review response, and `review-stage/ROUND_${ROUND}_REBUTTAL.md`; require it to
707
+ verify the cited files itself and return its updated Score/Verdict. Verify this
708
+ verdict-bearing ruling with distinct run-scoped
709
+ `..._ROUND_<round>_REBUTTAL.challenge.json`, `.evidence.json`, and
710
+ `.response.md` artifacts. Use that evidence (not the pre-debate evidence) in
711
+ the stop gate and trace.
712
+
713
+ *For compatibility copilot:* fresh `copilot --agent` subprocess with the same profile + `review-stage/REVIEWER_MEMORY.md` context:
714
+ ```bash
715
+ # Store the generated rebuttal as data; never paste memory/rebuttal text into
716
+ # a heredoc body, because either may contain a line matching its delimiter.
717
+ MEMORY_FILE="review-stage/REVIEWER_MEMORY.md"
718
+ REBUTTAL_FILE="review-stage/ROUND_${ROUND}_REBUTTAL.md"
719
+ [[ -f "$MEMORY_FILE" && -f "$REBUTTAL_FILE" ]] || {
720
+ echo "REVIEW_UNAVAILABLE: missing memory or rebuttal artifact" >&2
721
+ exit 1
722
+ }
723
+ PROMPTFILE="$(mktemp)" || { echo "REVIEW_UNAVAILABLE: mktemp failed" >&2; exit 1; }
724
+ trap 'rm -f "$PROMPTFILE"' EXIT
725
+ {
726
+ cat <<'ARIS_REBUTTAL_HEADER'
727
+ [Rebuttal ruling — same reviewer]
728
+
729
+ ## Your Memory From Previous Rounds
730
+ ARIS_REBUTTAL_HEADER
731
+ cat -- "$MEMORY_FILE"
732
+ cat <<'ARIS_REBUTTAL_MIDDLE'
733
+
734
+ The author rebuts your review:
735
+ ARIS_REBUTTAL_MIDDLE
736
+ cat -- "$REBUTTAL_FILE"
737
+ cat <<'ARIS_REBUTTAL_FOOTER'
738
+
739
+ For each rebuttal, rule:
740
+ - SUSTAINED (author's argument is valid, withdraw this weakness)
741
+ - OVERRULED (your original criticism stands, explain why)
742
+ - PARTIALLY SUSTAINED (revise the weakness to a narrower scope)
743
+
744
+ Then update your score if any weaknesses were withdrawn.
745
+ Include a Memory Update section at the end of your response.
746
+ ARIS_REBUTTAL_FOOTER
747
+ } > "$PROMPTFILE"
748
+ copilot --agent "$REVIEWER_PROFILE" --model "$REVIEWER_MODEL" \
749
+ --effort xhigh --allow-tool=read --prompt "$(cat "$PROMPTFILE")"
750
+ ```
751
+
752
+ *For codex:*
753
+ ```
754
+ mcp__codex__codex-reply:
755
+ threadId: [saved]
756
+ # inherits the thread's model/effort — do not re-send
757
+ prompt: |
758
+ The author rebuts your review:
759
+ ```
760
+
761
+ *For manual:* use `mcp__manual_review__review_reply` with the same `threadId` and prompt.
762
+
763
+ The prompt content:
764
+
765
+ ```
766
+ The author rebuts your review:
767
+
768
+ [paste executor's rebuttal]
769
+
770
+ For each rebuttal, rule:
771
+ - SUSTAINED (author's argument is valid, withdraw this weakness)
772
+ - OVERRULED (your original criticism stands, explain why)
773
+ - PARTIALLY SUSTAINED (revise the weakness to a narrower scope)
774
+
775
+ Then update your score if any weaknesses were withdrawn.
776
+ ```
777
+
778
+ *Nightmare mode (codex exec):*
779
+ ```bash
780
+ codex exec "$(cat <<'PROMPT'
781
+ You are the same adversarial reviewer. The author rebuts your review:
782
+
783
+ [paste executor's rebuttal]
784
+
785
+ VERIFY the author's evidence claims yourself — read the files they reference.
786
+ Do NOT take their word for it.
787
+
788
+ For each rebuttal, rule:
789
+ - SUSTAINED (verified and valid)
790
+ - OVERRULED (evidence doesn't check out or argument is weak)
791
+ - PARTIALLY SUSTAINED (partially valid, narrow the weakness)
792
+
793
+ Update your score. Update your memory.
794
+ PROMPT
795
+ )" --skip-git-repo-check 2>&1
796
+ ```
797
+
798
+ **Step 3 — Update score and action items** based on the ruling:
799
+ - SUSTAINED weaknesses: remove from action items
800
+ - OVERRULED: keep as-is
801
+ - PARTIALLY SUSTAINED: revise scope
802
+
803
+ Append the full debate transcript to `review-stage/AUTO_REVIEW.md` under the round's entry.
804
+
805
+ #### Human Checkpoint (if enabled)
806
+
807
+ **Skip this step entirely if `HUMAN_CHECKPOINT = false`.**
808
+
809
+ When `HUMAN_CHECKPOINT = true`, present the review results and wait for user input:
810
+
811
+ ```
812
+ 📋 Round N/MAX_ROUNDS review complete.
813
+
814
+ Score: X/10 — [verdict]
815
+ Top weaknesses:
816
+ 1. [weakness 1]
817
+ 2. [weakness 2]
818
+ 3. [weakness 3]
819
+
820
+ Suggested fixes:
821
+ 1. [fix 1]
822
+ 2. [fix 2]
823
+ 3. [fix 3]
824
+
825
+ Options:
826
+ - Reply "go" or "continue" → implement all suggested fixes
827
+ - Reply with custom instructions → implement your modifications instead
828
+ - Reply "skip 2" → skip fix #2, implement the rest
829
+ - Reply "stop" → end the loop, document current state
830
+ ```
831
+
832
+ Wait for the user's response. Parse their input:
833
+ - **Approval** ("go", "continue", "ok", "proceed"): proceed to Phase C with all suggested fixes
834
+ - **Custom instructions** (any other text): treat as additional/replacement guidance for Phase C. Merge with reviewer suggestions where appropriate
835
+ - **Skip specific fixes** ("skip 1,3"): remove those fixes from the action list
836
+ - **Stop** ("stop", "enough", "done"): terminate the loop, jump to Termination
837
+
838
+ #### Feishu Notification (if configured)
839
+
840
+ After parsing the score, check if `~/.claude/feishu.json` exists and mode is not `"off"`:
841
+ - Send a `review_scored` notification: "Round N: X/10 — [verdict]" with top 3 weaknesses
842
+ - If **interactive** mode and verdict is "almost": send as checkpoint, wait for user reply on whether to continue or stop
843
+ - If config absent or mode off: skip entirely (no-op)
844
+
845
+ #### Phase C: Implement Fixes (if not stopping)
846
+
847
+ For each action item (highest priority first):
848
+
849
+ 1. **Code changes**: Write/modify experiment scripts, model code, analysis scripts
850
+ 2. **Run experiments**: Deploy to GPU server via SSH + screen/tmux
851
+ 3. **Analysis**: Run evaluation, collect results, update figures/tables
852
+ 4. **Documentation**: Update project notes and review document
853
+
854
+ Prioritization rules:
855
+ - Skip fixes requiring excessive compute (flag for manual follow-up)
856
+ - Skip fixes requiring external data/models not available
857
+ - Prefer reframing/analysis over new experiments when both address the concern
858
+ - Always implement metric additions (cheap, high impact)
859
+
860
+ #### Phase D: Wait for Results
861
+
862
+ If experiments were launched:
863
+ - Monitor remote sessions for completion
864
+ - Collect results from output files and logs
865
+ - **Training quality check** — if W&B is configured, invoke `/training-check` to verify training was healthy (no NaN, no divergence, no plateau). If W&B not available, skip silently. Flag any quality issues in the next review round.
866
+
867
+ #### Phase E: Document Round
868
+
869
+ Append to `review-stage/AUTO_REVIEW.md`:
870
+
871
+ ```markdown
872
+ ## Round N (timestamp)
873
+
874
+ ### Assessment (Summary)
875
+ - Score: X/10
876
+ - Verdict: [ready/almost/not ready]
877
+ - Key criticisms: [bullet list]
878
+
879
+ ### Reviewer Raw Response
880
+
881
+ <details>
882
+ <summary>Click to expand full reviewer response</summary>
883
+
884
+ [Paste the COMPLETE raw response from the reviewer here — verbatim, unedited.
885
+ This is the authoritative record. Do NOT truncate or paraphrase.]
886
+
887
+ </details>
888
+
889
+ ### Debate Transcript (hard + nightmare only)
890
+
891
+ <details>
892
+ <summary>Click to expand debate</summary>
893
+
894
+ **Executor Rebuttal:**
895
+ [paste rebuttal]
896
+
897
+ **Reviewer Ruling:**
898
+ [paste ruling — SUSTAINED / OVERRULED / PARTIALLY SUSTAINED for each]
899
+
900
+ **Score adjustment**: X/10 → Y/10
901
+
902
+ </details>
903
+
904
+ ### Actions Taken
905
+ - [what was implemented/changed]
906
+
907
+ ### Results
908
+ - [experiment outcomes, if any]
909
+
910
+ ### Status
911
+ - [continuing to round N+1 / stopping]
912
+ - Difficulty: [medium/hard/nightmare]
913
+ ```
914
+
915
+ **Write `review-stage/REVIEW_STATE.json`** with current `run_id`, round, threadId, score, verdict, `reviewer_backend`, `requires_external_acquittal`, and any pending experiments. The `run_id` field MUST persist unchanged from initialization; do NOT regenerate it per round.
916
+
917
+ **Backend labeling for the state file:** The `reviewer_backend` field in `REVIEW_STATE.json` controls the continuation mechanism for the NEXT round (used on resume), not the round just documented. During Phase E:
918
+ - Use `round_backend` (snapshotted at round start, step 0) to label the CURRENT round in `AUTO_REVIEW.md` documentation (e.g., "Reviewer backend: copilot-native").
919
+ - Use `round_requires_external_acquittal` to decide whether the CURRENT round was a Copilot-triggered finalizer. Write `requires_external_acquittal` in state as the forward-looking obligation for the NEXT round.
920
+ - Write `reviewer_backend` in `REVIEW_STATE.json` to the value that should control the NEXT round — this is either (a) unchanged from the current round's backend if no escalation occurred, or (b) the escalation backend set during Phase B.5.1. Never substitute `round_backend` for this forward-looking field.
921
+ - When no escalation happened, `round_backend == reviewer_backend` (trivially safe).
922
+ - For a native round, persist `native_evidence_id`, `native_evidence_path`, both
923
+ host-event model IDs/sources, `family_relation: different`, and
924
+ `identity_assurance: host_event_verified`. Never copy these fields from prose.
925
+
926
+ **If `round_backend ∈ {codex, manual}` AND `round_requires_external_acquittal = true` AND score >= 6 AND verdict ∈ {"ready", "almost"}:** append one external-finalizer line to `review-stage/ACQUITTAL_LOG.jsonl`:
927
+ ```
928
+ {"run_id":"<current-run_id>","round":<N>,"backend":"<codex|manual>","effort":"xhigh","verdict":"<ready|almost>","score":<score>,"executor_model":"<from-trace>","executor_model_source":"caller-declared","executor_family":"<derived-from-executor_model>","reviewer_model":"<from-trace-or-manual-Reviewer-Model>","reviewer_model_source":"<requested|backend-reported>","reviewer_family":"<derived-from-reviewer_model>","family_relation":"different","identity_assurance":"caller_declared","independence_verified":"unverified","trace_id":"<skill>/<YYYY-MM-DD>_run<NN>","timestamp":"<ISO8601>"}
929
+ ```
930
+ Use `>>` (append), never `>`. Re-derive both families from model strings and reject unknown/same-family pairs, but copy the model-source and assurance fields without upgrading them. The `trace_id` MUST be the actual trace directory path relative to `.aris/traces/` (e.g., `auto-review-loop/2026-07-13_run01`), matching the RUN_ID format from `save_trace.sh`: `<YYYY-MM-DD>_run<NN>` with the skill-name subdirectory prefix. Do NOT fabricate a synthetic `trace_...` identifier.
931
+
932
+ **Append to `findings.md`** (when `COMPACT = true`): one-line entry per key finding this round:
933
+
934
+ ```markdown
935
+ - [Round N] [positive/negative/unexpected]: [one-sentence finding] (metric: X.XX → Y.YY)
936
+ ```
937
+
938
+ Increment round counter → back to Phase A.
939
+
940
+ ### Termination
941
+
942
+ When loop ends (positive assessment or max rounds):
943
+
944
+ 1. Update `review-stage/REVIEW_STATE.json` with `"status": "completed"`
945
+ 2. Write final summary to `review-stage/AUTO_REVIEW.md`
946
+ 3. Update project notes with conclusions
947
+ 4. **Write method/pipeline description** to `review-stage/AUTO_REVIEW.md` under a `## Method Description` section — a concise 1-2 paragraph description of the final method, its architecture, and data flow. This serves as input for `/paper-illustration` in Workflow 3 (so it can generate architecture diagrams automatically).
948
+ 5. **Generate claims from results** — invoke `/result-to-claim` to convert experiment results from `review-stage/AUTO_REVIEW.md` into structured paper claims. Output: `CLAIMS_FROM_RESULTS.md`. This bridges Workflow 2 → Workflow 3 so `/paper-plan` can directly use validated claims instead of extracting them from scratch. If `/result-to-claim` is not installed, skip this step (no `CLAIMS_FROM_RESULTS.md` is produced; `/paper-plan` extracts claims from the narrative as before) — but NEVER fabricate the file or its verdict. If it ran but its output starts with `verdict: REVIEW_UNAVAILABLE`, keep that file AS-IS (do not overwrite or paraphrase it) and record in `AUTO_REVIEW.md` that claims are UNADJUDICATED — downstream paper stages must not treat them as validated.
949
+ 6. If stopped at max rounds without positive assessment:
950
+ - List remaining blockers
951
+ - Estimate effort needed for each
952
+ - Suggest whether to continue manually or pivot
953
+ 7. **Feishu notification** (if configured): Send `pipeline_done` with final score progression table
954
+ 8. **Render HTML view** (if `RENDER_HTML = true`, default): invoke `/render-html` on the cumulative review log:
955
+ ```
956
+ /render-html "review-stage/AUTO_REVIEW.md" --no-review --state review-stage/REVIEW_STATE.json
957
+ ```
958
+ Pass `--state` explicitly (the helper does not auto-discover the sidecar). Drop the `--state` flag if `REVIEW_STATE.json` doesn't exist. HTML lands at `review-stage/AUTO_REVIEW.html` with embedded source SHA256. **Non-blocking**: if `/render-html` fails, log the error and continue — the HTML is a convenience, not a termination prerequisite. Skip if `RENDER_HTML = false`.
959
+
960
+ ## Key Rules
961
+
962
+ - **Large file handling**: If the Write tool fails due to file size, immediately retry using Bash (`cat << 'EOF' > file`) to write in chunks. Do NOT ask the user for permission — just do it silently.
963
+
964
+ - ALWAYS use `config: {"model_reasoning_effort": "xhigh"}` for maximum reasoning depth
965
+ - **Native Copilot is an evidence-gated acceptance backend.** It never pins a
966
+ reviewer model: Copilot selects the complementary rubber-duck model, and the
967
+ helper verifies the actual cross-family pair from host events. A native
968
+ positive needs no external finalizer.
969
+ - **Explicit compatibility Copilot remains drive-only.** Its `copilot --agent`
970
+ calls pin profile model/xhigh/read-only access and require a traced
971
+ Codex/manual finalizer; caller-declared identity remains unverified.
972
+ - Save `threadId` (codex/manual), fresh evidence path/ID (`copilot-native`), or `reviewer_profile` (compatibility copilot); use the appropriate continuation mechanism
973
+ - **Anti-hallucination citations**: When adding references during fixes, NEVER fabricate BibTeX. Use the same DBLP → CrossRef → `[VERIFY]` chain as `/paper-write`: (1) `curl -s "https://dblp.org/search/publ/api?q=TITLE&format=json"` → get key → `curl -s "https://dblp.org/rec/{key}.bib"`, (2) if not found, `curl -sLH "Accept: application/x-bibtex" "https://doi.org/{doi}"`, (3) if both fail, mark with `% [VERIFY]`. Do NOT generate BibTeX from memory.
974
+ - Be honest — include negative results and failed experiments
975
+ - Do NOT hide weaknesses to game a positive score
976
+ - Implement fixes BEFORE re-reviewing (don't just promise to fix)
977
+ - **Exhaust before surrendering** — before marking any reviewer concern as "cannot address": (1) try at least 2 different solution paths, (2) for experiment issues, adjust hyperparameters or try an alternative baseline, (3) for theory issues, provide a weaker version of the result or an alternative argument, (4) only then concede narrowly and bound the damage. Never give up on the first attempt.
978
+ - If an experiment takes > 30 minutes, launch it and continue with other fixes while waiting
979
+ - Document EVERYTHING — the review log should be self-contained
980
+ - Update project notes after each round, not just at the end
981
+
982
+ ## Prompt Template for Round 2+
983
+
984
+ Use the selected backend. *For copilot-native:* fresh
985
+ marker/challenge/rubber-duck/evidence cycle with a new run-scoped REVIEW
986
+ artifact set, with paths to
987
+ `review-stage/REVIEWER_MEMORY.md` and current inputs. *For compatibility
988
+ copilot:* fresh `copilot --agent` subprocess with the same profile + memory
989
+ artifact. *For codex:* `mcp__codex__codex-reply` with the saved threadId. *For
990
+ manual:* `mcp__manual_review__review_reply` with the saved threadId.
991
+
992
+ Before invoking the Copilot subprocess, use the **Write tool** (not Bash,
993
+ `echo`, a heredoc, or generated shell assignments) to overwrite
994
+ `review-stage/CURRENT_REVIEW_INPUTS.md`. Put the exact changed paths, diff
995
+ artifact/range, and result paths under static labels in that file. Repository
996
+ paths are untrusted data: never splice any byte from this artifact into shell
997
+ source. The fixed filename below is the only value the shell template needs.
998
+
999
+ ```
1000
+ [For copilot:]
1001
+
1002
+ # ARIS_ROUND2_COPILOT_BEGIN
1003
+ # Dynamic values were written with the Write tool; shell only reads them as data.
1004
+ MEMORY_FILE="review-stage/REVIEWER_MEMORY.md"
1005
+ ROUND_INPUT_FILE="review-stage/CURRENT_REVIEW_INPUTS.md"
1006
+ [[ -f "$MEMORY_FILE" && -f "$ROUND_INPUT_FILE" ]] || {
1007
+ echo "REVIEW_UNAVAILABLE: missing reviewer memory or round inputs" >&2
1008
+ exit 1
1009
+ }
1010
+ PROMPTFILE="$(mktemp)" || { echo "REVIEW_UNAVAILABLE: mktemp failed" >&2; exit 1; }
1011
+ trap 'rm -f "$PROMPTFILE"' EXIT
1012
+ {
1013
+ cat <<'ARIS_ROUND_HEADER'
1014
+ [Round N update]
1015
+
1016
+ ## Your Memory From Previous Rounds
1017
+ ARIS_ROUND_HEADER
1018
+ cat -- "$MEMORY_FILE"
1019
+ cat <<'ARIS_ROUND_STATE'
1020
+
1021
+ Since your last review these files changed — read them yourself; do not
1022
+ take my word for what changed or whether it worked:
1023
+ ARIS_ROUND_STATE
1024
+ cat -- "$ROUND_INPUT_FILE"
1025
+ cat <<'ARIS_ROUND_INSTRUCTIONS'
1026
+
1027
+ Please re-score and re-assess. Are the remaining concerns addressed?
1028
+ Same format: Score, Verdict, Remaining Weaknesses, Minimum Fixes.
1029
+
1030
+ === SCOPE LIMITS (these bound what you PROPOSE, never what you look for) ===
1031
+ Report anything that is actually wrong here — including a rare-looking case, if
1032
+ this repo actually produces it. Then keep the fix in scope:
1033
+ 1. This is a RESEARCH-WORKFLOW tool, not a security paper. Verification is
1034
+ welcome; over-defense is not. Assume a cooperating operator on their own
1035
+ machine — a malicious local user is NOT in the threat model.
1036
+ 2. Do NOT propose SHA / hash / content-fingerprint / digest-binding schemes.
1037
+ Reporting a real defect in hashing code that already exists is fine.
1038
+ 3. NO speculative machinery: do not add feature flags, migration frameworks,
1039
+ compat layers, wrappers, pins, or similar mechanisms unless evidence shows
1040
+ a current repo defect they fix or an explicit existing invariant they must
1041
+ preserve. "Load-bearing", "compatibility", and "not scaffolding" are labels,
1042
+ not evidence. Point to the failing path/artifact or invariant, and check the
1043
+ proposal's factual premises, such as whether a named package version exists.
1044
+ 4. NO corner-case obsession: exotic encodings, symlink races, RTL text and
1045
+ millisecond races are out of scope unless you can show the case arises here.
1046
+ 5. Where a rubric or checklist is genuinely needed, do not over-mechanize
1047
+ judgement. A clear sentence a human reads beats a scored table nobody
1048
+ maintains.
1049
+ Exception: code that runs remote commands, starts a network service, or installs
1050
+ an MCP server runs on the user's machine with their credentials — trust-boundary
1051
+ findings there are in scope and the default is strict.
1052
+ Say plainly when something is correct. Do not manufacture findings.
1053
+
1054
+ At the end of your review, include a Memory Update section — this will
1055
+ be passed back to you next round.
1056
+ ARIS_ROUND_INSTRUCTIONS
1057
+ } > "$PROMPTFILE"
1058
+ copilot --agent "$REVIEWER_PROFILE" --model "$REVIEWER_MODEL" \
1059
+ --effort xhigh --allow-tool=read --prompt "$(cat "$PROMPTFILE")"
1060
+ # ARIS_ROUND2_COPILOT_END
1061
+
1062
+ [For codex:] mcp__codex__codex-reply:
1063
+ threadId: [saved from round 1]
1064
+ # inherits the thread's model/effort — do not re-send
1065
+ prompt: |
1066
+ [Round N update]
1067
+
1068
+ Since your last review these files changed — read them yourself; do not
1069
+ take my word for what changed or whether it worked:
1070
+ - Changed files: <paths>
1071
+ - Raw diff: <path, or the `git diff` range>
1072
+ - Updated raw results: <result-file paths> (verbatim files, not a pasted table)
1073
+
1074
+ Please re-score and re-assess. Are the remaining concerns addressed?
1075
+ Same format: Score, Verdict, Remaining Weaknesses, Minimum Fixes.
1076
+
1077
+ === SCOPE LIMITS (these bound what you PROPOSE, never what you look for) ===
1078
+ Report anything that is actually wrong here — including a rare-looking case, if
1079
+ this repo actually produces it. Then keep the fix in scope:
1080
+ 1. This is a RESEARCH-WORKFLOW tool, not a security paper. Verification is
1081
+ welcome; over-defense is not. Assume a cooperating operator on their own
1082
+ machine — a malicious local user is NOT in the threat model.
1083
+ 2. Do NOT propose SHA / hash / content-fingerprint / digest-binding schemes.
1084
+ Reporting a real defect in hashing code that already exists is fine.
1085
+ 3. NO speculative machinery: do not add feature flags, migration frameworks,
1086
+ compat layers, wrappers, pins, or similar mechanisms unless evidence shows
1087
+ a current repo defect they fix or an explicit existing invariant they must
1088
+ preserve. "Load-bearing", "compatibility", and "not scaffolding" are labels,
1089
+ not evidence. Point to the failing path/artifact or invariant, and check the
1090
+ proposal's factual premises, such as whether a named package version exists.
1091
+ 4. NO corner-case obsession: exotic encodings, symlink races, RTL text and
1092
+ millisecond races are out of scope unless you can show the case arises here.
1093
+ 5. Where a rubric or checklist is genuinely needed, do not over-mechanize
1094
+ judgement. A clear sentence a human reads beats a scored table nobody
1095
+ maintains.
1096
+ Exception: code that runs remote commands, starts a network service, or installs
1097
+ an MCP server runs on the user's machine with their credentials — trust-boundary
1098
+ findings there are in scope and the default is strict.
1099
+ Say plainly when something is correct. Do not manufacture findings.
1100
+ ```
1101
+
1102
+ ## Review Tracing
1103
+
1104
+ After each reviewer call (`task(agent_type=rubber-duck)` for `copilot-native`,
1105
+ Codex/manual MCP calls, or the compatibility `copilot --agent` subprocess),
1106
+ save the trace following `shared-references/review-tracing.md` (Policy C —
1107
+ forensic; never silently skip). Native calls MUST pass `--backend
1108
+ copilot-native --native-evidence "$NATIVE_EVIDENCE"`; the helper revalidates
1109
+ and supplies the response and actual model pair. The sole exception is a native
1110
+ dispatch that failed before evidence existed: trace it with `--backend
1111
+ copilot-native --status error --fallback-reason <reason>` and no evidence, then
1112
+ trace any actual fallback reviewer separately. Use `save_trace.sh` resolved
1113
+ through the canonical chain, or write the same schema directly only if that
1114
+ forensic helper is unreachable. Respect `--- trace:` (default `full`).
1115
+
1116
+ ## Stop-Gate State-Transition Tests
1117
+
1118
+ The canonical transition table is `tools/review_gate.py` in the ARIS repository (resolved at runtime as `review_gate.py` through the helper chain) and is covered by `tests/test_review_gate.py`. The required cases are:
1119
+
1120
+ 1. Default Codex positive → `stop` even when executor model identity is absent (backward compatibility).
1121
+ 2. High score + `not ready` → `continue`.
1122
+ 3. Verified native Copilot positive → `stop` with
1123
+ `identity_assurance=host_event_verified`.
1124
+ 4. Native negative → `continue` on `copilot-native`; missing/invalid/mismatched
1125
+ evidence → `REVIEW_UNAVAILABLE`.
1126
+ 5. Compatibility Copilot positive under declared Anthropic/Google executor →
1127
+ `escalate` to Codex and set `requires_external_acquittal=true`.
1128
+ 6. Compatibility Copilot positive under declared OpenAI executor → `escalate`
1129
+ to manual; Codex is forbidden as same-family.
1130
+ 7. Compatibility Copilot negative → `continue` on compatibility Copilot.
1131
+ 8. Unknown executor family, unavailable finalizer, same-family finalizer, or
1132
+ manual finalizer without `Reviewer-Model:` → `REVIEW_UNAVAILABLE`.
1133
+ 9. Positive finalizer with known, different declared families → `stop`, while
1134
+ identity assurance remains `caller_declared` / `unverified`.
1135
+
1136
+ `ACQUITTAL_LOG.jsonl` is tested as append-only compatibility-drive audit output;
1137
+ it is never consulted for native termination.