paperlint 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (762) hide show
  1. package/.github/dependabot.yml +72 -0
  2. package/.github/workflows/ci.yml +297 -0
  3. package/.github/workflows/dependabot-automerge.yml +70 -0
  4. package/.github/workflows/pr-title.yml +59 -0
  5. package/.github/workflows/release.yml +54 -0
  6. package/CLAUDE.md +598 -0
  7. package/CONTRIBUTING.md +159 -0
  8. package/LICENSE +21 -0
  9. package/README.md +240 -0
  10. package/action.harness.mjs +287 -0
  11. package/action.mutations.mjs +162 -0
  12. package/action.yml +138 -0
  13. package/bin/rpp.mjs +43 -0
  14. package/dist/action-ref.d.ts +12 -0
  15. package/dist/action-ref.d.ts.map +1 -0
  16. package/dist/action-ref.js +16 -0
  17. package/dist/action-ref.js.map +1 -0
  18. package/dist/adapters/banal/failure.d.ts +73 -0
  19. package/dist/adapters/banal/failure.d.ts.map +1 -0
  20. package/dist/adapters/banal/failure.js +58 -0
  21. package/dist/adapters/banal/failure.js.map +1 -0
  22. package/dist/adapters/banal/index.d.ts +17 -0
  23. package/dist/adapters/banal/index.d.ts.map +1 -0
  24. package/dist/adapters/banal/index.js +56 -0
  25. package/dist/adapters/banal/index.js.map +1 -0
  26. package/dist/adapters/banal/install.d.ts +26 -0
  27. package/dist/adapters/banal/install.d.ts.map +1 -0
  28. package/dist/adapters/banal/install.js +15 -0
  29. package/dist/adapters/banal/install.js.map +1 -0
  30. package/dist/adapters/banal/invocation.d.ts +48 -0
  31. package/dist/adapters/banal/invocation.d.ts.map +1 -0
  32. package/dist/adapters/banal/invocation.js +43 -0
  33. package/dist/adapters/banal/invocation.js.map +1 -0
  34. package/dist/adapters/banal/locate.d.ts +50 -0
  35. package/dist/adapters/banal/locate.d.ts.map +1 -0
  36. package/dist/adapters/banal/locate.js +34 -0
  37. package/dist/adapters/banal/locate.js.map +1 -0
  38. package/dist/adapters/banal/output.d.ts +27 -0
  39. package/dist/adapters/banal/output.d.ts.map +1 -0
  40. package/dist/adapters/banal/output.js +112 -0
  41. package/dist/adapters/banal/output.js.map +1 -0
  42. package/dist/adapters/banal/pin.d.ts +19 -0
  43. package/dist/adapters/banal/pin.d.ts.map +1 -0
  44. package/dist/adapters/banal/pin.js +15 -0
  45. package/dist/adapters/banal/pin.js.map +1 -0
  46. package/dist/adapters/banal/probe.d.ts +12 -0
  47. package/dist/adapters/banal/probe.d.ts.map +1 -0
  48. package/dist/adapters/banal/probe.js +27 -0
  49. package/dist/adapters/banal/probe.js.map +1 -0
  50. package/dist/adapters/banal/run.d.ts +89 -0
  51. package/dist/adapters/banal/run.d.ts.map +1 -0
  52. package/dist/adapters/banal/run.js +104 -0
  53. package/dist/adapters/banal/run.js.map +1 -0
  54. package/dist/adapters/banal/settings.d.ts +18 -0
  55. package/dist/adapters/banal/settings.d.ts.map +1 -0
  56. package/dist/adapters/banal/settings.js +29 -0
  57. package/dist/adapters/banal/settings.js.map +1 -0
  58. package/dist/adapters/banal/xml.d.ts +48 -0
  59. package/dist/adapters/banal/xml.d.ts.map +1 -0
  60. package/dist/adapters/banal/xml.js +67 -0
  61. package/dist/adapters/banal/xml.js.map +1 -0
  62. package/dist/adapters/curl/download.io.d.ts +14 -0
  63. package/dist/adapters/curl/download.io.d.ts.map +1 -0
  64. package/dist/adapters/curl/download.io.js +69 -0
  65. package/dist/adapters/curl/download.io.js.map +1 -0
  66. package/dist/adapters/curl/index.d.ts +6 -0
  67. package/dist/adapters/curl/index.d.ts.map +1 -0
  68. package/dist/adapters/curl/index.js +6 -0
  69. package/dist/adapters/curl/index.js.map +1 -0
  70. package/dist/adapters/memory/index.d.ts +43 -0
  71. package/dist/adapters/memory/index.d.ts.map +1 -0
  72. package/dist/adapters/memory/index.js +79 -0
  73. package/dist/adapters/memory/index.js.map +1 -0
  74. package/dist/adapters/node/files.io.d.ts +3 -0
  75. package/dist/adapters/node/files.io.d.ts.map +1 -0
  76. package/dist/adapters/node/files.io.js +31 -0
  77. package/dist/adapters/node/files.io.js.map +1 -0
  78. package/dist/adapters/node/host.io.d.ts +3 -0
  79. package/dist/adapters/node/host.io.d.ts.map +1 -0
  80. package/dist/adapters/node/host.io.js +14 -0
  81. package/dist/adapters/node/host.io.js.map +1 -0
  82. package/dist/adapters/node/index.d.ts +25 -0
  83. package/dist/adapters/node/index.d.ts.map +1 -0
  84. package/dist/adapters/node/index.js +14 -0
  85. package/dist/adapters/node/index.js.map +1 -0
  86. package/dist/adapters/node/process.io.d.ts +14 -0
  87. package/dist/adapters/node/process.io.d.ts.map +1 -0
  88. package/dist/adapters/node/process.io.js +41 -0
  89. package/dist/adapters/node/process.io.js.map +1 -0
  90. package/dist/adapters/node/workspace.io.d.ts +4 -0
  91. package/dist/adapters/node/workspace.io.d.ts.map +1 -0
  92. package/dist/adapters/node/workspace.io.js +33 -0
  93. package/dist/adapters/node/workspace.io.js.map +1 -0
  94. package/dist/adapters/pdfjs/fill.d.ts +42 -0
  95. package/dist/adapters/pdfjs/fill.d.ts.map +1 -0
  96. package/dist/adapters/pdfjs/fill.js +91 -0
  97. package/dist/adapters/pdfjs/fill.js.map +1 -0
  98. package/dist/build-engine.d.ts +48 -0
  99. package/dist/build-engine.d.ts.map +1 -0
  100. package/dist/build-engine.js +148 -0
  101. package/dist/build-engine.js.map +1 -0
  102. package/dist/build.d.ts +163 -0
  103. package/dist/build.d.ts.map +1 -0
  104. package/dist/build.js +575 -0
  105. package/dist/build.js.map +1 -0
  106. package/dist/cli.d.ts +151 -0
  107. package/dist/cli.d.ts.map +1 -0
  108. package/dist/cli.js +951 -0
  109. package/dist/cli.js.map +1 -0
  110. package/dist/doctor.d.ts +42 -0
  111. package/dist/doctor.d.ts.map +1 -0
  112. package/dist/doctor.js +280 -0
  113. package/dist/doctor.js.map +1 -0
  114. package/dist/domain/geometry.d.ts +71 -0
  115. package/dist/domain/geometry.d.ts.map +1 -0
  116. package/dist/domain/geometry.js +35 -0
  117. package/dist/domain/geometry.js.map +1 -0
  118. package/dist/domain/host.d.ts +16 -0
  119. package/dist/domain/host.d.ts.map +1 -0
  120. package/dist/domain/host.js +8 -0
  121. package/dist/domain/host.js.map +1 -0
  122. package/dist/domain/page-layout.d.ts +34 -0
  123. package/dist/domain/page-layout.d.ts.map +1 -0
  124. package/dist/domain/page-layout.js +8 -0
  125. package/dist/domain/page-layout.js.map +1 -0
  126. package/dist/domain/paths.d.ts +5 -0
  127. package/dist/domain/paths.d.ts.map +1 -0
  128. package/dist/domain/paths.js +2 -0
  129. package/dist/domain/paths.js.map +1 -0
  130. package/dist/domain/result.d.ts +23 -0
  131. package/dist/domain/result.d.ts.map +1 -0
  132. package/dist/domain/result.js +10 -0
  133. package/dist/domain/result.js.map +1 -0
  134. package/dist/domain/sha256.d.ts +7 -0
  135. package/dist/domain/sha256.d.ts.map +1 -0
  136. package/dist/domain/sha256.js +14 -0
  137. package/dist/domain/sha256.js.map +1 -0
  138. package/dist/domain/text.d.ts +6 -0
  139. package/dist/domain/text.d.ts.map +1 -0
  140. package/dist/domain/text.js +7 -0
  141. package/dist/domain/text.js.map +1 -0
  142. package/dist/engine.d.ts +93 -0
  143. package/dist/engine.d.ts.map +1 -0
  144. package/dist/engine.js +119 -0
  145. package/dist/engine.js.map +1 -0
  146. package/dist/exit-code.d.ts +22 -0
  147. package/dist/exit-code.d.ts.map +1 -0
  148. package/dist/exit-code.js +10 -0
  149. package/dist/exit-code.js.map +1 -0
  150. package/dist/facts-file.d.ts +96 -0
  151. package/dist/facts-file.d.ts.map +1 -0
  152. package/dist/facts-file.js +134 -0
  153. package/dist/facts-file.js.map +1 -0
  154. package/dist/hooks-settings.d.ts +141 -0
  155. package/dist/hooks-settings.d.ts.map +1 -0
  156. package/dist/hooks-settings.js +306 -0
  157. package/dist/hooks-settings.js.map +1 -0
  158. package/dist/init.d.ts +201 -0
  159. package/dist/init.d.ts.map +1 -0
  160. package/dist/init.js +579 -0
  161. package/dist/init.js.map +1 -0
  162. package/dist/latex-log.d.ts +80 -0
  163. package/dist/latex-log.d.ts.map +1 -0
  164. package/dist/latex-log.js +187 -0
  165. package/dist/latex-log.js.map +1 -0
  166. package/dist/latex-loop.d.ts +129 -0
  167. package/dist/latex-loop.d.ts.map +1 -0
  168. package/dist/latex-loop.js +113 -0
  169. package/dist/latex-loop.js.map +1 -0
  170. package/dist/link-skills.d.ts +51 -0
  171. package/dist/link-skills.d.ts.map +1 -0
  172. package/dist/link-skills.js +199 -0
  173. package/dist/link-skills.js.map +1 -0
  174. package/dist/new-paper.d.ts +48 -0
  175. package/dist/new-paper.d.ts.map +1 -0
  176. package/dist/new-paper.js +110 -0
  177. package/dist/new-paper.js.map +1 -0
  178. package/dist/pdf-facts.d.ts +44 -0
  179. package/dist/pdf-facts.d.ts.map +1 -0
  180. package/dist/pdf-facts.js +239 -0
  181. package/dist/pdf-facts.js.map +1 -0
  182. package/dist/pdf-geometry.d.ts +170 -0
  183. package/dist/pdf-geometry.d.ts.map +1 -0
  184. package/dist/pdf-geometry.js +158 -0
  185. package/dist/pdf-geometry.js.map +1 -0
  186. package/dist/ports/download.d.ts +9 -0
  187. package/dist/ports/download.d.ts.map +1 -0
  188. package/dist/ports/download.js +2 -0
  189. package/dist/ports/download.js.map +1 -0
  190. package/dist/ports/files.d.ts +11 -0
  191. package/dist/ports/files.d.ts.map +1 -0
  192. package/dist/ports/files.js +2 -0
  193. package/dist/ports/files.js.map +1 -0
  194. package/dist/ports/measure-geometry.d.ts +8 -0
  195. package/dist/ports/measure-geometry.d.ts.map +1 -0
  196. package/dist/ports/measure-geometry.js +2 -0
  197. package/dist/ports/measure-geometry.js.map +1 -0
  198. package/dist/ports/process.d.ts +45 -0
  199. package/dist/ports/process.d.ts.map +1 -0
  200. package/dist/ports/process.js +2 -0
  201. package/dist/ports/process.js.map +1 -0
  202. package/dist/ports/tool-installer.d.ts +29 -0
  203. package/dist/ports/tool-installer.d.ts.map +1 -0
  204. package/dist/ports/tool-installer.js +2 -0
  205. package/dist/ports/tool-installer.js.map +1 -0
  206. package/dist/ports/workspace.d.ts +18 -0
  207. package/dist/ports/workspace.d.ts.map +1 -0
  208. package/dist/ports/workspace.js +2 -0
  209. package/dist/ports/workspace.js.map +1 -0
  210. package/dist/rules-config.d.ts +34 -0
  211. package/dist/rules-config.d.ts.map +1 -0
  212. package/dist/rules-config.js +132 -0
  213. package/dist/rules-config.js.map +1 -0
  214. package/dist/structure.d.ts +34 -0
  215. package/dist/structure.d.ts.map +1 -0
  216. package/dist/structure.js +149 -0
  217. package/dist/structure.js.map +1 -0
  218. package/dist/tex-requirements.d.ts +43 -0
  219. package/dist/tex-requirements.d.ts.map +1 -0
  220. package/dist/tex-requirements.js +127 -0
  221. package/dist/tex-requirements.js.map +1 -0
  222. package/dist/toolchain.d.ts +159 -0
  223. package/dist/toolchain.d.ts.map +1 -0
  224. package/dist/toolchain.js +542 -0
  225. package/dist/toolchain.js.map +1 -0
  226. package/dist/types.d.ts +110 -0
  227. package/dist/types.d.ts.map +1 -0
  228. package/dist/types.js +2 -0
  229. package/dist/types.js.map +1 -0
  230. package/docs/configuration.md +235 -0
  231. package/docs/e2e.md +152 -0
  232. package/docs/incidents.md +59 -0
  233. package/docs/install.md +170 -0
  234. package/docs/optional-rules.md +107 -0
  235. package/docs/package-shape-options.md +262 -0
  236. package/docs/prior-art/README.md +76 -0
  237. package/docs/prior-art/blocking-vs-advisory.md +83 -0
  238. package/docs/prior-art/content-delivery.md +124 -0
  239. package/docs/prior-art/multi-mode-tools.md +106 -0
  240. package/docs/prior-art/nondeterministic-checks.md +99 -0
  241. package/docs/prior-art/package-location.md +422 -0
  242. package/docs/prior-art/paper-folder-scaffolding.md +538 -0
  243. package/docs/prior-art/readme-structure.md +69 -0
  244. package/docs/prior-art/repro/README.md +92 -0
  245. package/docs/prior-art/repro/claim1-allowedtools.mjs +66 -0
  246. package/docs/prior-art/repro/claim1-at2.mjs +40 -0
  247. package/docs/prior-art/repro/claim1-crosschannel.mjs +54 -0
  248. package/docs/prior-art/repro/claim1-frontmatter.mjs +76 -0
  249. package/docs/prior-art/repro/claim1-hook-payload-reporter.mjs +10 -0
  250. package/docs/prior-art/repro/claim1-plugin-frontmatter.mjs +27 -0
  251. package/docs/prior-art/repro/claim1-plugin-skill.mjs +52 -0
  252. package/docs/prior-art/repro/claim1-project-skill.mjs +81 -0
  253. package/docs/prior-art/repro/claim2-marketplace-flat-asclaimed.json +1 -0
  254. package/docs/prior-art/repro/claim2-marketplace-negative-control.json +1 -0
  255. package/docs/prior-art/repro/claim2-marketplace-nested-exact.json +9 -0
  256. package/docs/prior-art/repro/claim2-marketplace-nested-noversion.json +9 -0
  257. package/docs/prior-art/repro/claim2-marketplace-nested-range.json +1 -0
  258. package/docs/prior-art/repro/claim3-imports.mjs +50 -0
  259. package/docs/prior-art/repro/claim4-find-package-json.mjs +8 -0
  260. package/docs/prior-art/repro/claim4-package-dir.mjs +39 -0
  261. package/docs/prior-art/repro/claim4-parent-arg.mjs +17 -0
  262. package/docs/prior-art/repro/claim4-resolve-apis.mjs +21 -0
  263. package/docs/prior-art/repro/claim4-setup-consumers.mjs +45 -0
  264. package/docs/prior-art/repro/claim4-yarn-pnp.mjs +70 -0
  265. package/docs/prior-art/repro/claim5-bin-launch.mjs +39 -0
  266. package/docs/prior-art/repro/claim5-exports-mutation.mjs +57 -0
  267. package/docs/prior-art/repro/claim5-resolved-location-and-bin.mjs +33 -0
  268. package/docs/prior-art/repro/claim6-candidate-ambiguity.mjs +17 -0
  269. package/docs/prior-art/repro/claim6-doc-path-candidates.mjs +27 -0
  270. package/docs/prior-art/test-tooling.md +131 -0
  271. package/docs/rules.md +58 -0
  272. package/docs/texlive-install-decision.md +230 -0
  273. package/docs/toolchain.md +152 -0
  274. package/eslint-rules/doc-fields.harness.mjs +336 -0
  275. package/eslint-rules/doc-fields.mjs +186 -0
  276. package/eslint-rules/doc-fields.mutations.mjs +96 -0
  277. package/eslint-rules/install-path-literals.harness.mjs +121 -0
  278. package/eslint-rules/install-path-literals.mjs +108 -0
  279. package/eslint-rules/install-path-literals.mutations.mjs +62 -0
  280. package/eslint-rules/latex-language.harness.mjs +599 -0
  281. package/eslint-rules/latex-language.mjs +591 -0
  282. package/eslint-rules/latex-language.mutations.mjs +196 -0
  283. package/eslint-rules/paper-research-question.harness.mjs +146 -0
  284. package/eslint-rules/paper-research-question.mjs +180 -0
  285. package/eslint-rules/paper-research-question.mutations.mjs +127 -0
  286. package/eslint-rules/paper-stages.harness.mjs +356 -0
  287. package/eslint-rules/paper-stages.mjs +455 -0
  288. package/eslint-rules/paper-stages.mutations.mjs +157 -0
  289. package/eslint-rules/paper-typography.harness.mjs +291 -0
  290. package/eslint-rules/paper-typography.mjs +313 -0
  291. package/eslint-rules/paper-typography.mutations.mjs +131 -0
  292. package/eslint-rules/papers.harness.mjs +259 -0
  293. package/eslint-rules/papers.mjs +166 -0
  294. package/eslint-rules/papers.mutations.mjs +186 -0
  295. package/eslint-rules/pdf-last-page-balance.harness.mjs +206 -0
  296. package/eslint-rules/pdf-last-page-balance.mjs +208 -0
  297. package/eslint-rules/review-findings-cause.harness.mjs +228 -0
  298. package/eslint-rules/review-findings-cause.mjs +135 -0
  299. package/eslint-rules/review-findings-cause.mutations.mjs +72 -0
  300. package/eslint-rules/temp-root-realpath.harness.mjs +176 -0
  301. package/eslint-rules/temp-root-realpath.mjs +129 -0
  302. package/eslint-rules/temp-root-realpath.mutations.mjs +99 -0
  303. package/eslint-rules/tex-build.harness.mjs +753 -0
  304. package/eslint-rules/tex-build.mjs +322 -0
  305. package/eslint-rules/tex-build.mutations.mjs +258 -0
  306. package/eslint.config.mjs +521 -0
  307. package/fixtures/build-e2e/acmart/PIPELINE-STATUS.md +3 -0
  308. package/fixtures/build-e2e/acmart/paper.tex +11 -0
  309. package/fixtures/build-e2e/acmart/venue.json +1 -0
  310. package/fixtures/build-e2e/broken/PIPELINE-STATUS.md +3 -0
  311. package/fixtures/build-e2e/broken/paper.tex +7 -0
  312. package/fixtures/build-e2e/cite/PIPELINE-STATUS.md +3 -0
  313. package/fixtures/build-e2e/cite/build.sh +5 -0
  314. package/fixtures/build-e2e/cite/paper.tex +10 -0
  315. package/fixtures/build-e2e/cite/refs.bib +9 -0
  316. package/fixtures/build-e2e/empty/PIPELINE-STATUS.md +3 -0
  317. package/fixtures/build-e2e/empty/paper.tex +6 -0
  318. package/fixtures/build-e2e/fallback/PIPELINE-STATUS.md +3 -0
  319. package/fixtures/build-e2e/fallback/paper.tex +11 -0
  320. package/fixtures/build-e2e/guards/PIPELINE-STATUS.md +3 -0
  321. package/fixtures/build-e2e/guards/paper.tex +10 -0
  322. package/fixtures/build-e2e/no-source/PIPELINE-STATUS.md +3 -0
  323. package/fixtures/build-e2e/unbalanced/PIPELINE-STATUS.md +3 -0
  324. package/fixtures/build-e2e/unbalanced/paper.tex +28 -0
  325. package/fixtures/build-e2e/unbalanced/refs.bib +269 -0
  326. package/fixtures/install-path-literals/clean.fixture.mjs +3 -0
  327. package/fixtures/install-path-literals/clean.md +15 -0
  328. package/fixtures/install-path-literals/defect.fixture.mjs +3 -0
  329. package/fixtures/install-path-literals/defect.md +14 -0
  330. package/fixtures/latex-language/clean.tex +50 -0
  331. package/fixtures/latex-language/defect.tex +52 -0
  332. package/fixtures/paper-research-question/comment-only/PIPELINE-STATUS.md +9 -0
  333. package/fixtures/paper-research-question/comment-only/paper.tex +7 -0
  334. package/fixtures/paper-research-question/declared-not-in-paper/PIPELINE-STATUS.md +10 -0
  335. package/fixtures/paper-research-question/declared-not-in-paper/paper.tex +6 -0
  336. package/fixtures/paper-research-question/draft/PIPELINE-STATUS.md +6 -0
  337. package/fixtures/paper-research-question/draft/paper.tex +2 -0
  338. package/fixtures/paper-research-question/markdown-no-rq/PIPELINE-STATUS.md +9 -0
  339. package/fixtures/paper-research-question/markdown-no-rq/paper.md +4 -0
  340. package/fixtures/paper-research-question/shipped-no-rq/PIPELINE-STATUS.md +12 -0
  341. package/fixtures/paper-research-question/shipped-no-rq/paper.tex +3 -0
  342. package/fixtures/paper-research-question/shipped-with-rq/PIPELINE-STATUS.md +10 -0
  343. package/fixtures/paper-research-question/shipped-with-rq/paper.tex +2 -0
  344. package/fixtures/paper-stages/authors-ran/PIPELINE-STATUS.md +16 -0
  345. package/fixtures/paper-stages/marker-in-prose/PIPELINE-STATUS.md +17 -0
  346. package/fixtures/paper-stages/nofile/PIPELINE-STATUS.md +8 -0
  347. package/fixtures/paper-stages/noheader/PIPELINE-STATUS.md +1 -0
  348. package/fixtures/paper-stages/noheader/versions/2026-07-22-submitted.pdf +0 -0
  349. package/fixtures/paper-stages/nothing/PIPELINE-STATUS.md +3 -0
  350. package/fixtures/paper-stages/ok/PIPELINE-STATUS.md +9 -0
  351. package/fixtures/paper-stages/ok/versions/2026-07-22-submitted.pdf +0 -0
  352. package/fixtures/paper-stages/stale/PIPELINE-STATUS.md +1 -0
  353. package/fixtures/paper-stages/stale/versions/2026-07-22-submitted.STALE-WRONG-FILE.pdf +0 -0
  354. package/fixtures/paper-stages/twice/PIPELINE-STATUS.md +14 -0
  355. package/fixtures/paper-stages/twice/versions/2026-08-06-submitted.pdf +0 -0
  356. package/fixtures/paper-stages/twice/versions/2026-10-24-submitted.pdf +0 -0
  357. package/fixtures/paper-stages/undeclared/PIPELINE-STATUS.md +8 -0
  358. package/fixtures/paper-stages/undeclared/versions/2026-07-22-submitted.pdf +0 -0
  359. package/fixtures/paper-stages/undeclared/versions/2026-08-29-camera-ready.pdf +0 -0
  360. package/fixtures/paper-stages/wrongsize/PIPELINE-STATUS.md +8 -0
  361. package/fixtures/paper-stages/wrongsize/versions/2026-07-22-submitted.pdf +0 -0
  362. package/fixtures/paper-typography/clean-paper/paper.tex +29 -0
  363. package/fixtures/paper-typography/messy-paper/paper.tex +27 -0
  364. package/fixtures/pdf-facts/README.md +22 -0
  365. package/fixtures/pdf-facts/corrupt-font.pdf +0 -0
  366. package/fixtures/pdf-facts/encrypted.pdf +0 -0
  367. package/fixtures/pdf-facts/hidden-text.pdf +0 -0
  368. package/fixtures/pdf-facts/hidden-text.tex +28 -0
  369. package/fixtures/pdf-facts/t3-all.pdf +0 -0
  370. package/fixtures/pdf-facts/t3-all.tex +8 -0
  371. package/fixtures/pdf-facts/t3-mixed.pdf +0 -0
  372. package/fixtures/pdf-facts/t3-mixed.tex +9 -0
  373. package/fixtures/pdf-facts/ttf.pdf +2240 -1
  374. package/fixtures/pdf-facts/ttf.tex +6 -0
  375. package/fixtures/real-markdown-paper/baseline.json +24 -0
  376. package/fixtures/real-markdown-paper/baseline.mjs +48 -0
  377. package/fixtures/render-paper/build-clean.sh +25 -0
  378. package/fixtures/render-paper/build-defect.sh +15 -0
  379. package/fixtures/review-findings-cause/clean.md +17 -0
  380. package/fixtures/review-findings-cause/defect.md +14 -0
  381. package/fixtures/review-findings-cause/old-debt.md +14 -0
  382. package/fixtures/review-findings-cause/quiet-in-fence.md +16 -0
  383. package/fixtures/tex-build/clean.tex +21 -0
  384. package/fixtures/tex-build/defect.tex +24 -0
  385. package/fixtures/tex-build/frontmatter-clean.tex +25 -0
  386. package/fixtures/tex-build/frontmatter-defect.tex +23 -0
  387. package/fixtures/toolchain-mirror/catalog.txt +5 -0
  388. package/fixtures/toolchain-mirror/install-tl +27 -0
  389. package/fixtures/toolchain-mirror/release-texlive.txt +3 -0
  390. package/fixtures/toolchain-mirror/release-year +1 -0
  391. package/fixtures/toolchain-mirror/stub-kpsewhich +8 -0
  392. package/fixtures/toolchain-mirror/stub-pdflatex +3 -0
  393. package/fixtures/toolchain-mirror/stub-tlmgr +44 -0
  394. package/hooks/hooks.harness.mjs +713 -0
  395. package/hooks/hooks.mutations.mjs +337 -0
  396. package/hooks/paper-edit-guard.hook.d.mts +13 -0
  397. package/hooks/paper-edit-guard.hook.mjs +457 -0
  398. package/hooks/paper-skills-nudge.hook.mjs +136 -0
  399. package/hooks/paper-status-gates.hook.mjs +156 -0
  400. package/hooks/paper-status-gates.sh +91 -0
  401. package/lib/agent-cli-version.harness.mjs +165 -0
  402. package/lib/agent-cli-version.mjs +106 -0
  403. package/lib/agent-cli-version.mutations.mjs +109 -0
  404. package/lib/markdown.mjs +386 -0
  405. package/lib/mutation-driver.harness.mjs +227 -0
  406. package/lib/mutation-driver.mjs +397 -0
  407. package/lib/mutation-driver.mutations.mjs +68 -0
  408. package/lib/paper-config.d.mts +34 -0
  409. package/lib/paper-config.harness.mjs +286 -0
  410. package/lib/paper-config.mjs +142 -0
  411. package/lib/paper-config.mutations.mjs +143 -0
  412. package/lib/skill-checks.mjs +701 -0
  413. package/lib/skill-corpus.mjs +403 -0
  414. package/lib/skill-eval-fixture.mjs +63 -0
  415. package/lib/skill-eval-kit.mjs +257 -0
  416. package/lib/skill-trigger-cases.harness.mjs +170 -0
  417. package/lib/skill-trigger-cases.mjs +446 -0
  418. package/lib/skill-trigger-cases.mutations.mjs +65 -0
  419. package/lib/trigger-ledger.mjs +215 -0
  420. package/package.json +97 -0
  421. package/plugin/.claude-plugin/plugin.json +8 -0
  422. package/plugin/hooks/hooks.json +30 -0
  423. package/scripts/check.harness.mjs +177 -0
  424. package/scripts/check.mjs +239 -0
  425. package/scripts/check.mutations.mjs +110 -0
  426. package/scripts/eslint-report-guard.mjs +82 -0
  427. package/scripts/exclusive.mjs +138 -0
  428. package/scripts/harness-api.frozen.json +76 -0
  429. package/scripts/harness-api.test.ts +175 -0
  430. package/scripts/layer-legacy-frozen.d.mts +28 -0
  431. package/scripts/layer-legacy-frozen.mjs +152 -0
  432. package/scripts/layer-legacy-frozen.test.ts +115 -0
  433. package/scripts/layer-legacy.frozen.json +50 -0
  434. package/scripts/mutation-batteries-frozen.harness.mjs +204 -0
  435. package/scripts/mutation-batteries-frozen.mjs +238 -0
  436. package/scripts/mutation-batteries.frozen.json +117 -0
  437. package/scripts/release-config.test.ts +90 -0
  438. package/scripts/rules-are-content-only.harness.mjs +113 -0
  439. package/scripts/rules-are-content-only.mjs +138 -0
  440. package/scripts/rules-are-content-only.mutations.mjs +81 -0
  441. package/scripts/rules-see-files.harness.mjs +115 -0
  442. package/scripts/rules-see-files.mjs +99 -0
  443. package/scripts/rules-see-files.mutations.mjs +131 -0
  444. package/scripts/run-mutations.mjs +100 -0
  445. package/scripts/semantic-release-plugins.d.ts +16 -0
  446. package/skills/README.md +15 -0
  447. package/skills/analyze-sibling-paper/SKILL.md +170 -0
  448. package/skills/analyze-sibling-paper/SKILL.md.spec.ts +186 -0
  449. package/skills/analyze-sibling-paper/analyze-sibling-paper.eval.mjs +19 -0
  450. package/skills/analyze-sibling-paper/analyze-sibling-paper.harness.mjs +23 -0
  451. package/skills/argument-arc/SKILL.md +177 -0
  452. package/skills/argument-arc/SKILL.md.spec.ts +192 -0
  453. package/skills/argument-arc/argument-arc.eval.mjs +19 -0
  454. package/skills/argument-arc/argument-arc.harness.mjs +23 -0
  455. package/skills/build-benchmark/SKILL.md +213 -0
  456. package/skills/build-benchmark/SKILL.md.spec.ts +220 -0
  457. package/skills/build-benchmark/build-benchmark.eval.mjs +19 -0
  458. package/skills/build-benchmark/build-benchmark.harness.mjs +23 -0
  459. package/skills/build-benchmark/references/adversarial-cold-repro.md +68 -0
  460. package/skills/camera-ready/SKILL.md +148 -0
  461. package/skills/camera-ready/SKILL.md.spec.ts +164 -0
  462. package/skills/camera-ready/camera-ready.eval.mjs +19 -0
  463. package/skills/camera-ready/camera-ready.harness.mjs +23 -0
  464. package/skills/cold-read-diff/SKILL.md +160 -0
  465. package/skills/cold-read-diff/SKILL.md.spec.ts +166 -0
  466. package/skills/cold-read-diff/cold-read-diff.eval.mjs +19 -0
  467. package/skills/cold-read-diff/cold-read-diff.harness.mjs +23 -0
  468. package/skills/draft-paper/SKILL.md +152 -0
  469. package/skills/draft-paper/SKILL.md.spec.ts +169 -0
  470. package/skills/draft-paper/draft-paper.eval.mjs +19 -0
  471. package/skills/draft-paper/draft-paper.harness.mjs +23 -0
  472. package/skills/extend-paper/SKILL.md +99 -0
  473. package/skills/extend-paper/SKILL.md.spec.ts +116 -0
  474. package/skills/extend-paper/extend-paper.eval.mjs +19 -0
  475. package/skills/extend-paper/extend-paper.harness.mjs +23 -0
  476. package/skills/find-venue/SKILL.md +128 -0
  477. package/skills/find-venue/SKILL.md.spec.ts +145 -0
  478. package/skills/find-venue/find-venue.eval.mjs +19 -0
  479. package/skills/find-venue/find-venue.harness.mjs +23 -0
  480. package/skills/grade-paper-writing/SKILL.md +436 -0
  481. package/skills/grade-paper-writing/SKILL.md.spec.ts +453 -0
  482. package/skills/grade-paper-writing/fixtures/control_gopen.txt +1 -0
  483. package/skills/grade-paper-writing/fixtures/control_human_paper.txt +1 -0
  484. package/skills/grade-paper-writing/fixtures/rewrite.txt +1 -0
  485. package/skills/grade-paper-writing/fixtures/specimen.txt +1 -0
  486. package/skills/grade-paper-writing/fixtures/structure-checks.md +22 -0
  487. package/skills/grade-paper-writing/grade-paper-writing.eval.mjs +19 -0
  488. package/skills/grade-paper-writing/grade-paper-writing.harness.mjs +23 -0
  489. package/skills/grade-paper-writing/prose-lint.mjs +713 -0
  490. package/skills/harden-paper/SKILL.md +318 -0
  491. package/skills/harden-paper/SKILL.md.spec.ts +336 -0
  492. package/skills/harden-paper/check-numbers.sh +33 -0
  493. package/skills/harden-paper/check-release-claims.sh +35 -0
  494. package/skills/harden-paper/fixtures/uncited-assertions-sample.md +43 -0
  495. package/skills/harden-paper/fixtures/uncited-assertions-sample.tex +77 -0
  496. package/skills/harden-paper/harden-paper.eval.mjs +19 -0
  497. package/skills/harden-paper/harden-paper.harness.mjs +23 -0
  498. package/skills/map-prior-work/SKILL.md +211 -0
  499. package/skills/map-prior-work/SKILL.md.spec.ts +227 -0
  500. package/skills/map-prior-work/map-prior-work.eval.mjs +19 -0
  501. package/skills/map-prior-work/map-prior-work.harness.mjs +23 -0
  502. package/skills/osf-artifact-upload/SKILL.md +52 -0
  503. package/skills/osf-artifact-upload/SKILL.md.spec.ts +59 -0
  504. package/skills/osf-artifact-upload/osf-artifact-upload.eval.mjs +22 -0
  505. package/skills/osf-artifact-upload/osf-artifact-upload.harness.mjs +103 -0
  506. package/skills/paper-adversarial-review/SKILL.md +126 -0
  507. package/skills/paper-adversarial-review/SKILL.md.spec.ts +142 -0
  508. package/skills/paper-adversarial-review/paper-adversarial-review.eval.mjs +19 -0
  509. package/skills/paper-adversarial-review/paper-adversarial-review.harness.mjs +23 -0
  510. package/skills/paper-pipeline/PIPELINE-MAP.md +371 -0
  511. package/skills/paper-pipeline/SKILL.md +499 -0
  512. package/skills/paper-pipeline/SKILL.md.spec.ts +517 -0
  513. package/skills/paper-pipeline/description-language.eval.mjs +347 -0
  514. package/skills/paper-pipeline/framing-vs-vocabulary.eval.mjs +891 -0
  515. package/skills/paper-pipeline/grade-paper-writing-ablation.eval.mjs +1254 -0
  516. package/skills/paper-pipeline/paper-pipeline.eval.mjs +22 -0
  517. package/skills/paper-pipeline/paper-pipeline.harness.mjs +143 -0
  518. package/skills/paper-pipeline/pipeline-firing.baseline.json +270 -0
  519. package/skills/paper-pipeline/pipeline-firing.eval.mjs +664 -0
  520. package/skills/paper-pipeline/pipeline-language.eval.mjs +672 -0
  521. package/skills/paper-pipeline/references/acceptance-gate.md +329 -0
  522. package/skills/paper-pipeline/references/acl-venue-rules.md +142 -0
  523. package/skills/paper-pipeline/references/anonymization.md +68 -0
  524. package/skills/paper-pipeline/references/artifact-checklist.md +93 -0
  525. package/skills/paper-pipeline/references/body-vs-appendix.md +97 -0
  526. package/skills/paper-pipeline/references/credit-criteria.md +69 -0
  527. package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/README.md +35 -0
  528. package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/run_retext.mjs +24 -0
  529. package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/sentences.txt +11 -0
  530. package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/test_sentences.py +25 -0
  531. package/skills/paper-pipeline/references/occupancy-2026-08-06-prose-checkers.md +538 -0
  532. package/skills/paper-pipeline/references/occupancy-2026-08-06-reproducible-tooling.md +431 -0
  533. package/skills/paper-pipeline/references/occupancy-2026-08-06-staleness-and-orchestration.md +592 -0
  534. package/skills/paper-pipeline/references/pipeline-status-template.md +162 -0
  535. package/skills/paper-pipeline/references/review-ratchet.md +36 -0
  536. package/skills/paper-pipeline/references/sweep-2026-08-09-ideal-pipeline.md +585 -0
  537. package/skills/paper-pipeline/references/writing-craft.md +448 -0
  538. package/skills/paper-pipeline/repro/2026-08-07-description-language-control.log +63 -0
  539. package/skills/paper-pipeline/repro/2026-08-07-fork-check.log +52 -0
  540. package/skills/paper-pipeline/repro/2026-08-07-fork-check2.log +33 -0
  541. package/skills/paper-pipeline/repro/2026-08-07-language-eval-pilot.log +33 -0
  542. package/skills/paper-pipeline/repro/2026-08-07-language-eval-raw.log +166 -0
  543. package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-oracle.json +338 -0
  544. package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-oracle.log +118 -0
  545. package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-raw.log +245 -0
  546. package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary.json +776 -0
  547. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-A6-oracle.log +53 -0
  548. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-A6-raw.log +89 -0
  549. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-oracle.json +450 -0
  550. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-oracle.log +136 -0
  551. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-raw.log +242 -0
  552. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-setupdiff.log +59 -0
  553. package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation.json +1032 -0
  554. package/skills/paper-pipeline/repro/2026-08-08-parent-replication-gpw.json +139 -0
  555. package/skills/paper-pipeline/repro/2026-08-08-parent-replication-gpw.log +98 -0
  556. package/skills/paper-pipeline/repro/2026-08-08-parent-replication.mjs +92 -0
  557. package/skills/paper-pipeline/repro/README.md +129 -0
  558. package/skills/paper-pipeline/repro/analyze-language-eval.py +116 -0
  559. package/skills/paper-pipeline/scripts/README.md +344 -0
  560. package/skills/paper-pipeline/scripts/announce.mjs +67 -0
  561. package/skills/paper-pipeline/scripts/artifact-coverage.harness.mjs +496 -0
  562. package/skills/paper-pipeline/scripts/artifact-coverage.mjs +397 -0
  563. package/skills/paper-pipeline/scripts/artifact-coverage.mutations.mjs +218 -0
  564. package/skills/paper-pipeline/scripts/check-provenance.mjs +184 -0
  565. package/skills/paper-pipeline/scripts/consumer.d.mts +32 -0
  566. package/skills/paper-pipeline/scripts/consumer.harness.mjs +562 -0
  567. package/skills/paper-pipeline/scripts/consumer.mjs +535 -0
  568. package/skills/paper-pipeline/scripts/consumer.mutations.mjs +190 -0
  569. package/skills/paper-pipeline/scripts/extract-ref-facts.harness.mjs +457 -0
  570. package/skills/paper-pipeline/scripts/extract-ref-facts.mjs +656 -0
  571. package/skills/paper-pipeline/scripts/extract-ref-facts.mutations.mjs +54 -0
  572. package/skills/paper-pipeline/scripts/fixtures/clean/PIPELINE-STATUS.md +51 -0
  573. package/skills/paper-pipeline/scripts/fixtures/dirty/PIPELINE-STATUS.md +52 -0
  574. package/skills/paper-pipeline/scripts/fixtures/dirty/paper.md +6 -0
  575. package/skills/paper-pipeline/scripts/fixtures/real-bib/refs.bib +153 -0
  576. package/skills/paper-pipeline/scripts/generated-code.harness.mjs +466 -0
  577. package/skills/paper-pipeline/scripts/generated-code.mjs +338 -0
  578. package/skills/paper-pipeline/scripts/generated-code.mutations.mjs +254 -0
  579. package/skills/paper-pipeline/scripts/ledger.mjs +623 -0
  580. package/skills/paper-pipeline/scripts/ledger.selftest.mjs +286 -0
  581. package/skills/paper-pipeline/scripts/pipeline-check.harness.mjs +389 -0
  582. package/skills/paper-pipeline/scripts/pipeline-check.mjs +737 -0
  583. package/skills/paper-pipeline/scripts/pipeline-check.mutations.mjs +54 -0
  584. package/skills/paper-pipeline/scripts/pipeline-edges.mjs +169 -0
  585. package/skills/paper-pipeline/scripts/population-map.harness.mjs +178 -0
  586. package/skills/paper-pipeline/scripts/population-map.mjs +181 -0
  587. package/skills/paper-pipeline/scripts/population-map.mutations.mjs +65 -0
  588. package/skills/paper-pipeline/scripts/population-map.selftest.mjs +122 -0
  589. package/skills/paper-pipeline/scripts/provenance.harness.mjs +240 -0
  590. package/skills/paper-pipeline/scripts/provenance.mutations.mjs +59 -0
  591. package/skills/paper-pipeline/scripts/round-diff.harness.mjs +881 -0
  592. package/skills/paper-pipeline/scripts/round-diff.mjs +576 -0
  593. package/skills/paper-pipeline/scripts/round-diff.mutations.mjs +276 -0
  594. package/skills/paper-pipeline/scripts/run-mechanical.mjs +633 -0
  595. package/skills/paper-pipeline/scripts/status.mjs +295 -0
  596. package/skills/paper-status/SKILL.md +183 -0
  597. package/skills/paper-status/SKILL.md.spec.ts +190 -0
  598. package/skills/paper-status/paper-status.eval.mjs +22 -0
  599. package/skills/paper-status/paper-status.harness.mjs +25 -0
  600. package/skills/pc-panel-review/SKILL.md +263 -0
  601. package/skills/pc-panel-review/SKILL.md.spec.ts +280 -0
  602. package/skills/pc-panel-review/pc-panel-review.eval.mjs +19 -0
  603. package/skills/pc-panel-review/pc-panel-review.harness.mjs +23 -0
  604. package/skills/plan-paper-timeline/SKILL.md +182 -0
  605. package/skills/plan-paper-timeline/SKILL.md.spec.ts +200 -0
  606. package/skills/plan-paper-timeline/fixtures/fake-google-calendar.mjs +239 -0
  607. package/skills/plan-paper-timeline/plan-paper-timeline.effects.harness.mjs +431 -0
  608. package/skills/plan-paper-timeline/plan-paper-timeline.effects.mutations.mjs +65 -0
  609. package/skills/plan-paper-timeline/plan-paper-timeline.eval.mjs +19 -0
  610. package/skills/plan-paper-timeline/plan-paper-timeline.harness.mjs +23 -0
  611. package/skills/render-paper/SKILL.md +159 -0
  612. package/skills/render-paper/SKILL.md.spec.ts +166 -0
  613. package/skills/render-paper/check-render.sh +419 -0
  614. package/skills/render-paper/checkers-requirements.txt +55 -0
  615. package/skills/render-paper/ensure-checkers.sh +69 -0
  616. package/skills/render-paper/extract-pdf-facts.harness.mjs +166 -0
  617. package/skills/render-paper/extract-pdf-facts.mjs +144 -0
  618. package/skills/render-paper/render-paper.eval.mjs +19 -0
  619. package/skills/render-paper/render-paper.harness.mjs +339 -0
  620. package/skills/research-ideate/SKILL.md +136 -0
  621. package/skills/research-ideate/SKILL.md.spec.ts +152 -0
  622. package/skills/research-ideate/research-ideate.eval.mjs +19 -0
  623. package/skills/research-ideate/research-ideate.harness.mjs +23 -0
  624. package/skills/skill-contract.mutations.mjs +179 -0
  625. package/skills/study-accepted-papers/SKILL.md +206 -0
  626. package/skills/study-accepted-papers/SKILL.md.spec.ts +223 -0
  627. package/skills/study-accepted-papers/study-accepted-papers.eval.mjs +19 -0
  628. package/skills/study-accepted-papers/study-accepted-papers.harness.mjs +23 -0
  629. package/skills/submit-paper/SKILL.md +182 -0
  630. package/skills/submit-paper/SKILL.md.spec.ts +199 -0
  631. package/skills/submit-paper/check-deanon.sh +149 -0
  632. package/skills/submit-paper/references/publishers/acm.md +92 -0
  633. package/skills/submit-paper/references/venues/agenticdev.jsonc +108 -0
  634. package/skills/submit-paper/references/venues/agenticdev.md +139 -0
  635. package/skills/submit-paper/references/venues/agenticdev.tex +19 -0
  636. package/skills/submit-paper/references/venues/aisec.jsonc +101 -0
  637. package/skills/submit-paper/references/venues/aisec.md +105 -0
  638. package/skills/submit-paper/references/venues/paper-guards.tex +41 -0
  639. package/skills/submit-paper/references/venues/realm.jsonc +81 -0
  640. package/skills/submit-paper/references/venues/realm.md +155 -0
  641. package/skills/submit-paper/references/venues/tex-base.jsonc +50 -0
  642. package/skills/submit-paper/references/venues/venue-profile.schema.json +74 -0
  643. package/skills/submit-paper/submit-paper.eval.mjs +19 -0
  644. package/skills/submit-paper/submit-paper.harness.mjs +23 -0
  645. package/skills/sweep-design-space/SKILL.md +269 -0
  646. package/skills/sweep-design-space/SKILL.md.spec.ts +285 -0
  647. package/skills/sweep-design-space/sweep-design-space.eval.mjs +19 -0
  648. package/skills/sweep-design-space/sweep-design-space.harness.mjs +23 -0
  649. package/skills/tighten-paper/SKILL.md +368 -0
  650. package/skills/tighten-paper/SKILL.md.spec.ts +384 -0
  651. package/skills/tighten-paper/structure.mjs +371 -0
  652. package/skills/tighten-paper/tighten-paper.eval.mjs +19 -0
  653. package/skills/tighten-paper/tighten-paper.harness.mjs +23 -0
  654. package/skills/verify-citations/SKILL.md +328 -0
  655. package/skills/verify-citations/SKILL.md.spec.ts +345 -0
  656. package/skills/verify-citations/scripts/bib-authors.mjs +479 -0
  657. package/skills/verify-citations/scripts/bib-authors.test.mjs +175 -0
  658. package/skills/verify-citations/scripts/verify-cites.mjs +1108 -0
  659. package/skills/verify-citations/scripts/verify-cites.test.mjs +735 -0
  660. package/skills/verify-citations/verify-citations.eval.mjs +19 -0
  661. package/skills/verify-citations/verify-citations.harness.mjs +23 -0
  662. package/src/CLAUDE.md +51 -0
  663. package/src/action-ref.test.ts +26 -0
  664. package/src/action-ref.ts +15 -0
  665. package/src/adapters/banal/failure.test.ts +63 -0
  666. package/src/adapters/banal/failure.ts +118 -0
  667. package/src/adapters/banal/index.test.ts +119 -0
  668. package/src/adapters/banal/index.ts +100 -0
  669. package/src/adapters/banal/install.test.ts +20 -0
  670. package/src/adapters/banal/install.ts +41 -0
  671. package/src/adapters/banal/invocation.test.ts +74 -0
  672. package/src/adapters/banal/invocation.ts +95 -0
  673. package/src/adapters/banal/locate.test.ts +52 -0
  674. package/src/adapters/banal/locate.ts +84 -0
  675. package/src/adapters/banal/output.test.ts +140 -0
  676. package/src/adapters/banal/output.ts +141 -0
  677. package/src/adapters/banal/pin.ts +30 -0
  678. package/src/adapters/banal/probe.ts +35 -0
  679. package/src/adapters/banal/run.test.ts +191 -0
  680. package/src/adapters/banal/run.ts +244 -0
  681. package/src/adapters/banal/settings.test.ts +31 -0
  682. package/src/adapters/banal/settings.ts +55 -0
  683. package/src/adapters/banal/xml.test.ts +111 -0
  684. package/src/adapters/banal/xml.ts +112 -0
  685. package/src/adapters/curl/download.io.ts +73 -0
  686. package/src/adapters/curl/download.test.ts +55 -0
  687. package/src/adapters/curl/index.ts +5 -0
  688. package/src/adapters/memory/index.ts +131 -0
  689. package/src/adapters/node/files.io.ts +39 -0
  690. package/src/adapters/node/files.test.ts +28 -0
  691. package/src/adapters/node/host.io.ts +15 -0
  692. package/src/adapters/node/index.ts +36 -0
  693. package/src/adapters/node/process.io.ts +49 -0
  694. package/src/adapters/node/process.test.ts +46 -0
  695. package/src/adapters/node/workspace.io.ts +40 -0
  696. package/src/adapters/node/workspace.test.ts +58 -0
  697. package/src/adapters/pdfjs/fill.test.ts +111 -0
  698. package/src/adapters/pdfjs/fill.ts +141 -0
  699. package/src/build-engine.harness.mjs +314 -0
  700. package/src/build-engine.ts +219 -0
  701. package/src/build.harness.mjs +631 -0
  702. package/src/build.mutations.mjs +195 -0
  703. package/src/build.ts +793 -0
  704. package/src/cli.harness.mjs +2007 -0
  705. package/src/cli.mutations.mjs +448 -0
  706. package/src/cli.ts +1189 -0
  707. package/src/doctor.harness.mjs +396 -0
  708. package/src/doctor.mutations.mjs +175 -0
  709. package/src/doctor.ts +356 -0
  710. package/src/domain/geometry.ts +108 -0
  711. package/src/domain/host.ts +23 -0
  712. package/src/domain/page-layout.ts +32 -0
  713. package/src/domain/paths.ts +5 -0
  714. package/src/domain/result.test.ts +26 -0
  715. package/src/domain/result.ts +29 -0
  716. package/src/domain/sha256.test.ts +12 -0
  717. package/src/domain/sha256.ts +21 -0
  718. package/src/domain/text.ts +11 -0
  719. package/src/engine.harness.mjs +252 -0
  720. package/src/engine.ts +176 -0
  721. package/src/exit-code.test.ts +21 -0
  722. package/src/exit-code.ts +38 -0
  723. package/src/facts-file.test.ts +240 -0
  724. package/src/facts-file.ts +241 -0
  725. package/src/hooks-settings.harness.mjs +386 -0
  726. package/src/hooks-settings.mutations.mjs +116 -0
  727. package/src/hooks-settings.ts +434 -0
  728. package/src/init.ts +900 -0
  729. package/src/latex-log.harness.mjs +226 -0
  730. package/src/latex-log.ts +234 -0
  731. package/src/latex-loop.harness.mjs +449 -0
  732. package/src/latex-loop.ts +211 -0
  733. package/src/link-skills.harness.mjs +273 -0
  734. package/src/link-skills.mutations.mjs +136 -0
  735. package/src/link-skills.ts +258 -0
  736. package/src/new-paper.harness.mjs +216 -0
  737. package/src/new-paper.mutations.mjs +79 -0
  738. package/src/new-paper.ts +158 -0
  739. package/src/pdf-facts.harness.mjs +188 -0
  740. package/src/pdf-facts.ts +327 -0
  741. package/src/pdf-geometry.harness.mjs +254 -0
  742. package/src/pdf-geometry.ts +300 -0
  743. package/src/ports/download.ts +10 -0
  744. package/src/ports/files.ts +11 -0
  745. package/src/ports/measure-geometry.ts +8 -0
  746. package/src/ports/process.ts +46 -0
  747. package/src/ports/tool-installer.ts +33 -0
  748. package/src/ports/workspace.ts +20 -0
  749. package/src/rules-config.harness.mjs +114 -0
  750. package/src/rules-config.ts +178 -0
  751. package/src/structure.harness.mjs +179 -0
  752. package/src/structure.mutations.mjs +83 -0
  753. package/src/structure.ts +166 -0
  754. package/src/tex-requirements.harness.mjs +238 -0
  755. package/src/tex-requirements.ts +181 -0
  756. package/src/toolchain.harness.mjs +651 -0
  757. package/src/toolchain.ts +755 -0
  758. package/src/types.ts +106 -0
  759. package/templates/paper/PIPELINE-STATUS.md +72 -0
  760. package/templates/paper/paper.md +4 -0
  761. package/templates/paper/paper.tex +8 -0
  762. package/tsconfig.json +23 -0
@@ -0,0 +1,664 @@
1
+ /**
2
+ * pipeline-firing.eval.mjs — do the paper-pipeline skills actually FIRE when they should?
3
+ *
4
+ * Run: node .claude/skills/paper-pipeline/pipeline-firing.eval.mjs [flags]
5
+ * --only <skill> run one case (repeatable: --only tighten-paper --only argument-arc)
6
+ * --trials N trials per prompt (default 1)
7
+ * --concurrency N parallel runs (default 3)
8
+ * --strict count a run as fired only if the COLLIDING skill stayed silent
9
+ * --update-baseline record this run as the committed baseline
10
+ * --no-gate report only; do not throw on a threshold breach
11
+ *
12
+ * ─────────────────────────────────────────────────────────────────────────────
13
+ * WHY THIS FILE EXISTS, AND WHY IT IS NOT A HARNESS TEST
14
+ *
15
+ * The owner's complaint: "we lack observability into pipeline status and whether skills are
16
+ * actually being called." `.claude/hooks/hooks.harness.mjs` and `.claude/skills/paper-pipeline/scripts/gates.harness.mjs`
17
+ * test machinery that was ALREADY invoked — they feed a hook an event and check the verdict. No
18
+ * deterministic test can answer the question above, because whether a skill is called is a
19
+ * property of a real model reading 37 competing descriptions and picking one. That is a
20
+ * measurement, it needs the real CLI, and it costs money.
21
+ *
22
+ * 🔴 NOT NOVEL, AND SAYING SO IS THE POINT. `vigiles/s47.md` records the prior art in this exact
23
+ * slot: `adewale/skill-eval-harness` (MIT, 53★, v0.4.2) drives the real `claude`/`codex` binaries
24
+ * and reports an autonomous-trigger-rate MATRIX split by should-fire / should-not-fire — i.e.
25
+ * recall AND precision, empirically, against the real harness. Scott Spence published a real
26
+ * sandboxed activation study against `claude -p` in Feb 2026. AWS `sample-agent-skill-eval` scores
27
+ * a 20%-weighted "Trigger" component. This file measures trigger rate; it does not invent the idea
28
+ * of measuring trigger rate, and nothing built on it should be written up as if it did.
29
+ *
30
+ * ─────────────────────────────────────────────────────────────────────────────
31
+ * 🔴 THE THREE WAYS `hooks.harness.mjs` LIED, AND WHERE EACH ONE LANDS HERE
32
+ *
33
+ * 1. "Assertions must run at MODULE TOP LEVEL — an exported `tests` object ran nothing and the
34
+ * runner printed ✓." Same trap, worse: an eval that silently measured nothing still prints a
35
+ * percentage. Everything here runs at top level under a top-level `await`, and the gate
36
+ * `assertTriggerRate` throws from top level. There is no exported entry point to forget to call.
37
+ *
38
+ * 2. "A probe built on `execFileSync` reported all three react hooks DEAD; the probe could not
39
+ * see stderr. A checker that can only report failure is worth less than no checker." The eval
40
+ * version of that bug is a `fired` predicate that can only return false — a wrong skill id, a
41
+ * wrong plugin namespace, and every rate is 0.00 while the file looks fine. Guarded two ways:
42
+ * `assertSkillIdsExist` fails loudly at startup if a case names a skill that is not installed,
43
+ * and a run where EVERY case scores 0.00 is reported as a suspected harness fault, not as a
44
+ * finding. (Verified live: a smoke run scored 0.50, so the predicate can return true.)
45
+ *
46
+ * 3. "`touches(['<papers-root>/'])` — a trailing slash never matched, and the guard was WEAKER
47
+ * than the grep it replaced while its header claimed the opposite." The analogue here is
48
+ * `EvalArm.plugin` vs `pluginDir`: `plugin` materialises a file subset that does NOT register
49
+ * skills, so a run using it would measure a model that cannot fire a skill at all and would
50
+ * report 0% as though it were news. This file never uses `plugin`. See the `skillsDir` note.
51
+ *
52
+ * ─────────────────────────────────────────────────────────────────────────────
53
+ * WHY `skillsDir` AND NOT `pluginDir`
54
+ *
55
+ * Both install NATIVELY (`claude --plugin-dir`), which is the whole point: the real model triggers
56
+ * a skill by its description. `pluginDir` wants a COMPLETE plugin (a `.claude-plugin/plugin.json`);
57
+ * `.claude/skills` is a loose skills directory, so the correct field is `skillsDir`, which vigiles
58
+ * packages into a throwaway `--plugin-dir` install for us (`packageSkillsDir`) and removes after.
59
+ * Verified: the run reports `whole-harness: measured against 36 competing skill(s)`. The installed
60
+ * namespace for a loose dir is `vigiles-loose-skills`, hence the skill ids below.
61
+ *
62
+ * The 36 competitors matter. An ISOLATED trigger rate (one skill, nothing to compete with)
63
+ * OVERSTATES recall and UNDERSTATES false positives, because selection is competitive and Claude
64
+ * Code evicts unused descriptions under a context budget. This eval is the whole-harness tier by
65
+ * construction — every skill in the repo is installed on every run.
66
+ *
67
+ * ─────────────────────────────────────────────────────────────────────────────
68
+ * 🔴 API MISMATCH vs. THE BRIEF: `interceptTools` IS NOT ON `TriggerRateSpec`
69
+ *
70
+ * `EvalArm.interceptTools` and `MeasureSpec.interceptTools` exist (`runEval` / `measure`).
71
+ * `TriggerRateSpec` — checked against `node_modules/vigiles/dist/eval.d.ts` — has no such field, so
72
+ * passing one would be silently ignored by an .mjs file and would read as a safety measure that is
73
+ * not there. It is deliberately absent below. Safety comes from three real properties instead:
74
+ *
75
+ * - `stubSkillBodies` defaults TRUE for trigger runs. Every SKILL.md is rewritten to frontmatter
76
+ * plus a no-op body before install, so a fired skill stops AT selection. It cannot spawn the
77
+ * paid subagent panels that `pc-panel-review` and `grade-paper-writing` open with. This is not
78
+ * a compromise: selection happens from name+description alone, before a body is ever loaded, so
79
+ * stubbing cannot change what is measured.
80
+ * - `allowedTools: ["Skill", "Read"]` — no Write, no Edit, no Bash, no Task. The model cannot
81
+ * write to a paper, push, or spawn a subagent even if it wanted to.
82
+ * - every run executes in a fresh throwaway cwd seeded only with FIXTURE below. The real
83
+ * papers tree is never in scope.
84
+ *
85
+ * What `interceptTools` would have added — recording a blocked ATTEMPT so its arguments land in the
86
+ * trace — is not needed here: the question is which Skill was selected, and that call is not
87
+ * blocked. And per its own doc, `interceptTools` prevents side effects only; it does NOT reduce
88
+ * model-call cost. Neither does anything else here. This eval costs real money on every run.
89
+ *
90
+ * ─────────────────────────────────────────────────────────────────────────────
91
+ * WHAT IT COSTS, MEASURED NOT GUESSED
92
+ *
93
+ * Smoke run 2026-08-07, 2 prompts, sonnet, whole-harness: ~$0.13 API-equivalent per run,
94
+ * ~12.6 s each, ~110k tokens per run of which ~95k came from prompt cache. The default grid is
95
+ * 8 cases × (4 should-fire + 4 should-not-fire) = 64 runs ≈ $8 API-equivalent, ~5 min at
96
+ * concurrency 3. That is why this is NOT wired into the push-triggered CI job — see the
97
+ * `skill-firing` job in .github/workflows/paper-gates.yml, which is `workflow_dispatch` + weekly.
98
+ *
99
+ * ─────────────────────────────────────────────────────────────────────────────
100
+ * ✅ API MISMATCH #3 IS CLOSED — kept because the reasoning was right and the fact expired
101
+ *
102
+ * It USED to read: a bare `npx vigiles eval` prints "No …eval.{mjs,…} files found", because the
103
+ * discovery glob did not descend into dot-directories and `.claude/` is where a Claude Code eval
104
+ * lives by definition. Fixed upstream (`dot: true`) and verified here 2026-08-08: a bare run now
105
+ * matches 5 eval files across the tree and refuses to fire them non-interactively — a consent
106
+ * gate, not a discovery failure. `vigiles audit` can therefore see this file, and the two
107
+ * consequences that used to follow no longer do.
108
+ *
109
+ * The note stays because the LESSON outlives the bug, and because it is the reason nothing here
110
+ * claimed a `Tested` improvement it had not earned: a metric counts what its own discovery can
111
+ * see, so "we wrote the eval" and "the score moved" are different statements. Asserting the
112
+ * second from the first is the "fixed the symptom past the measurement" defect that
113
+ * `compile-rules-2026` is about.
114
+ *
115
+ * ⚠️ Still run it as `node <path>` rather than `vigiles eval <path>` — not for discovery now, but
116
+ * because this file parses its own `--trials` and the CI job passes it.
117
+ *
118
+ * ─────────────────────────────────────────────────────────────────────────────
119
+ * FIRST RUN — 2026-08-07, sonnet, 64 runs, 1 trial/prompt, 36 competitors, ~$9.55 API-equivalent
120
+ * (billed to a Claude subscription, $0 metered). Recorded in pipeline-firing.baseline.json.
121
+ *
122
+ * skill recall FP-rate precision n
123
+ * tighten-paper 75% 0% 100% 4
124
+ * grade-paper-writing 50% 0% 100% 4
125
+ * argument-arc 75% 0% 100% 4
126
+ * paper-adversarial-review 75% 0% 100% 4
127
+ * pc-panel-review 75% 0% 100% 4
128
+ * cold-read-diff 50% 0% 100% 4
129
+ * verify-citations 100% 0% 100% 4
130
+ * map-prior-work 75% 0% 100% 4
131
+ *
132
+ * TWO THINGS THIS SAYS, and one it does not.
133
+ *
134
+ * 1. ROUTING IS NOT THE PROBLEM. Every colliding pair scored 0% false positives and 100%
135
+ * precision. `tighten-paper` never fired on a prose-craft prompt; `pc-panel-review` never fired
136
+ * on "one hostile reviewer"; `verify-citations` never fired on "who already did this". The
137
+ * "which skill?" callouts those SKILL.md files open with are doing their job. This was the
138
+ * hypothesis the case list was built to test, and it came back negative — the skills do not
139
+ * steal each other's work.
140
+ *
141
+ * 2. RECALL IS THE PROBLEM. 25 of 32 should-fire prompts fired: seven of eight skills miss at
142
+ * least one prompt a person would really type, and only `verify-citations` fired every time
143
+ * (pass^k = 1). A skill that fires 50-75% of the time is not broken, but it is also not the
144
+ * thing the pipeline docs assume when they say "run tighten-paper first".
145
+ *
146
+ * 3. 🔴 A HYPOTHESIS, EXPLICITLY NOT A RESULT — LANGUAGE. Eight of the nine misses were
147
+ * Russian-language prompts. Recall splits 10/18 (56%) on Russian against 13/14 (93%) on
148
+ * English; Fisher exact two-sided p = 0.044. That would matter a lot here, because the owner
149
+ * types Russian constantly and every skill description is written in English.
150
+ *
151
+ * Do NOT act on this number yet, for three reasons that are not hedging:
152
+ * - POST HOC. The split was noticed in the output, not designed for. A pattern found by
153
+ * looking at 32 outcomes and picking the one that stands out is worth p ≈ 0.044 much less
154
+ * than a pattern predicted in advance.
155
+ * - CONFOUNDED. The Russian prompts are not translations of the English ones — they are
156
+ * different prompts asking different things. Language is entangled with content, so the
157
+ * effect could be "these particular four questions are harder", not "Russian".
158
+ * - ONE TRIAL. At one trial per prompt each cell is a coin flip observed once. `tighten-paper`
159
+ * already flipped: a smoke run the same morning scored the "14 pages against a 9 page limit"
160
+ * prompt 0.00, the full run scored it 1.00.
161
+ *
162
+ * THE DESIGNED VERSION, if this is worth settling: take the 14 English prompts, translate each
163
+ * to Russian, and run both sets at --trials 5. That is a matched pair — same content, one
164
+ * variable — and it costs about the same as one full run above. Until then this is a lead.
165
+ */
166
+
167
+ import {
168
+ assertTriggerRate,
169
+ assertPromptDiversity,
170
+ checkPromptDiversity,
171
+ formatTriggerRateReport,
172
+ skillResolved,
173
+ readBaseline,
174
+ writeBaseline,
175
+ diffReports,
176
+ formatBaselineDiff,
177
+ assertNoRegression,
178
+ skip,
179
+ } from "vigiles";
180
+ import { paid_measureTriggerRate as measureTriggerRate } from "vigiles/eval";
181
+ import { existsSync, readFileSync } from "node:fs";
182
+ import { join } from "node:path";
183
+ import { frontmatterBlock } from "../../lib/markdown.mjs";
184
+ import { parseFm } from "../../lib/skill-corpus.mjs";
185
+ import { installedSkills } from "./scripts/consumer.mjs";
186
+
187
+ // The name a SKILL.md DECLARES, parsed rather than matched. The old expression took
188
+ // `(\S+)` after `name:`, which silently truncates a quoted name and cannot see one
189
+ // carried onto a continuation line — and this is a guard whose whole job is to fail
190
+ // when the declared name disagrees with the directory.
191
+ const declaredName = (md) => {
192
+ const block = frontmatterBlock(md);
193
+ if (block === null) return undefined;
194
+ const v = parseFm(block, "skill fixture").name;
195
+ return typeof v === "string" ? v : undefined;
196
+ };
197
+
198
+ import { execFileSync } from "node:child_process";
199
+
200
+ const ROOT = process.env.CLAUDE_PROJECT_DIR ?? process.cwd();
201
+ const SKILLS_DIR = join(ROOT, ".claude", "skills");
202
+ const BASELINE = join(
203
+ SKILLS_DIR,
204
+ "paper-pipeline",
205
+ "pipeline-firing.baseline.json",
206
+ );
207
+
208
+ /** The namespace `packageSkillsDir` installs a LOOSE skills dir under. Not a guess — see eval.js. */
209
+ const NS = "vigiles-loose-skills";
210
+ const id = (skill) => `${NS}:${skill}`;
211
+
212
+ // ── flags ────────────────────────────────────────────────────────────────────
213
+ const argv = process.argv.slice(2);
214
+ const flag = (name) => argv.includes(`--${name}`);
215
+ const val = (name, dflt) => {
216
+ const i = argv.indexOf(`--${name}`);
217
+ return i >= 0 && argv[i + 1] ? argv[i + 1] : dflt;
218
+ };
219
+ const onlys = argv.reduce(
220
+ (acc, a, i) => (a === "--only" && argv[i + 1] ? [...acc, argv[i + 1]] : acc),
221
+ [],
222
+ );
223
+ const TRIALS = Number(val("trials", "1"));
224
+ const CONCURRENCY = Number(val("concurrency", "3"));
225
+ const STRICT = flag("strict");
226
+ const GATE = !flag("no-gate");
227
+
228
+ // ── the filesystem CONTEXT the skills are measured in ────────────────────────
229
+ // The default empty cwd is faithful for opening-move skills but biased LOW for skills whose
230
+ // trigger is a repo STATE. These skills all presuppose "there is a paper here", so the paper is
231
+ // seeded. Deliberately small: it is scenery, not a document under test.
232
+ //
233
+ // 🔴 HONEST GAP: `cold-read-diff`'s real trigger is a DIRTY GIT TREE ("prose I just changed").
234
+ // `fixture` writes plain files; it cannot seed a git history or a diff. Its recall below is
235
+ // therefore a LOWER bound, and a low number for that case is partly an artifact of this gap, not
236
+ // necessarily a description defect. Do not report it as one.
237
+ const FIXTURE = {
238
+ "paper/paper.md": [
239
+ "# Prose Isn't Policy: Measuring Whether Agent-Config Rules Are Enforceable",
240
+ "",
241
+ "## Abstract",
242
+ "Agent configuration files state rules in prose and assume the model obeys them. We compile a",
243
+ "corpus of real rules and measure what fraction can be mechanically enforced. We find that 84%",
244
+ "of rules in our corpus are enforceable, and that LLM-authored checkers for the remainder leak",
245
+ "silently in 84-96% of adversarial cases \\cite{greshake2023}.",
246
+ "",
247
+ "## 1 Introduction",
248
+ "Every agent harness ships a natural-language rulebook. Nothing checks it. This is the same",
249
+ "mistake as a code comment that claims an invariant no test enforces \\cite{thompson1984}.",
250
+ "",
251
+ "## 2 Method",
252
+ "We gather rules from public repositories, classify each by enforceability, and build a",
253
+ "two-stage adversarial gate that validates a synthesized rule against a blind gold set.",
254
+ "",
255
+ "## 3 Results",
256
+ "See Table 1. The headline number is 84%.",
257
+ "",
258
+ "## 4 Discussion",
259
+ "The result generalizes beyond our corpus in the sense that the mechanism is not corpus-specific,",
260
+ "though of course the specific percentages are, and it is important to note in this context that",
261
+ "the framing itself may be what carries, rather than the measurement.",
262
+ "",
263
+ "## 5 Threats to Validity",
264
+ "Our corpus is drawn from public repositories and may not represent private configurations.",
265
+ "",
266
+ "## 6 Related Work",
267
+ "TODO",
268
+ "",
269
+ "## 7 Conclusion",
270
+ "Prose is not policy.",
271
+ ].join("\n"),
272
+ "paper/repro/README.md":
273
+ "# Reproduction artifact\n\n`python3 paper_numbers.py` recomputes every bolded figure in paper.md.\n",
274
+ "paper/PIPELINE-STATUS.md":
275
+ "# Pipeline status\n\n| gate | state |\n|---|---|\n| numbers | pass |\n| structure | not run |\n| citations | not run |\n",
276
+ };
277
+
278
+ // ── the cases ────────────────────────────────────────────────────────────────
279
+ // Chosen where descriptions make COMPETING CLAIMS on the same territory, because that is where
280
+ // triggering actually fails. Three of these skills open their SKILL.md with a "which skill?"
281
+ // callout precisely because they collide — that callout is prose, and prose is not policy, which
282
+ // is the thesis of the paper in the fixture above.
283
+ //
284
+ // `collides` names the sibling whose prompts become this case's `irrelevantPrompts`. That is how
285
+ // "the WRONG skill must not fire" is asserted: the sibling's own should-fire prompts are fed to
286
+ // this case, and any firing is a false positive. It is symmetric — each side of a collision is
287
+ // measured from both directions — and it is the API-native form of `assertToolNotUsed`, which
288
+ // cannot be used inside `fired` (it throws; `fired` must return a boolean).
289
+ const CASES = [
290
+ {
291
+ skill: "tighten-paper",
292
+ why: "structural bloat — cut/fold/merge, NOT sentence craft",
293
+ prompts: [
294
+ "статья раздулась, середина провисает — что резать?",
295
+ "this draft is 14 pages against a 9 page limit, what goes",
296
+ "после трёх раундов ревью там одна вода и хеджи, нужен план сокращения",
297
+ "sections 4 and 5 say the same thing twice, and nobody would skim any of it",
298
+ ],
299
+ collides: ["grade-paper-writing", "argument-arc"],
300
+ },
301
+ {
302
+ skill: "grade-paper-writing",
303
+ why: "prose craft — the sentence is the unit, NOT the section",
304
+ prompts: [
305
+ "оцени как написано — читается как стена жаргона",
306
+ "is the writing any good or does it read like shit",
307
+ "abstract звучит криво хотя по смыслу всё на месте, дай оценку прозе",
308
+ "grade the craft: title, abstract, sentence clarity, hedge stacking",
309
+ ],
310
+ collides: ["tighten-paper", "argument-arc"],
311
+ },
312
+ {
313
+ skill: "argument-arc",
314
+ why: "argument architecture — does one conclusion become inevitable",
315
+ prompts: [
316
+ "ревьюер второй раз пишет что мы кидаем в него идеи без связи",
317
+ "does the paper actually carry a reader to one conclusion or just list stuff",
318
+ "мы вводим пять именованных штук и три числа — по-моему это перебор",
319
+ "before the big rewrite I want one sentence per section, bottom up",
320
+ ],
321
+ collides: ["tighten-paper", "grade-paper-writing"],
322
+ },
323
+ {
324
+ skill: "paper-adversarial-review",
325
+ why: "ONE hostile reviewer, fast",
326
+ prompts: [
327
+ "red-team эту статью, чем будет бить reviewer 2",
328
+ "would reviewer 2 buy this claim about the hook finding",
329
+ "найди слабые места до сабмита — один злой но честный рецензент",
330
+ "what is our desk reject risk and where do we overclaim",
331
+ ],
332
+ collides: ["pc-panel-review"],
333
+ },
334
+ {
335
+ skill: "pc-panel-review",
336
+ why: "the WHOLE committee + an accept probability, not one reviewer",
337
+ prompts: [
338
+ "какая вероятность принятия у этой статьи, если честно",
339
+ "simulate the whole program committee, not one reviewer",
340
+ "что решат на PC discussion — accept или reject",
341
+ "нужно несколько независимых ревьюеров с разными линзами плюс мета-ревью от чейра",
342
+ ],
343
+ collides: ["paper-adversarial-review"],
344
+ },
345
+ {
346
+ skill: "cold-read-diff",
347
+ why: "fires AFTER a prose edit — the reader with no context",
348
+ prompts: [
349
+ "я переписал третий абзац intro — проверь что предложения вообще что-то значат",
350
+ "just edited the threats section, would a reader with no context get it",
351
+ "поправил формулировки в 4.2, прогони свежим читателем до того как я закрою правку",
352
+ "эти предложения короткие, правдивые и всё равно непонятно что они утверждают",
353
+ ],
354
+ collides: ["grade-paper-writing", "tighten-paper"],
355
+ },
356
+ {
357
+ skill: "verify-citations",
358
+ why: "are the cites REAL — a pre-submit metadata gate",
359
+ prompts: [
360
+ "проверь что все цитаты настоящие перед сабмитом",
361
+ "did we hallucinate any of these refs",
362
+ "сверь метаданные по каждому cite — год, венью, авторы, doi",
363
+ "one bibtex entry looks invented to me, check the whole bibliography",
364
+ ],
365
+ collides: ["map-prior-work"],
366
+ },
367
+ {
368
+ skill: "map-prior-work",
369
+ why: "who already did this — BEFORE drafting, reshapes the contribution",
370
+ prompts: [
371
+ "кто уже это сделал до нас — хочу знать до того как начну писать",
372
+ "sweep the landscape: everyone working on this, prior versus concurrent",
373
+ "нужен скелет related work и вердикт что мы ещё можем клеймить своим",
374
+ "find every competing group in this space and date them against our submission",
375
+ ],
376
+ collides: ["verify-citations"],
377
+ },
378
+ ];
379
+
380
+ /**
381
+ * Irrelevant set for a case: its colliders' should-fire prompts, sliced so no two cases receive
382
+ * the identical set (near-duplicate sets pass the per-set diversity gate but make two cases'
383
+ * false-positive rates the same measurement twice).
384
+ */
385
+ const bySkill = new Map(CASES.map((c) => [c.skill, c]));
386
+ const irrelevantFor = (c) => {
387
+ const take = c.collides.length === 1 ? 4 : 2;
388
+ return c.collides.flatMap((s, i) => {
389
+ const p = bySkill.get(s).prompts;
390
+ return take === 4 ? p : i === 0 ? p.slice(0, 2) : p.slice(2, 4);
391
+ });
392
+ };
393
+
394
+ // ── preflight: fail LOUDLY rather than measuring nothing ─────────────────────
395
+ // Lie #2's shape, transplanted. A misspelled skill or a changed namespace makes every `fired`
396
+ // predicate permanently false, and the run then reports a wall of confident 0.00s.
397
+ function assertSkillIdsExist() {
398
+ if (!existsSync(SKILLS_DIR))
399
+ throw new Error(`no skills dir at ${SKILLS_DIR}`);
400
+ // Through `installedSkills`, which follows the links `paperlint init` makes (rpp#62).
401
+ const installed = new Set(installedSkills(SKILLS_DIR));
402
+ const missing = CASES.map((c) => c.skill).filter((s) => !installed.has(s));
403
+ if (missing.length)
404
+ throw new Error(
405
+ `these cases name skills that are not installed under ${SKILLS_DIR}: ${missing.join(", ")}. ` +
406
+ `Every \`fired\` predicate for them would be permanently false and the run would report 0.00 as a finding.`,
407
+ );
408
+ // And the frontmatter `name:` must equal the directory name — the id the model reports is built
409
+ // from the directory, but a mismatch means the SKILL.md a human reads is not the one measured.
410
+ for (const c of CASES) {
411
+ const fm = readFileSync(join(SKILLS_DIR, c.skill, "SKILL.md"), "utf-8");
412
+ const declared = declaredName(fm);
413
+ if (declared && declared !== c.skill)
414
+ throw new Error(
415
+ `${c.skill}/SKILL.md declares name: ${declared} — id mismatch, fix one of them`,
416
+ );
417
+ }
418
+ return installed.size;
419
+ }
420
+
421
+ const installedCount = assertSkillIdsExist();
422
+
423
+ // Free, deterministic, and it runs BEFORE anything spends a token — the same order
424
+ // `measureTriggerRate` uses internally. A prompt set that all reads the same way makes a high
425
+ // trigger rate meaningless: the model would be answering one question four times.
426
+ for (const c of CASES) {
427
+ assertPromptDiversity(c.prompts, {
428
+ minPrompts: 4,
429
+ minDistance: 0.3,
430
+ label: `${c.skill}:should-fire`,
431
+ });
432
+ assertPromptDiversity(irrelevantFor(c), {
433
+ minPrompts: 4,
434
+ minDistance: 0.3,
435
+ label: `${c.skill}:should-not-fire`,
436
+ });
437
+ }
438
+ // Cross-set too: two cases whose should-fire sets are near-identical are not two measurements.
439
+ for (let i = 0; i < CASES.length; i++)
440
+ for (let j = i + 1; j < CASES.length; j++) {
441
+ const issues = checkPromptDiversity(
442
+ [...CASES[i].prompts, ...CASES[j].prompts],
443
+ {
444
+ minPrompts: 8,
445
+ minDistance: 0.25,
446
+ label: `${CASES[i].skill} × ${CASES[j].skill}`,
447
+ },
448
+ );
449
+ if (issues.length) throw new Error(issues.map((x) => x.message).join("\n"));
450
+ }
451
+
452
+ console.log(
453
+ `prompt sets: OK (${CASES.length} cases, ${CASES.length * 8} runs × ${TRIALS} trial(s), ` +
454
+ `${installedCount} skills installed → ${installedCount - 1} competitors per run)`,
455
+ );
456
+
457
+ // The eval tier needs the real binary. A missing CLI is a SKIP, never a silent pass.
458
+ try {
459
+ execFileSync("claude", ["--version"], { stdio: "ignore" });
460
+ } catch {
461
+ skip(
462
+ "`claude` CLI not on PATH — the eval tier drives the real harness and cannot be faked",
463
+ );
464
+ }
465
+
466
+ const selected = onlys.length
467
+ ? CASES.filter((c) => onlys.includes(c.skill))
468
+ : CASES;
469
+ if (selected.length === 0)
470
+ throw new Error(
471
+ `--only matched nothing. Known: ${CASES.map((c) => c.skill).join(", ")}`,
472
+ );
473
+
474
+ // ── the measurement ──────────────────────────────────────────────────────────
475
+ const results = [];
476
+ for (const c of selected) {
477
+ const colliderIds = c.collides.map(id);
478
+ const report = await measureTriggerRate({
479
+ name: `pipeline-firing:${c.skill}${STRICT ? ":strict" : ""}`,
480
+ // NOT `plugin` (materialises files, registers no skills) and NOT `pluginDir` (wants a complete
481
+ // plugin). `skillsDir` packages this loose dir into a real `--plugin-dir` install. See header.
482
+ skillsDir: SKILLS_DIR,
483
+ prompts: c.prompts,
484
+ irrelevantPrompts: irrelevantFor(c),
485
+ // In strict mode a run counts as fired only if the right skill fired AND every colliding
486
+ // sibling stayed silent — "the wrong one did NOT fire", read off the trace's skill list.
487
+ fired: STRICT
488
+ ? (t) =>
489
+ skillResolved(t, id(c.skill)) &&
490
+ !colliderIds.some((x) => skillResolved(t, x))
491
+ : (t) => skillResolved(t, id(c.skill)),
492
+ fixture: FIXTURE,
493
+ // 4, not the default 10: these are deliberately narrow skills and the grid is already 64 runs.
494
+ // Lowering it is a REAL loss of power — at n=4 a rate is ±0.25 per prompt, so read the
495
+ // per-prompt lines, not the third decimal of the mean.
496
+ minPrompts: 4,
497
+ minDistance: 0.3,
498
+ trials: TRIALS,
499
+ // Selection only. No Write/Edit/Bash/Task: this eval cannot touch a paper, push, or spawn a
500
+ // paid subagent. Read is allowed because a real user's prompt refers to a file.
501
+ allowedTools: ["Skill", "Read"],
502
+ // stubSkillBodies defaults true — every body is a no-op, so a fired panel skill stops at
503
+ // selection instead of opening N reviewer subagents. Left implicit deliberately: overriding it
504
+ // to false is what would need a justification, not leaving it on.
505
+ concurrency: CONCURRENCY,
506
+ spacingSec: 2,
507
+ timeoutMs: 180000,
508
+ });
509
+ console.log(`\n=== ${c.skill} — ${c.why}`);
510
+ console.log(` collides with: ${c.collides.join(", ")}`);
511
+ console.log(formatTriggerRateReport(report));
512
+ results.push({ case: c, report });
513
+ }
514
+
515
+ // ── read the result ──────────────────────────────────────────────────────────
516
+ const pct = (x) =>
517
+ x === undefined ? " — " : `${(x * 100).toFixed(0)}%`.padStart(5);
518
+ console.log(`\n${"skill".padEnd(26)} recall FP-rate precision n cost`);
519
+ for (const { case: c, report: r } of results)
520
+ console.log(
521
+ `${c.skill.padEnd(26)} ${pct(r.rate)} ${pct(r.falsePositiveRate)} ${pct(r.precision)} ` +
522
+ `${String(r.n).padStart(3)} $${r.usage.totalCostUsd.toFixed(2)}`,
523
+ );
524
+ const spend = results.reduce((s, x) => s + x.report.usage.totalCostUsd, 0);
525
+ console.log(
526
+ `${"".padEnd(26)} total $${spend.toFixed(2)}`,
527
+ );
528
+
529
+ // 🔴 DID A MEASUREMENT HAPPEN AT ALL? This has to be answered BEFORE the floor gate below,
530
+ // because the two failures look identical from the outside and mean opposite things.
531
+ //
532
+ // Observed 2026-08-10 and 2026-08-17 — the only two scheduled runs this canary has ever had,
533
+ // both `failure`, every case 0.00, total $0.00. That reads as "every skill stopped firing",
534
+ // which is a five-alarm finding. It was not: the model never ran. The CLI guard above passes in
535
+ // CI (the job installs the binary) and nothing checked for a CREDENTIAL, so a keyless run walked
536
+ // straight into the floor gate and reported an environment fault as a regression. Eight days of
537
+ // red that nobody could act on, because the message pointed at the wrong thing.
538
+ //
539
+ // The check is on the IMPOSSIBLE STATE rather than on a list of causes: real model calls cost
540
+ // money, so zero spend across every case means no call was billed — whatever the reason (absent
541
+ // key, revoked key, network, quota). Enumerating causes would leave the next one undetected.
542
+ if (results.length > 0 && spend === 0) {
543
+ const everythingZero = results.every(({ report: r }) => r.rate === 0);
544
+ throw new Error(
545
+ `NO MEASUREMENT HAPPENED — ${String(results.length)} case(s) ran and total spend is $0.00` +
546
+ (everythingZero ? " with every recall at 0.00" : "") +
547
+ `.\nReal model calls are billed, so zero spend means no call reached a model. This is an ` +
548
+ `ENVIRONMENT fault, not a trigger-rate regression — do not read the numbers below as a ` +
549
+ `finding about the skills.\nMost likely: no credential. ANTHROPIC_API_KEY is ` +
550
+ `${process.env.ANTHROPIC_API_KEY ? "set" : "NOT SET"} in this process. The \`claude --version\` ` +
551
+ `guard above cannot see this: the binary installs fine without a key.`,
552
+ );
553
+ }
554
+
555
+ // Lie #2 again: an eval that measured nothing must not read as a finding.
556
+ if (results.length > 1 && results.every((x) => x.report.rate === 0))
557
+ throw new Error(
558
+ "EVERY case scored 0.00. That is far more likely a harness fault (wrong namespace, skills not " +
559
+ "installed, `fired` predicate broken) than eight simultaneously dead descriptions. Do not " +
560
+ "record this as a baseline — check the trace of one run first.",
561
+ );
562
+
563
+ // ── baseline / regression ────────────────────────────────────────────────────
564
+ // 🔴 API MISMATCH #2: `writeBaseline`/`readBaseline`/`assertNoRegression` are typed on
565
+ // `EvalReport` (arms × metrics × MetricStat), while `measureTriggerRate` returns a
566
+ // `TriggerRateReport`. There is no adapter in the package, so one is written here: each case
567
+ // becomes an ARM, `recall` and `falsePositiveRate` become METRICS, and the Bernoulli stats are
568
+ // derived from the per-prompt fired counts. Nothing is invented — std is the sample std of the
569
+ // 0/1 outcomes, n is the real run count.
570
+ const bernoulli = (successes, n) => {
571
+ const mean = n > 0 ? successes / n : 0;
572
+ const std = n > 1 ? Math.sqrt((mean * (1 - mean) * n) / (n - 1)) : 0;
573
+ return {
574
+ mean,
575
+ std,
576
+ se: n > 0 ? std / Math.sqrt(n) : 0,
577
+ n,
578
+ passK: n > 0 && successes === n ? 1 : 0,
579
+ };
580
+ };
581
+ const asEvalReport = () => {
582
+ const arms = {};
583
+ for (const { case: c, report: r } of results) {
584
+ const fired = r.perPrompt.reduce((s, p) => s + p.fired, 0);
585
+ const irrFired = (r.perIrrelevant ?? []).reduce((s, p) => s + p.fired, 0);
586
+ const irrN = (r.perIrrelevant ?? []).reduce((s, p) => s + p.trials, 0);
587
+ const recall = bernoulli(fired, r.n);
588
+ const fp = bernoulli(irrFired, irrN);
589
+ arms[c.skill] = {
590
+ runs: r.n + irrN,
591
+ metrics: { recall: recall.mean, falsePositiveRate: fp.mean },
592
+ stats: { recall, falsePositiveRate: fp },
593
+ usage: r.usage,
594
+ };
595
+ }
596
+ return {
597
+ name: `pipeline-firing${STRICT ? ":strict" : ""}`,
598
+ trials: TRIALS,
599
+ arms,
600
+ totalCostUsd: spend,
601
+ aborted: false,
602
+ };
603
+ };
604
+
605
+ const current = [asEvalReport()];
606
+ if (flag("update-baseline")) {
607
+ writeBaseline(BASELINE, current);
608
+ console.log(`\nbaseline recorded → ${BASELINE}`);
609
+ } else {
610
+ const prior = readBaseline(BASELINE);
611
+ if (!prior) {
612
+ console.log(
613
+ `\nno baseline at ${BASELINE} — record one with --update-baseline`,
614
+ );
615
+ } else if (onlys.length) {
616
+ console.log(
617
+ "\n--only run: skipping the regression diff (a partial run is not comparable)",
618
+ );
619
+ } else {
620
+ const diff = diffReports(prior, current, {
621
+ lowerIsBetter: ["falsePositiveRate"],
622
+ });
623
+ console.log(`\nvs baseline recorded ${prior.recordedAt}:`);
624
+ console.log(formatBaselineDiff(diff));
625
+ // 🔴 READ THIS BEFORE TRUSTING THE GATE. Welch on 4 Bernoulli trials per metric has almost no
626
+ // power: a drop from 100% to 50% is not significant at n=4, so this gate catches only a
627
+ // COLLAPSE. Raise --trials (3 trials ⇒ n=12) for a gate that catches drift rather than death.
628
+ if (GATE && TRIALS >= 3)
629
+ assertNoRegression(current, prior, {
630
+ lowerIsBetter: ["falsePositiveRate"],
631
+ });
632
+ else if (GATE)
633
+ console.log(
634
+ " (regression gate not enforced: needs --trials 3 or more to have power)",
635
+ );
636
+ }
637
+ }
638
+
639
+ // ── the absolute floor ───────────────────────────────────────────────────────
640
+ // Deliberately LOW, and the low number is the honest one. These are not tuned thresholds — they
641
+ // are the line below which a description is BROKEN rather than noisy, and there is no measured
642
+ // history yet to justify anything tighter. The drift question belongs to the baseline diff above,
643
+ // not to a number invented here. Tighten these only against recorded runs.
644
+ //
645
+ // Precision is NOT gated. A colliding sibling firing alongside the right skill is often CORRECT
646
+ // chaining — grade-paper-writing's own description says to run tighten-paper first on a bloated
647
+ // draft — so a false positive here means "fired on a prompt another skill owns", which includes
648
+ // legitimate hand-offs. Gating it would punish the composition these skills are designed for.
649
+ const FLOOR = { min: 0.5, maxFalsePositive: 0.75 };
650
+ if (GATE) {
651
+ const failures = [];
652
+ for (const { case: c, report: r } of results) {
653
+ try {
654
+ assertTriggerRate(r, FLOOR);
655
+ } catch (e) {
656
+ failures.push(`${c.skill}: ${e.message}`);
657
+ }
658
+ }
659
+ if (failures.length)
660
+ throw new Error(`trigger floor breached:\n ${failures.join("\n ")}`);
661
+ console.log(
662
+ `\nfloor OK: every case ≥ ${FLOOR.min * 100}% recall, ≤ ${FLOOR.maxFalsePositive * 100}% false positives`,
663
+ );
664
+ }