paperlint 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/dependabot.yml +72 -0
- package/.github/workflows/ci.yml +297 -0
- package/.github/workflows/dependabot-automerge.yml +70 -0
- package/.github/workflows/pr-title.yml +59 -0
- package/.github/workflows/release.yml +54 -0
- package/CLAUDE.md +598 -0
- package/CONTRIBUTING.md +159 -0
- package/LICENSE +21 -0
- package/README.md +240 -0
- package/action.harness.mjs +287 -0
- package/action.mutations.mjs +162 -0
- package/action.yml +138 -0
- package/bin/rpp.mjs +43 -0
- package/dist/action-ref.d.ts +12 -0
- package/dist/action-ref.d.ts.map +1 -0
- package/dist/action-ref.js +16 -0
- package/dist/action-ref.js.map +1 -0
- package/dist/adapters/banal/failure.d.ts +73 -0
- package/dist/adapters/banal/failure.d.ts.map +1 -0
- package/dist/adapters/banal/failure.js +58 -0
- package/dist/adapters/banal/failure.js.map +1 -0
- package/dist/adapters/banal/index.d.ts +17 -0
- package/dist/adapters/banal/index.d.ts.map +1 -0
- package/dist/adapters/banal/index.js +56 -0
- package/dist/adapters/banal/index.js.map +1 -0
- package/dist/adapters/banal/install.d.ts +26 -0
- package/dist/adapters/banal/install.d.ts.map +1 -0
- package/dist/adapters/banal/install.js +15 -0
- package/dist/adapters/banal/install.js.map +1 -0
- package/dist/adapters/banal/invocation.d.ts +48 -0
- package/dist/adapters/banal/invocation.d.ts.map +1 -0
- package/dist/adapters/banal/invocation.js +43 -0
- package/dist/adapters/banal/invocation.js.map +1 -0
- package/dist/adapters/banal/locate.d.ts +50 -0
- package/dist/adapters/banal/locate.d.ts.map +1 -0
- package/dist/adapters/banal/locate.js +34 -0
- package/dist/adapters/banal/locate.js.map +1 -0
- package/dist/adapters/banal/output.d.ts +27 -0
- package/dist/adapters/banal/output.d.ts.map +1 -0
- package/dist/adapters/banal/output.js +112 -0
- package/dist/adapters/banal/output.js.map +1 -0
- package/dist/adapters/banal/pin.d.ts +19 -0
- package/dist/adapters/banal/pin.d.ts.map +1 -0
- package/dist/adapters/banal/pin.js +15 -0
- package/dist/adapters/banal/pin.js.map +1 -0
- package/dist/adapters/banal/probe.d.ts +12 -0
- package/dist/adapters/banal/probe.d.ts.map +1 -0
- package/dist/adapters/banal/probe.js +27 -0
- package/dist/adapters/banal/probe.js.map +1 -0
- package/dist/adapters/banal/run.d.ts +89 -0
- package/dist/adapters/banal/run.d.ts.map +1 -0
- package/dist/adapters/banal/run.js +104 -0
- package/dist/adapters/banal/run.js.map +1 -0
- package/dist/adapters/banal/settings.d.ts +18 -0
- package/dist/adapters/banal/settings.d.ts.map +1 -0
- package/dist/adapters/banal/settings.js +29 -0
- package/dist/adapters/banal/settings.js.map +1 -0
- package/dist/adapters/banal/xml.d.ts +48 -0
- package/dist/adapters/banal/xml.d.ts.map +1 -0
- package/dist/adapters/banal/xml.js +67 -0
- package/dist/adapters/banal/xml.js.map +1 -0
- package/dist/adapters/curl/download.io.d.ts +14 -0
- package/dist/adapters/curl/download.io.d.ts.map +1 -0
- package/dist/adapters/curl/download.io.js +69 -0
- package/dist/adapters/curl/download.io.js.map +1 -0
- package/dist/adapters/curl/index.d.ts +6 -0
- package/dist/adapters/curl/index.d.ts.map +1 -0
- package/dist/adapters/curl/index.js +6 -0
- package/dist/adapters/curl/index.js.map +1 -0
- package/dist/adapters/memory/index.d.ts +43 -0
- package/dist/adapters/memory/index.d.ts.map +1 -0
- package/dist/adapters/memory/index.js +79 -0
- package/dist/adapters/memory/index.js.map +1 -0
- package/dist/adapters/node/files.io.d.ts +3 -0
- package/dist/adapters/node/files.io.d.ts.map +1 -0
- package/dist/adapters/node/files.io.js +31 -0
- package/dist/adapters/node/files.io.js.map +1 -0
- package/dist/adapters/node/host.io.d.ts +3 -0
- package/dist/adapters/node/host.io.d.ts.map +1 -0
- package/dist/adapters/node/host.io.js +14 -0
- package/dist/adapters/node/host.io.js.map +1 -0
- package/dist/adapters/node/index.d.ts +25 -0
- package/dist/adapters/node/index.d.ts.map +1 -0
- package/dist/adapters/node/index.js +14 -0
- package/dist/adapters/node/index.js.map +1 -0
- package/dist/adapters/node/process.io.d.ts +14 -0
- package/dist/adapters/node/process.io.d.ts.map +1 -0
- package/dist/adapters/node/process.io.js +41 -0
- package/dist/adapters/node/process.io.js.map +1 -0
- package/dist/adapters/node/workspace.io.d.ts +4 -0
- package/dist/adapters/node/workspace.io.d.ts.map +1 -0
- package/dist/adapters/node/workspace.io.js +33 -0
- package/dist/adapters/node/workspace.io.js.map +1 -0
- package/dist/adapters/pdfjs/fill.d.ts +42 -0
- package/dist/adapters/pdfjs/fill.d.ts.map +1 -0
- package/dist/adapters/pdfjs/fill.js +91 -0
- package/dist/adapters/pdfjs/fill.js.map +1 -0
- package/dist/build-engine.d.ts +48 -0
- package/dist/build-engine.d.ts.map +1 -0
- package/dist/build-engine.js +148 -0
- package/dist/build-engine.js.map +1 -0
- package/dist/build.d.ts +163 -0
- package/dist/build.d.ts.map +1 -0
- package/dist/build.js +575 -0
- package/dist/build.js.map +1 -0
- package/dist/cli.d.ts +151 -0
- package/dist/cli.d.ts.map +1 -0
- package/dist/cli.js +951 -0
- package/dist/cli.js.map +1 -0
- package/dist/doctor.d.ts +42 -0
- package/dist/doctor.d.ts.map +1 -0
- package/dist/doctor.js +280 -0
- package/dist/doctor.js.map +1 -0
- package/dist/domain/geometry.d.ts +71 -0
- package/dist/domain/geometry.d.ts.map +1 -0
- package/dist/domain/geometry.js +35 -0
- package/dist/domain/geometry.js.map +1 -0
- package/dist/domain/host.d.ts +16 -0
- package/dist/domain/host.d.ts.map +1 -0
- package/dist/domain/host.js +8 -0
- package/dist/domain/host.js.map +1 -0
- package/dist/domain/page-layout.d.ts +34 -0
- package/dist/domain/page-layout.d.ts.map +1 -0
- package/dist/domain/page-layout.js +8 -0
- package/dist/domain/page-layout.js.map +1 -0
- package/dist/domain/paths.d.ts +5 -0
- package/dist/domain/paths.d.ts.map +1 -0
- package/dist/domain/paths.js +2 -0
- package/dist/domain/paths.js.map +1 -0
- package/dist/domain/result.d.ts +23 -0
- package/dist/domain/result.d.ts.map +1 -0
- package/dist/domain/result.js +10 -0
- package/dist/domain/result.js.map +1 -0
- package/dist/domain/sha256.d.ts +7 -0
- package/dist/domain/sha256.d.ts.map +1 -0
- package/dist/domain/sha256.js +14 -0
- package/dist/domain/sha256.js.map +1 -0
- package/dist/domain/text.d.ts +6 -0
- package/dist/domain/text.d.ts.map +1 -0
- package/dist/domain/text.js +7 -0
- package/dist/domain/text.js.map +1 -0
- package/dist/engine.d.ts +93 -0
- package/dist/engine.d.ts.map +1 -0
- package/dist/engine.js +119 -0
- package/dist/engine.js.map +1 -0
- package/dist/exit-code.d.ts +22 -0
- package/dist/exit-code.d.ts.map +1 -0
- package/dist/exit-code.js +10 -0
- package/dist/exit-code.js.map +1 -0
- package/dist/facts-file.d.ts +96 -0
- package/dist/facts-file.d.ts.map +1 -0
- package/dist/facts-file.js +134 -0
- package/dist/facts-file.js.map +1 -0
- package/dist/hooks-settings.d.ts +141 -0
- package/dist/hooks-settings.d.ts.map +1 -0
- package/dist/hooks-settings.js +306 -0
- package/dist/hooks-settings.js.map +1 -0
- package/dist/init.d.ts +201 -0
- package/dist/init.d.ts.map +1 -0
- package/dist/init.js +579 -0
- package/dist/init.js.map +1 -0
- package/dist/latex-log.d.ts +80 -0
- package/dist/latex-log.d.ts.map +1 -0
- package/dist/latex-log.js +187 -0
- package/dist/latex-log.js.map +1 -0
- package/dist/latex-loop.d.ts +129 -0
- package/dist/latex-loop.d.ts.map +1 -0
- package/dist/latex-loop.js +113 -0
- package/dist/latex-loop.js.map +1 -0
- package/dist/link-skills.d.ts +51 -0
- package/dist/link-skills.d.ts.map +1 -0
- package/dist/link-skills.js +199 -0
- package/dist/link-skills.js.map +1 -0
- package/dist/new-paper.d.ts +48 -0
- package/dist/new-paper.d.ts.map +1 -0
- package/dist/new-paper.js +110 -0
- package/dist/new-paper.js.map +1 -0
- package/dist/pdf-facts.d.ts +44 -0
- package/dist/pdf-facts.d.ts.map +1 -0
- package/dist/pdf-facts.js +239 -0
- package/dist/pdf-facts.js.map +1 -0
- package/dist/pdf-geometry.d.ts +170 -0
- package/dist/pdf-geometry.d.ts.map +1 -0
- package/dist/pdf-geometry.js +158 -0
- package/dist/pdf-geometry.js.map +1 -0
- package/dist/ports/download.d.ts +9 -0
- package/dist/ports/download.d.ts.map +1 -0
- package/dist/ports/download.js +2 -0
- package/dist/ports/download.js.map +1 -0
- package/dist/ports/files.d.ts +11 -0
- package/dist/ports/files.d.ts.map +1 -0
- package/dist/ports/files.js +2 -0
- package/dist/ports/files.js.map +1 -0
- package/dist/ports/measure-geometry.d.ts +8 -0
- package/dist/ports/measure-geometry.d.ts.map +1 -0
- package/dist/ports/measure-geometry.js +2 -0
- package/dist/ports/measure-geometry.js.map +1 -0
- package/dist/ports/process.d.ts +45 -0
- package/dist/ports/process.d.ts.map +1 -0
- package/dist/ports/process.js +2 -0
- package/dist/ports/process.js.map +1 -0
- package/dist/ports/tool-installer.d.ts +29 -0
- package/dist/ports/tool-installer.d.ts.map +1 -0
- package/dist/ports/tool-installer.js +2 -0
- package/dist/ports/tool-installer.js.map +1 -0
- package/dist/ports/workspace.d.ts +18 -0
- package/dist/ports/workspace.d.ts.map +1 -0
- package/dist/ports/workspace.js +2 -0
- package/dist/ports/workspace.js.map +1 -0
- package/dist/rules-config.d.ts +34 -0
- package/dist/rules-config.d.ts.map +1 -0
- package/dist/rules-config.js +132 -0
- package/dist/rules-config.js.map +1 -0
- package/dist/structure.d.ts +34 -0
- package/dist/structure.d.ts.map +1 -0
- package/dist/structure.js +149 -0
- package/dist/structure.js.map +1 -0
- package/dist/tex-requirements.d.ts +43 -0
- package/dist/tex-requirements.d.ts.map +1 -0
- package/dist/tex-requirements.js +127 -0
- package/dist/tex-requirements.js.map +1 -0
- package/dist/toolchain.d.ts +159 -0
- package/dist/toolchain.d.ts.map +1 -0
- package/dist/toolchain.js +542 -0
- package/dist/toolchain.js.map +1 -0
- package/dist/types.d.ts +110 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +2 -0
- package/dist/types.js.map +1 -0
- package/docs/configuration.md +235 -0
- package/docs/e2e.md +152 -0
- package/docs/incidents.md +59 -0
- package/docs/install.md +170 -0
- package/docs/optional-rules.md +107 -0
- package/docs/package-shape-options.md +262 -0
- package/docs/prior-art/README.md +76 -0
- package/docs/prior-art/blocking-vs-advisory.md +83 -0
- package/docs/prior-art/content-delivery.md +124 -0
- package/docs/prior-art/multi-mode-tools.md +106 -0
- package/docs/prior-art/nondeterministic-checks.md +99 -0
- package/docs/prior-art/package-location.md +422 -0
- package/docs/prior-art/paper-folder-scaffolding.md +538 -0
- package/docs/prior-art/readme-structure.md +69 -0
- package/docs/prior-art/repro/README.md +92 -0
- package/docs/prior-art/repro/claim1-allowedtools.mjs +66 -0
- package/docs/prior-art/repro/claim1-at2.mjs +40 -0
- package/docs/prior-art/repro/claim1-crosschannel.mjs +54 -0
- package/docs/prior-art/repro/claim1-frontmatter.mjs +76 -0
- package/docs/prior-art/repro/claim1-hook-payload-reporter.mjs +10 -0
- package/docs/prior-art/repro/claim1-plugin-frontmatter.mjs +27 -0
- package/docs/prior-art/repro/claim1-plugin-skill.mjs +52 -0
- package/docs/prior-art/repro/claim1-project-skill.mjs +81 -0
- package/docs/prior-art/repro/claim2-marketplace-flat-asclaimed.json +1 -0
- package/docs/prior-art/repro/claim2-marketplace-negative-control.json +1 -0
- package/docs/prior-art/repro/claim2-marketplace-nested-exact.json +9 -0
- package/docs/prior-art/repro/claim2-marketplace-nested-noversion.json +9 -0
- package/docs/prior-art/repro/claim2-marketplace-nested-range.json +1 -0
- package/docs/prior-art/repro/claim3-imports.mjs +50 -0
- package/docs/prior-art/repro/claim4-find-package-json.mjs +8 -0
- package/docs/prior-art/repro/claim4-package-dir.mjs +39 -0
- package/docs/prior-art/repro/claim4-parent-arg.mjs +17 -0
- package/docs/prior-art/repro/claim4-resolve-apis.mjs +21 -0
- package/docs/prior-art/repro/claim4-setup-consumers.mjs +45 -0
- package/docs/prior-art/repro/claim4-yarn-pnp.mjs +70 -0
- package/docs/prior-art/repro/claim5-bin-launch.mjs +39 -0
- package/docs/prior-art/repro/claim5-exports-mutation.mjs +57 -0
- package/docs/prior-art/repro/claim5-resolved-location-and-bin.mjs +33 -0
- package/docs/prior-art/repro/claim6-candidate-ambiguity.mjs +17 -0
- package/docs/prior-art/repro/claim6-doc-path-candidates.mjs +27 -0
- package/docs/prior-art/test-tooling.md +131 -0
- package/docs/rules.md +58 -0
- package/docs/texlive-install-decision.md +230 -0
- package/docs/toolchain.md +152 -0
- package/eslint-rules/doc-fields.harness.mjs +336 -0
- package/eslint-rules/doc-fields.mjs +186 -0
- package/eslint-rules/doc-fields.mutations.mjs +96 -0
- package/eslint-rules/install-path-literals.harness.mjs +121 -0
- package/eslint-rules/install-path-literals.mjs +108 -0
- package/eslint-rules/install-path-literals.mutations.mjs +62 -0
- package/eslint-rules/latex-language.harness.mjs +599 -0
- package/eslint-rules/latex-language.mjs +591 -0
- package/eslint-rules/latex-language.mutations.mjs +196 -0
- package/eslint-rules/paper-research-question.harness.mjs +146 -0
- package/eslint-rules/paper-research-question.mjs +180 -0
- package/eslint-rules/paper-research-question.mutations.mjs +127 -0
- package/eslint-rules/paper-stages.harness.mjs +356 -0
- package/eslint-rules/paper-stages.mjs +455 -0
- package/eslint-rules/paper-stages.mutations.mjs +157 -0
- package/eslint-rules/paper-typography.harness.mjs +291 -0
- package/eslint-rules/paper-typography.mjs +313 -0
- package/eslint-rules/paper-typography.mutations.mjs +131 -0
- package/eslint-rules/papers.harness.mjs +259 -0
- package/eslint-rules/papers.mjs +166 -0
- package/eslint-rules/papers.mutations.mjs +186 -0
- package/eslint-rules/pdf-last-page-balance.harness.mjs +206 -0
- package/eslint-rules/pdf-last-page-balance.mjs +208 -0
- package/eslint-rules/review-findings-cause.harness.mjs +228 -0
- package/eslint-rules/review-findings-cause.mjs +135 -0
- package/eslint-rules/review-findings-cause.mutations.mjs +72 -0
- package/eslint-rules/temp-root-realpath.harness.mjs +176 -0
- package/eslint-rules/temp-root-realpath.mjs +129 -0
- package/eslint-rules/temp-root-realpath.mutations.mjs +99 -0
- package/eslint-rules/tex-build.harness.mjs +753 -0
- package/eslint-rules/tex-build.mjs +322 -0
- package/eslint-rules/tex-build.mutations.mjs +258 -0
- package/eslint.config.mjs +521 -0
- package/fixtures/build-e2e/acmart/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/acmart/paper.tex +11 -0
- package/fixtures/build-e2e/acmart/venue.json +1 -0
- package/fixtures/build-e2e/broken/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/broken/paper.tex +7 -0
- package/fixtures/build-e2e/cite/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/cite/build.sh +5 -0
- package/fixtures/build-e2e/cite/paper.tex +10 -0
- package/fixtures/build-e2e/cite/refs.bib +9 -0
- package/fixtures/build-e2e/empty/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/empty/paper.tex +6 -0
- package/fixtures/build-e2e/fallback/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/fallback/paper.tex +11 -0
- package/fixtures/build-e2e/guards/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/guards/paper.tex +10 -0
- package/fixtures/build-e2e/no-source/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/unbalanced/PIPELINE-STATUS.md +3 -0
- package/fixtures/build-e2e/unbalanced/paper.tex +28 -0
- package/fixtures/build-e2e/unbalanced/refs.bib +269 -0
- package/fixtures/install-path-literals/clean.fixture.mjs +3 -0
- package/fixtures/install-path-literals/clean.md +15 -0
- package/fixtures/install-path-literals/defect.fixture.mjs +3 -0
- package/fixtures/install-path-literals/defect.md +14 -0
- package/fixtures/latex-language/clean.tex +50 -0
- package/fixtures/latex-language/defect.tex +52 -0
- package/fixtures/paper-research-question/comment-only/PIPELINE-STATUS.md +9 -0
- package/fixtures/paper-research-question/comment-only/paper.tex +7 -0
- package/fixtures/paper-research-question/declared-not-in-paper/PIPELINE-STATUS.md +10 -0
- package/fixtures/paper-research-question/declared-not-in-paper/paper.tex +6 -0
- package/fixtures/paper-research-question/draft/PIPELINE-STATUS.md +6 -0
- package/fixtures/paper-research-question/draft/paper.tex +2 -0
- package/fixtures/paper-research-question/markdown-no-rq/PIPELINE-STATUS.md +9 -0
- package/fixtures/paper-research-question/markdown-no-rq/paper.md +4 -0
- package/fixtures/paper-research-question/shipped-no-rq/PIPELINE-STATUS.md +12 -0
- package/fixtures/paper-research-question/shipped-no-rq/paper.tex +3 -0
- package/fixtures/paper-research-question/shipped-with-rq/PIPELINE-STATUS.md +10 -0
- package/fixtures/paper-research-question/shipped-with-rq/paper.tex +2 -0
- package/fixtures/paper-stages/authors-ran/PIPELINE-STATUS.md +16 -0
- package/fixtures/paper-stages/marker-in-prose/PIPELINE-STATUS.md +17 -0
- package/fixtures/paper-stages/nofile/PIPELINE-STATUS.md +8 -0
- package/fixtures/paper-stages/noheader/PIPELINE-STATUS.md +1 -0
- package/fixtures/paper-stages/noheader/versions/2026-07-22-submitted.pdf +0 -0
- package/fixtures/paper-stages/nothing/PIPELINE-STATUS.md +3 -0
- package/fixtures/paper-stages/ok/PIPELINE-STATUS.md +9 -0
- package/fixtures/paper-stages/ok/versions/2026-07-22-submitted.pdf +0 -0
- package/fixtures/paper-stages/stale/PIPELINE-STATUS.md +1 -0
- package/fixtures/paper-stages/stale/versions/2026-07-22-submitted.STALE-WRONG-FILE.pdf +0 -0
- package/fixtures/paper-stages/twice/PIPELINE-STATUS.md +14 -0
- package/fixtures/paper-stages/twice/versions/2026-08-06-submitted.pdf +0 -0
- package/fixtures/paper-stages/twice/versions/2026-10-24-submitted.pdf +0 -0
- package/fixtures/paper-stages/undeclared/PIPELINE-STATUS.md +8 -0
- package/fixtures/paper-stages/undeclared/versions/2026-07-22-submitted.pdf +0 -0
- package/fixtures/paper-stages/undeclared/versions/2026-08-29-camera-ready.pdf +0 -0
- package/fixtures/paper-stages/wrongsize/PIPELINE-STATUS.md +8 -0
- package/fixtures/paper-stages/wrongsize/versions/2026-07-22-submitted.pdf +0 -0
- package/fixtures/paper-typography/clean-paper/paper.tex +29 -0
- package/fixtures/paper-typography/messy-paper/paper.tex +27 -0
- package/fixtures/pdf-facts/README.md +22 -0
- package/fixtures/pdf-facts/corrupt-font.pdf +0 -0
- package/fixtures/pdf-facts/encrypted.pdf +0 -0
- package/fixtures/pdf-facts/hidden-text.pdf +0 -0
- package/fixtures/pdf-facts/hidden-text.tex +28 -0
- package/fixtures/pdf-facts/t3-all.pdf +0 -0
- package/fixtures/pdf-facts/t3-all.tex +8 -0
- package/fixtures/pdf-facts/t3-mixed.pdf +0 -0
- package/fixtures/pdf-facts/t3-mixed.tex +9 -0
- package/fixtures/pdf-facts/ttf.pdf +2240 -1
- package/fixtures/pdf-facts/ttf.tex +6 -0
- package/fixtures/real-markdown-paper/baseline.json +24 -0
- package/fixtures/real-markdown-paper/baseline.mjs +48 -0
- package/fixtures/render-paper/build-clean.sh +25 -0
- package/fixtures/render-paper/build-defect.sh +15 -0
- package/fixtures/review-findings-cause/clean.md +17 -0
- package/fixtures/review-findings-cause/defect.md +14 -0
- package/fixtures/review-findings-cause/old-debt.md +14 -0
- package/fixtures/review-findings-cause/quiet-in-fence.md +16 -0
- package/fixtures/tex-build/clean.tex +21 -0
- package/fixtures/tex-build/defect.tex +24 -0
- package/fixtures/tex-build/frontmatter-clean.tex +25 -0
- package/fixtures/tex-build/frontmatter-defect.tex +23 -0
- package/fixtures/toolchain-mirror/catalog.txt +5 -0
- package/fixtures/toolchain-mirror/install-tl +27 -0
- package/fixtures/toolchain-mirror/release-texlive.txt +3 -0
- package/fixtures/toolchain-mirror/release-year +1 -0
- package/fixtures/toolchain-mirror/stub-kpsewhich +8 -0
- package/fixtures/toolchain-mirror/stub-pdflatex +3 -0
- package/fixtures/toolchain-mirror/stub-tlmgr +44 -0
- package/hooks/hooks.harness.mjs +713 -0
- package/hooks/hooks.mutations.mjs +337 -0
- package/hooks/paper-edit-guard.hook.d.mts +13 -0
- package/hooks/paper-edit-guard.hook.mjs +457 -0
- package/hooks/paper-skills-nudge.hook.mjs +136 -0
- package/hooks/paper-status-gates.hook.mjs +156 -0
- package/hooks/paper-status-gates.sh +91 -0
- package/lib/agent-cli-version.harness.mjs +165 -0
- package/lib/agent-cli-version.mjs +106 -0
- package/lib/agent-cli-version.mutations.mjs +109 -0
- package/lib/markdown.mjs +386 -0
- package/lib/mutation-driver.harness.mjs +227 -0
- package/lib/mutation-driver.mjs +397 -0
- package/lib/mutation-driver.mutations.mjs +68 -0
- package/lib/paper-config.d.mts +34 -0
- package/lib/paper-config.harness.mjs +286 -0
- package/lib/paper-config.mjs +142 -0
- package/lib/paper-config.mutations.mjs +143 -0
- package/lib/skill-checks.mjs +701 -0
- package/lib/skill-corpus.mjs +403 -0
- package/lib/skill-eval-fixture.mjs +63 -0
- package/lib/skill-eval-kit.mjs +257 -0
- package/lib/skill-trigger-cases.harness.mjs +170 -0
- package/lib/skill-trigger-cases.mjs +446 -0
- package/lib/skill-trigger-cases.mutations.mjs +65 -0
- package/lib/trigger-ledger.mjs +215 -0
- package/package.json +97 -0
- package/plugin/.claude-plugin/plugin.json +8 -0
- package/plugin/hooks/hooks.json +30 -0
- package/scripts/check.harness.mjs +177 -0
- package/scripts/check.mjs +239 -0
- package/scripts/check.mutations.mjs +110 -0
- package/scripts/eslint-report-guard.mjs +82 -0
- package/scripts/exclusive.mjs +138 -0
- package/scripts/harness-api.frozen.json +76 -0
- package/scripts/harness-api.test.ts +175 -0
- package/scripts/layer-legacy-frozen.d.mts +28 -0
- package/scripts/layer-legacy-frozen.mjs +152 -0
- package/scripts/layer-legacy-frozen.test.ts +115 -0
- package/scripts/layer-legacy.frozen.json +50 -0
- package/scripts/mutation-batteries-frozen.harness.mjs +204 -0
- package/scripts/mutation-batteries-frozen.mjs +238 -0
- package/scripts/mutation-batteries.frozen.json +117 -0
- package/scripts/release-config.test.ts +90 -0
- package/scripts/rules-are-content-only.harness.mjs +113 -0
- package/scripts/rules-are-content-only.mjs +138 -0
- package/scripts/rules-are-content-only.mutations.mjs +81 -0
- package/scripts/rules-see-files.harness.mjs +115 -0
- package/scripts/rules-see-files.mjs +99 -0
- package/scripts/rules-see-files.mutations.mjs +131 -0
- package/scripts/run-mutations.mjs +100 -0
- package/scripts/semantic-release-plugins.d.ts +16 -0
- package/skills/README.md +15 -0
- package/skills/analyze-sibling-paper/SKILL.md +170 -0
- package/skills/analyze-sibling-paper/SKILL.md.spec.ts +186 -0
- package/skills/analyze-sibling-paper/analyze-sibling-paper.eval.mjs +19 -0
- package/skills/analyze-sibling-paper/analyze-sibling-paper.harness.mjs +23 -0
- package/skills/argument-arc/SKILL.md +177 -0
- package/skills/argument-arc/SKILL.md.spec.ts +192 -0
- package/skills/argument-arc/argument-arc.eval.mjs +19 -0
- package/skills/argument-arc/argument-arc.harness.mjs +23 -0
- package/skills/build-benchmark/SKILL.md +213 -0
- package/skills/build-benchmark/SKILL.md.spec.ts +220 -0
- package/skills/build-benchmark/build-benchmark.eval.mjs +19 -0
- package/skills/build-benchmark/build-benchmark.harness.mjs +23 -0
- package/skills/build-benchmark/references/adversarial-cold-repro.md +68 -0
- package/skills/camera-ready/SKILL.md +148 -0
- package/skills/camera-ready/SKILL.md.spec.ts +164 -0
- package/skills/camera-ready/camera-ready.eval.mjs +19 -0
- package/skills/camera-ready/camera-ready.harness.mjs +23 -0
- package/skills/cold-read-diff/SKILL.md +160 -0
- package/skills/cold-read-diff/SKILL.md.spec.ts +166 -0
- package/skills/cold-read-diff/cold-read-diff.eval.mjs +19 -0
- package/skills/cold-read-diff/cold-read-diff.harness.mjs +23 -0
- package/skills/draft-paper/SKILL.md +152 -0
- package/skills/draft-paper/SKILL.md.spec.ts +169 -0
- package/skills/draft-paper/draft-paper.eval.mjs +19 -0
- package/skills/draft-paper/draft-paper.harness.mjs +23 -0
- package/skills/extend-paper/SKILL.md +99 -0
- package/skills/extend-paper/SKILL.md.spec.ts +116 -0
- package/skills/extend-paper/extend-paper.eval.mjs +19 -0
- package/skills/extend-paper/extend-paper.harness.mjs +23 -0
- package/skills/find-venue/SKILL.md +128 -0
- package/skills/find-venue/SKILL.md.spec.ts +145 -0
- package/skills/find-venue/find-venue.eval.mjs +19 -0
- package/skills/find-venue/find-venue.harness.mjs +23 -0
- package/skills/grade-paper-writing/SKILL.md +436 -0
- package/skills/grade-paper-writing/SKILL.md.spec.ts +453 -0
- package/skills/grade-paper-writing/fixtures/control_gopen.txt +1 -0
- package/skills/grade-paper-writing/fixtures/control_human_paper.txt +1 -0
- package/skills/grade-paper-writing/fixtures/rewrite.txt +1 -0
- package/skills/grade-paper-writing/fixtures/specimen.txt +1 -0
- package/skills/grade-paper-writing/fixtures/structure-checks.md +22 -0
- package/skills/grade-paper-writing/grade-paper-writing.eval.mjs +19 -0
- package/skills/grade-paper-writing/grade-paper-writing.harness.mjs +23 -0
- package/skills/grade-paper-writing/prose-lint.mjs +713 -0
- package/skills/harden-paper/SKILL.md +318 -0
- package/skills/harden-paper/SKILL.md.spec.ts +336 -0
- package/skills/harden-paper/check-numbers.sh +33 -0
- package/skills/harden-paper/check-release-claims.sh +35 -0
- package/skills/harden-paper/fixtures/uncited-assertions-sample.md +43 -0
- package/skills/harden-paper/fixtures/uncited-assertions-sample.tex +77 -0
- package/skills/harden-paper/harden-paper.eval.mjs +19 -0
- package/skills/harden-paper/harden-paper.harness.mjs +23 -0
- package/skills/map-prior-work/SKILL.md +211 -0
- package/skills/map-prior-work/SKILL.md.spec.ts +227 -0
- package/skills/map-prior-work/map-prior-work.eval.mjs +19 -0
- package/skills/map-prior-work/map-prior-work.harness.mjs +23 -0
- package/skills/osf-artifact-upload/SKILL.md +52 -0
- package/skills/osf-artifact-upload/SKILL.md.spec.ts +59 -0
- package/skills/osf-artifact-upload/osf-artifact-upload.eval.mjs +22 -0
- package/skills/osf-artifact-upload/osf-artifact-upload.harness.mjs +103 -0
- package/skills/paper-adversarial-review/SKILL.md +126 -0
- package/skills/paper-adversarial-review/SKILL.md.spec.ts +142 -0
- package/skills/paper-adversarial-review/paper-adversarial-review.eval.mjs +19 -0
- package/skills/paper-adversarial-review/paper-adversarial-review.harness.mjs +23 -0
- package/skills/paper-pipeline/PIPELINE-MAP.md +371 -0
- package/skills/paper-pipeline/SKILL.md +499 -0
- package/skills/paper-pipeline/SKILL.md.spec.ts +517 -0
- package/skills/paper-pipeline/description-language.eval.mjs +347 -0
- package/skills/paper-pipeline/framing-vs-vocabulary.eval.mjs +891 -0
- package/skills/paper-pipeline/grade-paper-writing-ablation.eval.mjs +1254 -0
- package/skills/paper-pipeline/paper-pipeline.eval.mjs +22 -0
- package/skills/paper-pipeline/paper-pipeline.harness.mjs +143 -0
- package/skills/paper-pipeline/pipeline-firing.baseline.json +270 -0
- package/skills/paper-pipeline/pipeline-firing.eval.mjs +664 -0
- package/skills/paper-pipeline/pipeline-language.eval.mjs +672 -0
- package/skills/paper-pipeline/references/acceptance-gate.md +329 -0
- package/skills/paper-pipeline/references/acl-venue-rules.md +142 -0
- package/skills/paper-pipeline/references/anonymization.md +68 -0
- package/skills/paper-pipeline/references/artifact-checklist.md +93 -0
- package/skills/paper-pipeline/references/body-vs-appendix.md +97 -0
- package/skills/paper-pipeline/references/credit-criteria.md +69 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/README.md +35 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/run_retext.mjs +24 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/sentences.txt +11 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-probe/test_sentences.py +25 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-prose-checkers.md +538 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-reproducible-tooling.md +431 -0
- package/skills/paper-pipeline/references/occupancy-2026-08-06-staleness-and-orchestration.md +592 -0
- package/skills/paper-pipeline/references/pipeline-status-template.md +162 -0
- package/skills/paper-pipeline/references/review-ratchet.md +36 -0
- package/skills/paper-pipeline/references/sweep-2026-08-09-ideal-pipeline.md +585 -0
- package/skills/paper-pipeline/references/writing-craft.md +448 -0
- package/skills/paper-pipeline/repro/2026-08-07-description-language-control.log +63 -0
- package/skills/paper-pipeline/repro/2026-08-07-fork-check.log +52 -0
- package/skills/paper-pipeline/repro/2026-08-07-fork-check2.log +33 -0
- package/skills/paper-pipeline/repro/2026-08-07-language-eval-pilot.log +33 -0
- package/skills/paper-pipeline/repro/2026-08-07-language-eval-raw.log +166 -0
- package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-oracle.json +338 -0
- package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-oracle.log +118 -0
- package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary-raw.log +245 -0
- package/skills/paper-pipeline/repro/2026-08-08-framing-vs-vocabulary.json +776 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-A6-oracle.log +53 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-A6-raw.log +89 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-oracle.json +450 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-oracle.log +136 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-raw.log +242 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-setupdiff.log +59 -0
- package/skills/paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation.json +1032 -0
- package/skills/paper-pipeline/repro/2026-08-08-parent-replication-gpw.json +139 -0
- package/skills/paper-pipeline/repro/2026-08-08-parent-replication-gpw.log +98 -0
- package/skills/paper-pipeline/repro/2026-08-08-parent-replication.mjs +92 -0
- package/skills/paper-pipeline/repro/README.md +129 -0
- package/skills/paper-pipeline/repro/analyze-language-eval.py +116 -0
- package/skills/paper-pipeline/scripts/README.md +344 -0
- package/skills/paper-pipeline/scripts/announce.mjs +67 -0
- package/skills/paper-pipeline/scripts/artifact-coverage.harness.mjs +496 -0
- package/skills/paper-pipeline/scripts/artifact-coverage.mjs +397 -0
- package/skills/paper-pipeline/scripts/artifact-coverage.mutations.mjs +218 -0
- package/skills/paper-pipeline/scripts/check-provenance.mjs +184 -0
- package/skills/paper-pipeline/scripts/consumer.d.mts +32 -0
- package/skills/paper-pipeline/scripts/consumer.harness.mjs +562 -0
- package/skills/paper-pipeline/scripts/consumer.mjs +535 -0
- package/skills/paper-pipeline/scripts/consumer.mutations.mjs +190 -0
- package/skills/paper-pipeline/scripts/extract-ref-facts.harness.mjs +457 -0
- package/skills/paper-pipeline/scripts/extract-ref-facts.mjs +656 -0
- package/skills/paper-pipeline/scripts/extract-ref-facts.mutations.mjs +54 -0
- package/skills/paper-pipeline/scripts/fixtures/clean/PIPELINE-STATUS.md +51 -0
- package/skills/paper-pipeline/scripts/fixtures/dirty/PIPELINE-STATUS.md +52 -0
- package/skills/paper-pipeline/scripts/fixtures/dirty/paper.md +6 -0
- package/skills/paper-pipeline/scripts/fixtures/real-bib/refs.bib +153 -0
- package/skills/paper-pipeline/scripts/generated-code.harness.mjs +466 -0
- package/skills/paper-pipeline/scripts/generated-code.mjs +338 -0
- package/skills/paper-pipeline/scripts/generated-code.mutations.mjs +254 -0
- package/skills/paper-pipeline/scripts/ledger.mjs +623 -0
- package/skills/paper-pipeline/scripts/ledger.selftest.mjs +286 -0
- package/skills/paper-pipeline/scripts/pipeline-check.harness.mjs +389 -0
- package/skills/paper-pipeline/scripts/pipeline-check.mjs +737 -0
- package/skills/paper-pipeline/scripts/pipeline-check.mutations.mjs +54 -0
- package/skills/paper-pipeline/scripts/pipeline-edges.mjs +169 -0
- package/skills/paper-pipeline/scripts/population-map.harness.mjs +178 -0
- package/skills/paper-pipeline/scripts/population-map.mjs +181 -0
- package/skills/paper-pipeline/scripts/population-map.mutations.mjs +65 -0
- package/skills/paper-pipeline/scripts/population-map.selftest.mjs +122 -0
- package/skills/paper-pipeline/scripts/provenance.harness.mjs +240 -0
- package/skills/paper-pipeline/scripts/provenance.mutations.mjs +59 -0
- package/skills/paper-pipeline/scripts/round-diff.harness.mjs +881 -0
- package/skills/paper-pipeline/scripts/round-diff.mjs +576 -0
- package/skills/paper-pipeline/scripts/round-diff.mutations.mjs +276 -0
- package/skills/paper-pipeline/scripts/run-mechanical.mjs +633 -0
- package/skills/paper-pipeline/scripts/status.mjs +295 -0
- package/skills/paper-status/SKILL.md +183 -0
- package/skills/paper-status/SKILL.md.spec.ts +190 -0
- package/skills/paper-status/paper-status.eval.mjs +22 -0
- package/skills/paper-status/paper-status.harness.mjs +25 -0
- package/skills/pc-panel-review/SKILL.md +263 -0
- package/skills/pc-panel-review/SKILL.md.spec.ts +280 -0
- package/skills/pc-panel-review/pc-panel-review.eval.mjs +19 -0
- package/skills/pc-panel-review/pc-panel-review.harness.mjs +23 -0
- package/skills/plan-paper-timeline/SKILL.md +182 -0
- package/skills/plan-paper-timeline/SKILL.md.spec.ts +200 -0
- package/skills/plan-paper-timeline/fixtures/fake-google-calendar.mjs +239 -0
- package/skills/plan-paper-timeline/plan-paper-timeline.effects.harness.mjs +431 -0
- package/skills/plan-paper-timeline/plan-paper-timeline.effects.mutations.mjs +65 -0
- package/skills/plan-paper-timeline/plan-paper-timeline.eval.mjs +19 -0
- package/skills/plan-paper-timeline/plan-paper-timeline.harness.mjs +23 -0
- package/skills/render-paper/SKILL.md +159 -0
- package/skills/render-paper/SKILL.md.spec.ts +166 -0
- package/skills/render-paper/check-render.sh +419 -0
- package/skills/render-paper/checkers-requirements.txt +55 -0
- package/skills/render-paper/ensure-checkers.sh +69 -0
- package/skills/render-paper/extract-pdf-facts.harness.mjs +166 -0
- package/skills/render-paper/extract-pdf-facts.mjs +144 -0
- package/skills/render-paper/render-paper.eval.mjs +19 -0
- package/skills/render-paper/render-paper.harness.mjs +339 -0
- package/skills/research-ideate/SKILL.md +136 -0
- package/skills/research-ideate/SKILL.md.spec.ts +152 -0
- package/skills/research-ideate/research-ideate.eval.mjs +19 -0
- package/skills/research-ideate/research-ideate.harness.mjs +23 -0
- package/skills/skill-contract.mutations.mjs +179 -0
- package/skills/study-accepted-papers/SKILL.md +206 -0
- package/skills/study-accepted-papers/SKILL.md.spec.ts +223 -0
- package/skills/study-accepted-papers/study-accepted-papers.eval.mjs +19 -0
- package/skills/study-accepted-papers/study-accepted-papers.harness.mjs +23 -0
- package/skills/submit-paper/SKILL.md +182 -0
- package/skills/submit-paper/SKILL.md.spec.ts +199 -0
- package/skills/submit-paper/check-deanon.sh +149 -0
- package/skills/submit-paper/references/publishers/acm.md +92 -0
- package/skills/submit-paper/references/venues/agenticdev.jsonc +108 -0
- package/skills/submit-paper/references/venues/agenticdev.md +139 -0
- package/skills/submit-paper/references/venues/agenticdev.tex +19 -0
- package/skills/submit-paper/references/venues/aisec.jsonc +101 -0
- package/skills/submit-paper/references/venues/aisec.md +105 -0
- package/skills/submit-paper/references/venues/paper-guards.tex +41 -0
- package/skills/submit-paper/references/venues/realm.jsonc +81 -0
- package/skills/submit-paper/references/venues/realm.md +155 -0
- package/skills/submit-paper/references/venues/tex-base.jsonc +50 -0
- package/skills/submit-paper/references/venues/venue-profile.schema.json +74 -0
- package/skills/submit-paper/submit-paper.eval.mjs +19 -0
- package/skills/submit-paper/submit-paper.harness.mjs +23 -0
- package/skills/sweep-design-space/SKILL.md +269 -0
- package/skills/sweep-design-space/SKILL.md.spec.ts +285 -0
- package/skills/sweep-design-space/sweep-design-space.eval.mjs +19 -0
- package/skills/sweep-design-space/sweep-design-space.harness.mjs +23 -0
- package/skills/tighten-paper/SKILL.md +368 -0
- package/skills/tighten-paper/SKILL.md.spec.ts +384 -0
- package/skills/tighten-paper/structure.mjs +371 -0
- package/skills/tighten-paper/tighten-paper.eval.mjs +19 -0
- package/skills/tighten-paper/tighten-paper.harness.mjs +23 -0
- package/skills/verify-citations/SKILL.md +328 -0
- package/skills/verify-citations/SKILL.md.spec.ts +345 -0
- package/skills/verify-citations/scripts/bib-authors.mjs +479 -0
- package/skills/verify-citations/scripts/bib-authors.test.mjs +175 -0
- package/skills/verify-citations/scripts/verify-cites.mjs +1108 -0
- package/skills/verify-citations/scripts/verify-cites.test.mjs +735 -0
- package/skills/verify-citations/verify-citations.eval.mjs +19 -0
- package/skills/verify-citations/verify-citations.harness.mjs +23 -0
- package/src/CLAUDE.md +51 -0
- package/src/action-ref.test.ts +26 -0
- package/src/action-ref.ts +15 -0
- package/src/adapters/banal/failure.test.ts +63 -0
- package/src/adapters/banal/failure.ts +118 -0
- package/src/adapters/banal/index.test.ts +119 -0
- package/src/adapters/banal/index.ts +100 -0
- package/src/adapters/banal/install.test.ts +20 -0
- package/src/adapters/banal/install.ts +41 -0
- package/src/adapters/banal/invocation.test.ts +74 -0
- package/src/adapters/banal/invocation.ts +95 -0
- package/src/adapters/banal/locate.test.ts +52 -0
- package/src/adapters/banal/locate.ts +84 -0
- package/src/adapters/banal/output.test.ts +140 -0
- package/src/adapters/banal/output.ts +141 -0
- package/src/adapters/banal/pin.ts +30 -0
- package/src/adapters/banal/probe.ts +35 -0
- package/src/adapters/banal/run.test.ts +191 -0
- package/src/adapters/banal/run.ts +244 -0
- package/src/adapters/banal/settings.test.ts +31 -0
- package/src/adapters/banal/settings.ts +55 -0
- package/src/adapters/banal/xml.test.ts +111 -0
- package/src/adapters/banal/xml.ts +112 -0
- package/src/adapters/curl/download.io.ts +73 -0
- package/src/adapters/curl/download.test.ts +55 -0
- package/src/adapters/curl/index.ts +5 -0
- package/src/adapters/memory/index.ts +131 -0
- package/src/adapters/node/files.io.ts +39 -0
- package/src/adapters/node/files.test.ts +28 -0
- package/src/adapters/node/host.io.ts +15 -0
- package/src/adapters/node/index.ts +36 -0
- package/src/adapters/node/process.io.ts +49 -0
- package/src/adapters/node/process.test.ts +46 -0
- package/src/adapters/node/workspace.io.ts +40 -0
- package/src/adapters/node/workspace.test.ts +58 -0
- package/src/adapters/pdfjs/fill.test.ts +111 -0
- package/src/adapters/pdfjs/fill.ts +141 -0
- package/src/build-engine.harness.mjs +314 -0
- package/src/build-engine.ts +219 -0
- package/src/build.harness.mjs +631 -0
- package/src/build.mutations.mjs +195 -0
- package/src/build.ts +793 -0
- package/src/cli.harness.mjs +2007 -0
- package/src/cli.mutations.mjs +448 -0
- package/src/cli.ts +1189 -0
- package/src/doctor.harness.mjs +396 -0
- package/src/doctor.mutations.mjs +175 -0
- package/src/doctor.ts +356 -0
- package/src/domain/geometry.ts +108 -0
- package/src/domain/host.ts +23 -0
- package/src/domain/page-layout.ts +32 -0
- package/src/domain/paths.ts +5 -0
- package/src/domain/result.test.ts +26 -0
- package/src/domain/result.ts +29 -0
- package/src/domain/sha256.test.ts +12 -0
- package/src/domain/sha256.ts +21 -0
- package/src/domain/text.ts +11 -0
- package/src/engine.harness.mjs +252 -0
- package/src/engine.ts +176 -0
- package/src/exit-code.test.ts +21 -0
- package/src/exit-code.ts +38 -0
- package/src/facts-file.test.ts +240 -0
- package/src/facts-file.ts +241 -0
- package/src/hooks-settings.harness.mjs +386 -0
- package/src/hooks-settings.mutations.mjs +116 -0
- package/src/hooks-settings.ts +434 -0
- package/src/init.ts +900 -0
- package/src/latex-log.harness.mjs +226 -0
- package/src/latex-log.ts +234 -0
- package/src/latex-loop.harness.mjs +449 -0
- package/src/latex-loop.ts +211 -0
- package/src/link-skills.harness.mjs +273 -0
- package/src/link-skills.mutations.mjs +136 -0
- package/src/link-skills.ts +258 -0
- package/src/new-paper.harness.mjs +216 -0
- package/src/new-paper.mutations.mjs +79 -0
- package/src/new-paper.ts +158 -0
- package/src/pdf-facts.harness.mjs +188 -0
- package/src/pdf-facts.ts +327 -0
- package/src/pdf-geometry.harness.mjs +254 -0
- package/src/pdf-geometry.ts +300 -0
- package/src/ports/download.ts +10 -0
- package/src/ports/files.ts +11 -0
- package/src/ports/measure-geometry.ts +8 -0
- package/src/ports/process.ts +46 -0
- package/src/ports/tool-installer.ts +33 -0
- package/src/ports/workspace.ts +20 -0
- package/src/rules-config.harness.mjs +114 -0
- package/src/rules-config.ts +178 -0
- package/src/structure.harness.mjs +179 -0
- package/src/structure.mutations.mjs +83 -0
- package/src/structure.ts +166 -0
- package/src/tex-requirements.harness.mjs +238 -0
- package/src/tex-requirements.ts +181 -0
- package/src/toolchain.harness.mjs +651 -0
- package/src/toolchain.ts +755 -0
- package/src/types.ts +106 -0
- package/templates/paper/PIPELINE-STATUS.md +72 -0
- package/templates/paper/paper.md +4 -0
- package/templates/paper/paper.tex +8 -0
- package/tsconfig.json +23 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* paper-status — the PAID tier: does this skill's description actually fire?
|
|
3
|
+
*
|
|
4
|
+
* COLOCATED ON PURPOSE (vigiles decides coverage by placement as of 2026-08-11).
|
|
5
|
+
* The prompts live in `.claude/lib/skill-trigger-cases.mjs` so every case is
|
|
6
|
+
* reviewed as one table where collisions between siblings are visible; copying
|
|
7
|
+
* them here would recreate the drift that rule exists to prevent.
|
|
8
|
+
*
|
|
9
|
+
* Measures recall (fires on its own territory) AND precision (stays quiet on a
|
|
10
|
+
* colliding sibling's territory), against the REAL `.claude` harness so the skill
|
|
11
|
+
* competes with every other installed description — an isolated run overstates
|
|
12
|
+
* recall and understates false positives.
|
|
13
|
+
*
|
|
14
|
+
* ⚠️ NEVER RUN. This case was written 2026-08-11 with the other two the coverage
|
|
15
|
+
* sweep surfaced; its rate is UNKNOWN, not assumed good.
|
|
16
|
+
*
|
|
17
|
+
* Costs money; not CI.
|
|
18
|
+
* node .claude/skills/paper-status/paper-status.eval.mjs [trials]
|
|
19
|
+
*/
|
|
20
|
+
import { runSkillTriggerEval } from "../../lib/skill-eval-kit.mjs";
|
|
21
|
+
|
|
22
|
+
await runSkillTriggerEval("paper-status");
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* paper-status — the free, deterministic tier. No model, no network.
|
|
3
|
+
*
|
|
4
|
+
* 🔴 THIS SKILL WAS NOT MERELY UNTESTED — IT WAS UNREACHABLE BY THE HARNESS.
|
|
5
|
+
* `skill-corpus.mjs` carried `paper-status` in an EXCLUDED map whose recorded
|
|
6
|
+
* reason is "it has no gate row", a statement about ONE assertion. The code
|
|
7
|
+
* filtered it out of the checked set ENTIRELY, so it also skipped the strict-YAML
|
|
8
|
+
* frontmatter parse, the tool contract, the script-paths check and the identity
|
|
9
|
+
* check. Adding a file here would have thrown ("not a wired pipeline skill")
|
|
10
|
+
* rather than testing anything. Fixed 2026-08-11: the exclusion now applies at
|
|
11
|
+
* assertion 12b alone, and the sweep went 21 skills to 22.
|
|
12
|
+
*
|
|
13
|
+
* That is the general shape worth remembering: a surface can read as untested
|
|
14
|
+
* when the missing piece is not a test file but a filter upstream of it.
|
|
15
|
+
*
|
|
16
|
+
* The assertions live in `.claude/lib/skill-checks.mjs` and are CALLED with this
|
|
17
|
+
* skill's name — not copied. 22 copies of the same checks is the drift that module
|
|
18
|
+
* exists to prevent.
|
|
19
|
+
*
|
|
20
|
+
* What it does NOT prove: that the skill fires, or that its report is right.
|
|
21
|
+
* See `paper-status.eval.mjs`.
|
|
22
|
+
*/
|
|
23
|
+
import { checkSkill } from "../../lib/skill-checks.mjs";
|
|
24
|
+
|
|
25
|
+
await checkSkill("paper-status");
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: pc-panel-review
|
|
3
|
+
description: Use when asking "what would the program committee decide?" / "simulate the reviewers" / "what's this paper's accept probability?" on a drafted paper + artifact. Spawns N independent reviewers with DISTINCT lenses (one actually RUNS the artifact), then synthesizes a PC-chair meta-review into an accept/reject decision, a calibrated probability, and consensus must-fixes. Requires grade-paper-writing's persona stall inventory + tighten-paper's structural verdict as inputs (blocked until they exist — the panel cannot feel reader fatigue on its own). Also has a lightweight single-reviewer VENUE-FIT MODE for a quick CFP-fit spot-check. NOT a single fast defect hunt (paper-adversarial-review), a writing grade (grade-paper-writing), or the full pre-submit gate (harden-paper, which calls this) — this models the PC decision itself.
|
|
4
|
+
allowed-tools: [Read, Write, Grep, Glob, Bash, WebSearch, WebFetch, Agent, Skill]
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
<!-- vigiles:sha256:9622701a349ea249 compiled from skills/pc-panel-review/SKILL.md.spec.ts -->
|
|
8
|
+
|
|
9
|
+
# pc-panel-review — model the whole PC, not one reviewer
|
|
10
|
+
|
|
11
|
+
> **Which review skill?** `paper-adversarial-review` = one hostile reviewer, fast defect hunt.
|
|
12
|
+
> `pc-panel-review` (you are here) = the whole PC (N independent lenses incl. an artifact-runner) + a
|
|
13
|
+
> chair meta-review — the real pre-submission GATE, run LAST. This skill also has a **Venue-fit mode**
|
|
14
|
+
> (below): ONE reviewer scoring a named venue's CFP rubric to predict its accept/reject — the quick
|
|
15
|
+
> single-lens spot-check (formerly the standalone `venue-review-sim` skill). Reach for Venue-fit mode or
|
|
16
|
+
> `paper-adversarial-review` for a fast spot-check; run the full panel (this) for the decision.
|
|
17
|
+
|
|
18
|
+
## Run me
|
|
19
|
+
|
|
20
|
+
🔴 FIRST, before any other step:
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
node .claude/skills/paper-pipeline/scripts/announce.mjs pc-panel-review <paper-dir>
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
An advisory pass cannot be observed failing — silence is both its error state and its normal
|
|
27
|
+
state — so starting is an event, and events get written down.
|
|
28
|
+
|
|
29
|
+
The Venue-fit mode gives ONE reviewer's fit score; `paper-adversarial-review` gives ONE defect hunt.
|
|
30
|
+
A real decision is made by **2–4 reviewers with different priorities + a meta-review** that weighs
|
|
31
|
+
consensus over any single voice. This skill runs that. The payoff over a single review: a **must-fix
|
|
32
|
+
is what ≥2 independent reviewers flag** — that filters real blockers from one reviewer's hobbyhorse —
|
|
33
|
+
and at least one reviewer **executes the artifact**, which surfaces paper↔artifact mismatches no
|
|
34
|
+
prose-only read can.
|
|
35
|
+
|
|
36
|
+
## How to run it
|
|
37
|
+
|
|
38
|
+
1. **Get the venue rubric + bar** (fetch the CFP; quote the criteria). Calibrate to the *track*: a
|
|
39
|
+
WIP/short/workshop bar rewards promise and discussion value; a full-paper/top-tier bar demands
|
|
40
|
+
completeness. Do not import top-tier standards into a WIP review.
|
|
41
|
+
2. **Collect the two readability/structure artifacts — the panel is BLOCKED without them.** Before any
|
|
42
|
+
reviewer spawns, there must exist for THIS draft: (a) the **persona stall inventory** from
|
|
43
|
+
`grade-paper-writing` (the "Sam" per-section cold read), and (b) the **structural verdict** from
|
|
44
|
+
`tighten-paper` (length / sag / TMI / cut-plan). If either doesn't exist yet, STOP and run those
|
|
45
|
+
skills first.
|
|
46
|
+
|
|
47
|
+
🔴 **Invoke them BY NAME with the Skill tool — do not describe the need and hope.** Whether
|
|
48
|
+
`grade-paper-writing` gets selected from its description is **unstable**, and that is worse for a
|
|
49
|
+
hard gate than a low rate would be: identical prompts, identical description, identical roster
|
|
50
|
+
measured **21%** in the morning of 2026-08-08 and **50–54%** the same afternoon, in two
|
|
51
|
+
independently written implementations
|
|
52
|
+
(`../paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-raw.log`,
|
|
53
|
+
`2026-08-08-parent-replication.log`). Seven rewrites of the description — shorter, negations
|
|
54
|
+
stripped, machinery stripped, persona-only — all landed inside noise, so there is nothing to fix
|
|
55
|
+
in the text. Naming the skill explicitly does not depend on selection at all, so this gate stops
|
|
56
|
+
resting on the one part of the chain that does. Writing "run grade-paper-writing" and leaving the
|
|
57
|
+
invocation to a description match is the failure this line exists to remove.
|
|
58
|
+
|
|
59
|
+
Why hard-block: a frontier-LLM reviewer knows every term and cannot feel reader
|
|
60
|
+
fatigue, so a panel scoring on its own felt read will score a bloated, unreadable paper high
|
|
61
|
+
(proven: ~88% accept on a paper the human reader found exhausting). These two artifacts are the
|
|
62
|
+
panel's only legitimate source for the readability/structure signal.
|
|
63
|
+
*(Scorecard: this gate is row **`panel`**, and it declares **`structure`** (tighten-paper) + **`writing`**
|
|
64
|
+
(grade-paper-writing) as its required inputs — `pipeline-check.mjs` reports `panel` marked done while
|
|
65
|
+
either input is missing or newer than it.)*
|
|
66
|
+
3. **Pick 3 (or 4) reviewer lenses** that a real PC for THIS venue would field. Menu — choose by fit:
|
|
67
|
+
- **Methods/stats skeptic** — sample size, construct validity, multiple comparisons, overclaims,
|
|
68
|
+
whether the CIs support the claims.
|
|
69
|
+
- **Domain practitioner** — relevance to the CFP, novelty vs the nearest prior work, does the
|
|
70
|
+
finding change behavior, generalizability, fairness/tone toward anyone named.
|
|
71
|
+
- **Artifact / reproducibility reviewer (MANDATORY if an artifact is attached)** — *actually
|
|
72
|
+
`cd` into the artifact and run it*; verify every headline number reproduces; read the code for
|
|
73
|
+
circularity (echoing stored fields vs recomputing); grep all files (incl. data + compiled
|
|
74
|
+
caches) for anonymization leaks; scope the badge (Available / Functional / Reproduced).
|
|
75
|
+
- **Security/ethics reviewer** (security venues) — threat model soundness, responsible disclosure,
|
|
76
|
+
punching-down risk.
|
|
77
|
+
- **Novelty/related-work reviewer** — is the delta over the nearest neighbor explicit; obvious
|
|
78
|
+
uncited work.
|
|
79
|
+
4. **Spawn them in PARALLEL, each on a separate model instance** (e.g. Fable via subagent), each blind
|
|
80
|
+
to the others — independence is the whole point. Give each: the paper (`.tex`/PDF), the artifact
|
|
81
|
+
path, the quoted rubric, its lens, **and the two step-2 artifacts (persona stall inventory +
|
|
82
|
+
tighten-paper structural verdict)**. Force a filled review form (below).
|
|
83
|
+
5. **Synthesize the meta-review yourself** (PC chair) once all land — see output.
|
|
84
|
+
|
|
85
|
+
## Each reviewer returns a review form
|
|
86
|
+
- Per-criterion score on the venue's scale (map to 1–5 if unstated), one line each, grounded.
|
|
87
|
+
- Overall rec (Reject … Strong Accept) + confidence (1–5).
|
|
88
|
+
- **Readability's implicit drag — model the halo effect, DRIVEN BY THE PERSONA INVENTORY, not felt
|
|
89
|
+
judgment.** A real reviewer who finds the paper a slog silently loses confidence in the *science* and
|
|
90
|
+
lowers the OVERALL score, while writing "methodology concerns" in the box — the penalty rarely shows
|
|
91
|
+
up as an explicit "clarity" mark. Each reviewer must let prose/structure quality move their overall
|
|
92
|
+
rec and confidence — but the SOURCE of that signal is the handed-in **persona stall inventory +
|
|
93
|
+
tighten-paper verdict**, never the reviewer's own felt read: a frontier LLM knows every term and
|
|
94
|
+
cannot experience reader fatigue, so its felt judgment systematically under-fires. If the inventory
|
|
95
|
+
shows high stall density / walls, or the structural verdict says bloated, the overall score takes a
|
|
96
|
+
real hit even when every explicit criterion passes. Never score a slog as if it read cleanly — that
|
|
97
|
+
is the gap between this sim and a real PC.
|
|
98
|
+
- Top 3 issues for PC discussion, ranked, with severity (blocker/major/minor).
|
|
99
|
+
- Any must-fix before *they'd* accept (or "none").
|
|
100
|
+
- Artifact reviewer only: run outcome (does it execute? every number reproduce? PASS/FAIL per check),
|
|
101
|
+
anonymization verdict, badge scope.
|
|
102
|
+
- One-line meta: does this belong at this venue?
|
|
103
|
+
|
|
104
|
+
## The meta-review (PC-chair synthesis) — the actual deliverable
|
|
105
|
+
- **Score table:** each reviewer × each criterion + overall + confidence, at a glance.
|
|
106
|
+
- **Decision:** the rec the PC discussion converges on (weight by confidence; a lone low-confidence
|
|
107
|
+
outlier doesn't sink two confident accepts), + a **calibrated accept probability** for this venue
|
|
108
|
+
and edition.
|
|
109
|
+
- **Consensus must-fixes:** issues **≥2 reviewers independently raised** — these are the real ones;
|
|
110
|
+
apply before submitting.
|
|
111
|
+
- **Single-reviewer flags:** noted, triaged (fix if cheap, else defer to camera-ready).
|
|
112
|
+
- **Split calls:** where reviewers genuinely disagree, and which way the chair leans + why.
|
|
113
|
+
- **Pre-submission checklist:** the surgical edits, ranked, that move the probability most.
|
|
114
|
+
|
|
115
|
+
## 🔴 The ceiling is not a sentence — it is a PLAN, and the panel writes it
|
|
116
|
+
|
|
117
|
+
**The corpus owner, 2026-08-05: "weak accept doesn't work for us. skills should suggest what to do to fix
|
|
118
|
+
the situation."** He is right that this was missing. Three panels in a row named the same ceiling, it was
|
|
119
|
+
faithfully recorded three times, and it never once became work. The fourth panel named it again — and
|
|
120
|
+
the author wrote all three of its prongs off as *"facts, not worth chasing before the deadline"*. Two
|
|
121
|
+
of the three were closed by an hour of editing that same afternoon.
|
|
122
|
+
|
|
123
|
+
So a panel that reports Weak Accept or Borderline **must** end with a ceiling plan: every prong
|
|
124
|
+
classified, no exceptions, using exactly these three labels.
|
|
125
|
+
|
|
126
|
+
| label | means | what the panel owes |
|
|
127
|
+
|---|---|---|
|
|
128
|
+
| **TEXT** | an edit closes it — the evidence already exists and the paper fails to join it up, frame it, or show the reader what the alternative would look like | the actual sentence or paragraph to write, and where it goes |
|
|
129
|
+
| **EXPERIMENT** | only new measurement closes it | what would have to be run, and a rough cost, so the decision to skip is informed |
|
|
130
|
+
| **IMMOVABLE** | it is what the work *is* — the scope, the population, the yield | say so plainly, so nobody re-opens it next round |
|
|
131
|
+
|
|
132
|
+
**Why the labels and not prose.** "This is a fact about the work" is the cheapest thing a reviewer can
|
|
133
|
+
write and the hardest to argue with, so it is where an unwilling author hides. Forcing a label makes
|
|
134
|
+
the claim falsifiable: *TEXT* invites "then write it", and *IMMOVABLE* invites "is it really?".
|
|
135
|
+
|
|
136
|
+
**Two failure shapes to check yourself against**, both observed on `the reference paper`:
|
|
137
|
+
|
|
138
|
+
- **A number reported as a rate when it is a floor.** "8 contradictions in 1,836 repositories" read as
|
|
139
|
+
a small phenomenon; the same paper measured its own extractor's recall at under half, and never
|
|
140
|
+
joined the two. That is TEXT, not EXPERIMENT.
|
|
141
|
+
- **A position on a ladder given without the rungs above it.** "Ours reaches rung three of four" reads
|
|
142
|
+
as a shortfall until the paper says what rungs one and two would require and that nobody has built
|
|
143
|
+
them. Also TEXT.
|
|
144
|
+
|
|
145
|
+
🔴 **A deadline is a constraint on how much you rework, never an argument for the current form.** If
|
|
146
|
+
the author invokes it against a TEXT prong, that is the failure this section exists to catch.
|
|
147
|
+
|
|
148
|
+
**Enforced** by `../paper-pipeline/scripts/pipeline-check.mjs` → `ceiling-unplanned`: a ceiling recorded
|
|
149
|
+
in the scorecard whose prongs carry none of the three labels is a finding.
|
|
150
|
+
|
|
151
|
+
## Record the verdict
|
|
152
|
+
|
|
153
|
+
🔴 LAST step, once the deliverable exists:
|
|
154
|
+
|
|
155
|
+
```
|
|
156
|
+
node .claude/skills/paper-pipeline/scripts/ledger.mjs record pc-panel-review <paper-dir> FINDING <count> <report-path>
|
|
157
|
+
node .claude/skills/paper-pipeline/scripts/ledger.mjs record pc-panel-review <paper-dir> ABSTAINED <reason> "<one line>"
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
**FINDING** — `<count>` is the number of **consensus** must-fixes (flagged by ≥2 independent
|
|
161
|
+
reviewers, not the union of everything anyone said), `<report-path>` is the meta-review. Add
|
|
162
|
+
`--blocking` when the chair's decision is reject.
|
|
163
|
+
**ABSTAINED** — `blocked`: the panel never ran because its required inputs (the persona stall
|
|
164
|
+
inventory, the structural verdict) do not exist. `no-witness`: the panel ran and reached no
|
|
165
|
+
consensus must-fix.
|
|
166
|
+
|
|
167
|
+
🔴 **THIS IS THE SKILL THE DELETION OF `PASS` WAS WRITTEN FOR.** Five model reviewers from one
|
|
168
|
+
vendor agreeing was being recorded as an acquittal — and correlated agreement between instances of
|
|
169
|
+
one model is not independent evidence of anything. There is now no constructor that can say it.
|
|
170
|
+
"Accept with no must-fix" is `ABSTAINED no-witness`: the panel produced no finding, which is a fact
|
|
171
|
+
about the panel and not a fact about the paper.
|
|
172
|
+
|
|
173
|
+
🔴 **The `blocked` row is the one that matters most here.** A blocked panel and a panel nobody
|
|
174
|
+
launched read identically in prose, and on 2026-08-04 a status line said the panel was running when
|
|
175
|
+
it had never been launched at all. Record the block; do not leave the check silent.
|
|
176
|
+
|
|
177
|
+
Venue-fit mode records under the same skill name with the fit verdict and `<count>` 0 — it is one
|
|
178
|
+
reviewer's spot-check, and its row should not be mistaken for the panel's decision.
|
|
179
|
+
|
|
180
|
+
## Rules
|
|
181
|
+
- **Independence is non-negotiable** — never let reviewers see each other's reviews before the
|
|
182
|
+
meta-review; shared context collapses the panel into one voice.
|
|
183
|
+
- **Consensus > volume.** One reviewer with ten nits loses to two reviewers naming the same blocker.
|
|
184
|
+
- **The artifact reviewer must actually run the code**, not read about it — that lens exists to catch
|
|
185
|
+
what prose review can't (dead links, circular self-checks, paper↔artifact number mismatches, leaks).
|
|
186
|
+
- Judge against the venue's ACTUAL bar; reward fit/discussion-value for workshops.
|
|
187
|
+
- Report the run honestly — if the artifact fails or a number doesn't reproduce, that IS the finding.
|
|
188
|
+
- Don't fabricate the CFP or citations; fetch/verify.
|
|
189
|
+
- **Review-ratchet rule — when the panel's must-fixes include an overclaim, inline-hedging is the LAST
|
|
190
|
+
resort.** Tighten → cut → move to Threats → only then hedge; and run `tighten-paper` after the round
|
|
191
|
+
to strip what it deposited. Full rule, with the cost of skipping it:
|
|
192
|
+
`paper-pipeline/references/review-ratchet.md`.
|
|
193
|
+
|
|
194
|
+
## Venue-fit mode (single-lens spot-check — the merged `venue-review-sim`)
|
|
195
|
+
When a full panel is overkill and you just want to predict **"given THIS venue's criteria, bar, and
|
|
196
|
+
culture, would a real reviewer accept it, and what scores?"**, run **one** venue-fit reviewer scoring
|
|
197
|
+
the specific CFP rubric — the fast fit-prediction lens (this absorbs the former standalone
|
|
198
|
+
`venue-review-sim` skill). It answers a different question than the defect hunt: fit-to-CFP and matching
|
|
199
|
+
the venue's expectation (a WIP workshop short paper vs a top-tier full paper) decide most workshop
|
|
200
|
+
outcomes, often more than raw defect count.
|
|
201
|
+
|
|
202
|
+
Run it:
|
|
203
|
+
1. **Fetch the venue's actual review criteria** — the CFP topics, paper types, stated review criteria,
|
|
204
|
+
workshop goals, blind model. Quote the exact criteria. If none stated, use the venue-class default
|
|
205
|
+
(workshops: relevance / originality / technical quality / clarity; security venues add threat-model
|
|
206
|
+
soundness + ethics). Do NOT fabricate the CFP — fetch it; say when you fall back to the default.
|
|
207
|
+
2. **Adopt the reviewer persona for THAT venue** — a PC member who knows the sub-field, cares about the
|
|
208
|
+
workshop's goals (e.g. "foster discussion", "bridge research–practice"), and calibrates to the paper
|
|
209
|
+
TYPE (a short/WIP/position paper is judged for promise and discussion value, NOT the completeness
|
|
210
|
+
demanded of a full paper — don't reject a WIP for "only 2 tools" if the venue invites WIP).
|
|
211
|
+
3. **Use a separate model as the reviewer** (e.g. Fable via subagent) so the author isn't grading
|
|
212
|
+
itself. Give it the full paper text + the quoted CFP criteria.
|
|
213
|
+
|
|
214
|
+
Output — fill the venue's review form:
|
|
215
|
+
- **Per-criterion score** on the venue's scale (map to 1–5 if unstated), one for EACH stated criterion,
|
|
216
|
+
each with a one-line justification grounded in the paper.
|
|
217
|
+
- **Overall recommendation** (Accept / Weak Accept / Borderline / Weak Reject / Reject) + **reviewer
|
|
218
|
+
confidence** (low/med/high).
|
|
219
|
+
- **Fit-to-CFP**: which listed topics it hits, and whether the framing foregrounds them (a
|
|
220
|
+
trustworthiness/verification workshop wants that word in the abstract).
|
|
221
|
+
- **Type-appropriateness**: is it pitched right for its track (WIP/vision/position/short vs full)? Call
|
|
222
|
+
out over-reach for a short paper or an under-sold real contribution.
|
|
223
|
+
- **What the PC discussion would say** — the 2–3 meta-review sentences that decide it, as a PC chair
|
|
224
|
+
would summarize.
|
|
225
|
+
- **Minimum changes to flip a borderline to accept** — the specific, venue-relevant edits (often:
|
|
226
|
+
foreground the on-topic framing, right-size claims to the track, add the one obviously-expected
|
|
227
|
+
citation) — NOT a full defect list.
|
|
228
|
+
- **Predicted outcome + probability**, calibrated to the venue's selectivity and edition (first-edition
|
|
229
|
+
workshops are more welcoming; established ones more competitive).
|
|
230
|
+
|
|
231
|
+
Rules: judge against the venue's ACTUAL bar and goals, not an abstract ideal; reward fit and
|
|
232
|
+
discussion-value for workshops, reserve full-paper rigor for full-paper tracks; be honest about
|
|
233
|
+
reject-risk without rubber-stamping. For the real decision, escalate to the full panel below.
|
|
234
|
+
|
|
235
|
+
## Compose with
|
|
236
|
+
- `paper-adversarial-review` (defect hunt) as an *input* — this skill is the layer above it (and above
|
|
237
|
+
its own Venue-fit mode). Run the full panel LAST, after a hardening pass, as the final pre-submission
|
|
238
|
+
gate.
|
|
239
|
+
- Append each panel's result to the paper's review ledger
|
|
240
|
+
(`papers/research/<date>-fable-rereview-<venue>.md`) so accept-probability progression is tracked
|
|
241
|
+
across rounds.
|
|
242
|
+
|
|
243
|
+
## The reproduce loop (why this beats a prose review)
|
|
244
|
+
This is "double-blind review as a real venue does it, WITH reproduction." The artifact-runner reviewer
|
|
245
|
+
is the difference-maker on both runs so far: it doesn't judge the paper's numbers, it re-derives them.
|
|
246
|
+
Run the panel, apply the consensus fixes, then **re-run the panel** (same lenses, told what changed) to
|
|
247
|
+
confirm the fixes landed and nothing regressed — the artifact-runner re-executes every harness each
|
|
248
|
+
round. Track the accept-probability progression in the review ledger. Stop when the panel converges to
|
|
249
|
+
accept and the artifact reproduces clean; don't loop past one confirming round.
|
|
250
|
+
|
|
251
|
+
## Provenance (two papers, battle-tested)
|
|
252
|
+
- **AgenticDev 2026** (2026-07-12): 3 lenses (stats / practitioner / artifact-runner) + chair. Caught
|
|
253
|
+
what two prior single-lens rounds missed — a paper↔artifact mismatch ("raw 140-run data" vs shipped
|
|
254
|
+
per-arm aggregates) and a mislabeled "paired" test, both flagged by ≥2 reviewers. Round→confirm:
|
|
255
|
+
~85%→90%.
|
|
256
|
+
- **AISec 2026 @ CCS** (2026-07-13): same shape at a harder (top-tier security) bar. The artifact-runner
|
|
257
|
+
ran all four harnesses AND stress-tested the release gate (planted a leak, confirmed it caught it);
|
|
258
|
+
the ethics reviewer caught the release-vs-paper anonymization contradiction (the released artifact
|
|
259
|
+
named 46 maintainers) — the single highest reject-vector, invisible to a paper-only read. After the
|
|
260
|
+
fixes + a confirm round, all three moved to Accept (~87–88%, from Weak-Accept/conditional/leak-risk).
|
|
261
|
+
Lesson: the two things that most move a security paper — an ethics/disclosure contradiction and a
|
|
262
|
+
reproduction/anonymization leak — live in the ARTIFACT, so the artifact-runner + an ethics lens are
|
|
263
|
+
non-optional at security venues.
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
// Compiled to SKILL.md by `vigiles compile`. Edit THIS file, never the markdown.
|
|
2
|
+
//
|
|
3
|
+
// Adopted 2026-08-17 (batch 3). Body carried over VERBATIM so the compiled diff shows
|
|
4
|
+
// only what the compiler adds. No `disallowedTools` fence yet — the field landed on
|
|
5
|
+
// `SkillSpec` in vigiles branch `claude/skill-disallowed-tools` and is not in a release
|
|
6
|
+
// this repo installs, so writing one here would not compile.
|
|
7
|
+
import { experimental_skill } from "vigiles/spec";
|
|
8
|
+
|
|
9
|
+
export default experimental_skill({
|
|
10
|
+
name: "pc-panel-review",
|
|
11
|
+
description:
|
|
12
|
+
'Use when asking "what would the program committee decide?" / "simulate the reviewers" / "what\'s this paper\'s accept probability?" on a drafted paper + artifact. Spawns N independent reviewers with DISTINCT lenses (one actually RUNS the artifact), then synthesizes a PC-chair meta-review into an accept/reject decision, a calibrated probability, and consensus must-fixes. Requires grade-paper-writing\'s persona stall inventory + tighten-paper\'s structural verdict as inputs (blocked until they exist — the panel cannot feel reader fatigue on its own). Also has a lightweight single-reviewer VENUE-FIT MODE for a quick CFP-fit spot-check. NOT a single fast defect hunt (paper-adversarial-review), a writing grade (grade-paper-writing), or the full pre-submit gate (harden-paper, which calls this) — this models the PC decision itself.',
|
|
13
|
+
tools: [
|
|
14
|
+
"Read",
|
|
15
|
+
"Write",
|
|
16
|
+
"Grep",
|
|
17
|
+
"Glob",
|
|
18
|
+
"Bash",
|
|
19
|
+
"WebSearch",
|
|
20
|
+
"WebFetch",
|
|
21
|
+
"Agent",
|
|
22
|
+
"Skill",
|
|
23
|
+
],
|
|
24
|
+
body: `
|
|
25
|
+
# pc-panel-review — model the whole PC, not one reviewer
|
|
26
|
+
|
|
27
|
+
> **Which review skill?** \`paper-adversarial-review\` = one hostile reviewer, fast defect hunt.
|
|
28
|
+
> \`pc-panel-review\` (you are here) = the whole PC (N independent lenses incl. an artifact-runner) + a
|
|
29
|
+
> chair meta-review — the real pre-submission GATE, run LAST. This skill also has a **Venue-fit mode**
|
|
30
|
+
> (below): ONE reviewer scoring a named venue's CFP rubric to predict its accept/reject — the quick
|
|
31
|
+
> single-lens spot-check (formerly the standalone \`venue-review-sim\` skill). Reach for Venue-fit mode or
|
|
32
|
+
> \`paper-adversarial-review\` for a fast spot-check; run the full panel (this) for the decision.
|
|
33
|
+
|
|
34
|
+
## Run me
|
|
35
|
+
|
|
36
|
+
🔴 FIRST, before any other step:
|
|
37
|
+
|
|
38
|
+
\`\`\`
|
|
39
|
+
node .claude/skills/paper-pipeline/scripts/announce.mjs pc-panel-review <paper-dir>
|
|
40
|
+
\`\`\`
|
|
41
|
+
|
|
42
|
+
An advisory pass cannot be observed failing — silence is both its error state and its normal
|
|
43
|
+
state — so starting is an event, and events get written down.
|
|
44
|
+
|
|
45
|
+
The Venue-fit mode gives ONE reviewer's fit score; \`paper-adversarial-review\` gives ONE defect hunt.
|
|
46
|
+
A real decision is made by **2–4 reviewers with different priorities + a meta-review** that weighs
|
|
47
|
+
consensus over any single voice. This skill runs that. The payoff over a single review: a **must-fix
|
|
48
|
+
is what ≥2 independent reviewers flag** — that filters real blockers from one reviewer's hobbyhorse —
|
|
49
|
+
and at least one reviewer **executes the artifact**, which surfaces paper↔artifact mismatches no
|
|
50
|
+
prose-only read can.
|
|
51
|
+
|
|
52
|
+
## How to run it
|
|
53
|
+
|
|
54
|
+
1. **Get the venue rubric + bar** (fetch the CFP; quote the criteria). Calibrate to the *track*: a
|
|
55
|
+
WIP/short/workshop bar rewards promise and discussion value; a full-paper/top-tier bar demands
|
|
56
|
+
completeness. Do not import top-tier standards into a WIP review.
|
|
57
|
+
2. **Collect the two readability/structure artifacts — the panel is BLOCKED without them.** Before any
|
|
58
|
+
reviewer spawns, there must exist for THIS draft: (a) the **persona stall inventory** from
|
|
59
|
+
\`grade-paper-writing\` (the "Sam" per-section cold read), and (b) the **structural verdict** from
|
|
60
|
+
\`tighten-paper\` (length / sag / TMI / cut-plan). If either doesn't exist yet, STOP and run those
|
|
61
|
+
skills first.
|
|
62
|
+
|
|
63
|
+
🔴 **Invoke them BY NAME with the Skill tool — do not describe the need and hope.** Whether
|
|
64
|
+
\`grade-paper-writing\` gets selected from its description is **unstable**, and that is worse for a
|
|
65
|
+
hard gate than a low rate would be: identical prompts, identical description, identical roster
|
|
66
|
+
measured **21%** in the morning of 2026-08-08 and **50–54%** the same afternoon, in two
|
|
67
|
+
independently written implementations
|
|
68
|
+
(\`../paper-pipeline/repro/2026-08-08-grade-paper-writing-ablation-raw.log\`,
|
|
69
|
+
\`2026-08-08-parent-replication.log\`). Seven rewrites of the description — shorter, negations
|
|
70
|
+
stripped, machinery stripped, persona-only — all landed inside noise, so there is nothing to fix
|
|
71
|
+
in the text. Naming the skill explicitly does not depend on selection at all, so this gate stops
|
|
72
|
+
resting on the one part of the chain that does. Writing "run grade-paper-writing" and leaving the
|
|
73
|
+
invocation to a description match is the failure this line exists to remove.
|
|
74
|
+
|
|
75
|
+
Why hard-block: a frontier-LLM reviewer knows every term and cannot feel reader
|
|
76
|
+
fatigue, so a panel scoring on its own felt read will score a bloated, unreadable paper high
|
|
77
|
+
(proven: ~88% accept on a paper the human reader found exhausting). These two artifacts are the
|
|
78
|
+
panel's only legitimate source for the readability/structure signal.
|
|
79
|
+
*(Scorecard: this gate is row **\`panel\`**, and it declares **\`structure\`** (tighten-paper) + **\`writing\`**
|
|
80
|
+
(grade-paper-writing) as its required inputs — \`pipeline-check.mjs\` reports \`panel\` marked done while
|
|
81
|
+
either input is missing or newer than it.)*
|
|
82
|
+
3. **Pick 3 (or 4) reviewer lenses** that a real PC for THIS venue would field. Menu — choose by fit:
|
|
83
|
+
- **Methods/stats skeptic** — sample size, construct validity, multiple comparisons, overclaims,
|
|
84
|
+
whether the CIs support the claims.
|
|
85
|
+
- **Domain practitioner** — relevance to the CFP, novelty vs the nearest prior work, does the
|
|
86
|
+
finding change behavior, generalizability, fairness/tone toward anyone named.
|
|
87
|
+
- **Artifact / reproducibility reviewer (MANDATORY if an artifact is attached)** — *actually
|
|
88
|
+
\`cd\` into the artifact and run it*; verify every headline number reproduces; read the code for
|
|
89
|
+
circularity (echoing stored fields vs recomputing); grep all files (incl. data + compiled
|
|
90
|
+
caches) for anonymization leaks; scope the badge (Available / Functional / Reproduced).
|
|
91
|
+
- **Security/ethics reviewer** (security venues) — threat model soundness, responsible disclosure,
|
|
92
|
+
punching-down risk.
|
|
93
|
+
- **Novelty/related-work reviewer** — is the delta over the nearest neighbor explicit; obvious
|
|
94
|
+
uncited work.
|
|
95
|
+
4. **Spawn them in PARALLEL, each on a separate model instance** (e.g. Fable via subagent), each blind
|
|
96
|
+
to the others — independence is the whole point. Give each: the paper (\`.tex\`/PDF), the artifact
|
|
97
|
+
path, the quoted rubric, its lens, **and the two step-2 artifacts (persona stall inventory +
|
|
98
|
+
tighten-paper structural verdict)**. Force a filled review form (below).
|
|
99
|
+
5. **Synthesize the meta-review yourself** (PC chair) once all land — see output.
|
|
100
|
+
|
|
101
|
+
## Each reviewer returns a review form
|
|
102
|
+
- Per-criterion score on the venue's scale (map to 1–5 if unstated), one line each, grounded.
|
|
103
|
+
- Overall rec (Reject … Strong Accept) + confidence (1–5).
|
|
104
|
+
- **Readability's implicit drag — model the halo effect, DRIVEN BY THE PERSONA INVENTORY, not felt
|
|
105
|
+
judgment.** A real reviewer who finds the paper a slog silently loses confidence in the *science* and
|
|
106
|
+
lowers the OVERALL score, while writing "methodology concerns" in the box — the penalty rarely shows
|
|
107
|
+
up as an explicit "clarity" mark. Each reviewer must let prose/structure quality move their overall
|
|
108
|
+
rec and confidence — but the SOURCE of that signal is the handed-in **persona stall inventory +
|
|
109
|
+
tighten-paper verdict**, never the reviewer's own felt read: a frontier LLM knows every term and
|
|
110
|
+
cannot experience reader fatigue, so its felt judgment systematically under-fires. If the inventory
|
|
111
|
+
shows high stall density / walls, or the structural verdict says bloated, the overall score takes a
|
|
112
|
+
real hit even when every explicit criterion passes. Never score a slog as if it read cleanly — that
|
|
113
|
+
is the gap between this sim and a real PC.
|
|
114
|
+
- Top 3 issues for PC discussion, ranked, with severity (blocker/major/minor).
|
|
115
|
+
- Any must-fix before *they'd* accept (or "none").
|
|
116
|
+
- Artifact reviewer only: run outcome (does it execute? every number reproduce? PASS/FAIL per check),
|
|
117
|
+
anonymization verdict, badge scope.
|
|
118
|
+
- One-line meta: does this belong at this venue?
|
|
119
|
+
|
|
120
|
+
## The meta-review (PC-chair synthesis) — the actual deliverable
|
|
121
|
+
- **Score table:** each reviewer × each criterion + overall + confidence, at a glance.
|
|
122
|
+
- **Decision:** the rec the PC discussion converges on (weight by confidence; a lone low-confidence
|
|
123
|
+
outlier doesn't sink two confident accepts), + a **calibrated accept probability** for this venue
|
|
124
|
+
and edition.
|
|
125
|
+
- **Consensus must-fixes:** issues **≥2 reviewers independently raised** — these are the real ones;
|
|
126
|
+
apply before submitting.
|
|
127
|
+
- **Single-reviewer flags:** noted, triaged (fix if cheap, else defer to camera-ready).
|
|
128
|
+
- **Split calls:** where reviewers genuinely disagree, and which way the chair leans + why.
|
|
129
|
+
- **Pre-submission checklist:** the surgical edits, ranked, that move the probability most.
|
|
130
|
+
|
|
131
|
+
## 🔴 The ceiling is not a sentence — it is a PLAN, and the panel writes it
|
|
132
|
+
|
|
133
|
+
**The corpus owner, 2026-08-05: "weak accept doesn't work for us. skills should suggest what to do to fix
|
|
134
|
+
the situation."** He is right that this was missing. Three panels in a row named the same ceiling, it was
|
|
135
|
+
faithfully recorded three times, and it never once became work. The fourth panel named it again — and
|
|
136
|
+
the author wrote all three of its prongs off as *"facts, not worth chasing before the deadline"*. Two
|
|
137
|
+
of the three were closed by an hour of editing that same afternoon.
|
|
138
|
+
|
|
139
|
+
So a panel that reports Weak Accept or Borderline **must** end with a ceiling plan: every prong
|
|
140
|
+
classified, no exceptions, using exactly these three labels.
|
|
141
|
+
|
|
142
|
+
| label | means | what the panel owes |
|
|
143
|
+
|---|---|---|
|
|
144
|
+
| **TEXT** | an edit closes it — the evidence already exists and the paper fails to join it up, frame it, or show the reader what the alternative would look like | the actual sentence or paragraph to write, and where it goes |
|
|
145
|
+
| **EXPERIMENT** | only new measurement closes it | what would have to be run, and a rough cost, so the decision to skip is informed |
|
|
146
|
+
| **IMMOVABLE** | it is what the work *is* — the scope, the population, the yield | say so plainly, so nobody re-opens it next round |
|
|
147
|
+
|
|
148
|
+
**Why the labels and not prose.** "This is a fact about the work" is the cheapest thing a reviewer can
|
|
149
|
+
write and the hardest to argue with, so it is where an unwilling author hides. Forcing a label makes
|
|
150
|
+
the claim falsifiable: *TEXT* invites "then write it", and *IMMOVABLE* invites "is it really?".
|
|
151
|
+
|
|
152
|
+
**Two failure shapes to check yourself against**, both observed on \`the reference paper\`:
|
|
153
|
+
|
|
154
|
+
- **A number reported as a rate when it is a floor.** "8 contradictions in 1,836 repositories" read as
|
|
155
|
+
a small phenomenon; the same paper measured its own extractor's recall at under half, and never
|
|
156
|
+
joined the two. That is TEXT, not EXPERIMENT.
|
|
157
|
+
- **A position on a ladder given without the rungs above it.** "Ours reaches rung three of four" reads
|
|
158
|
+
as a shortfall until the paper says what rungs one and two would require and that nobody has built
|
|
159
|
+
them. Also TEXT.
|
|
160
|
+
|
|
161
|
+
🔴 **A deadline is a constraint on how much you rework, never an argument for the current form.** If
|
|
162
|
+
the author invokes it against a TEXT prong, that is the failure this section exists to catch.
|
|
163
|
+
|
|
164
|
+
**Enforced** by \`../paper-pipeline/scripts/pipeline-check.mjs\` → \`ceiling-unplanned\`: a ceiling recorded
|
|
165
|
+
in the scorecard whose prongs carry none of the three labels is a finding.
|
|
166
|
+
|
|
167
|
+
## Record the verdict
|
|
168
|
+
|
|
169
|
+
🔴 LAST step, once the deliverable exists:
|
|
170
|
+
|
|
171
|
+
\`\`\`
|
|
172
|
+
node .claude/skills/paper-pipeline/scripts/ledger.mjs record pc-panel-review <paper-dir> FINDING <count> <report-path>
|
|
173
|
+
node .claude/skills/paper-pipeline/scripts/ledger.mjs record pc-panel-review <paper-dir> ABSTAINED <reason> "<one line>"
|
|
174
|
+
\`\`\`
|
|
175
|
+
|
|
176
|
+
**FINDING** — \`<count>\` is the number of **consensus** must-fixes (flagged by ≥2 independent
|
|
177
|
+
reviewers, not the union of everything anyone said), \`<report-path>\` is the meta-review. Add
|
|
178
|
+
\`--blocking\` when the chair's decision is reject.
|
|
179
|
+
**ABSTAINED** — \`blocked\`: the panel never ran because its required inputs (the persona stall
|
|
180
|
+
inventory, the structural verdict) do not exist. \`no-witness\`: the panel ran and reached no
|
|
181
|
+
consensus must-fix.
|
|
182
|
+
|
|
183
|
+
🔴 **THIS IS THE SKILL THE DELETION OF \`PASS\` WAS WRITTEN FOR.** Five model reviewers from one
|
|
184
|
+
vendor agreeing was being recorded as an acquittal — and correlated agreement between instances of
|
|
185
|
+
one model is not independent evidence of anything. There is now no constructor that can say it.
|
|
186
|
+
"Accept with no must-fix" is \`ABSTAINED no-witness\`: the panel produced no finding, which is a fact
|
|
187
|
+
about the panel and not a fact about the paper.
|
|
188
|
+
|
|
189
|
+
🔴 **The \`blocked\` row is the one that matters most here.** A blocked panel and a panel nobody
|
|
190
|
+
launched read identically in prose, and on 2026-08-04 a status line said the panel was running when
|
|
191
|
+
it had never been launched at all. Record the block; do not leave the check silent.
|
|
192
|
+
|
|
193
|
+
Venue-fit mode records under the same skill name with the fit verdict and \`<count>\` 0 — it is one
|
|
194
|
+
reviewer's spot-check, and its row should not be mistaken for the panel's decision.
|
|
195
|
+
|
|
196
|
+
## Rules
|
|
197
|
+
- **Independence is non-negotiable** — never let reviewers see each other's reviews before the
|
|
198
|
+
meta-review; shared context collapses the panel into one voice.
|
|
199
|
+
- **Consensus > volume.** One reviewer with ten nits loses to two reviewers naming the same blocker.
|
|
200
|
+
- **The artifact reviewer must actually run the code**, not read about it — that lens exists to catch
|
|
201
|
+
what prose review can't (dead links, circular self-checks, paper↔artifact number mismatches, leaks).
|
|
202
|
+
- Judge against the venue's ACTUAL bar; reward fit/discussion-value for workshops.
|
|
203
|
+
- Report the run honestly — if the artifact fails or a number doesn't reproduce, that IS the finding.
|
|
204
|
+
- Don't fabricate the CFP or citations; fetch/verify.
|
|
205
|
+
- **Review-ratchet rule — when the panel's must-fixes include an overclaim, inline-hedging is the LAST
|
|
206
|
+
resort.** Tighten → cut → move to Threats → only then hedge; and run \`tighten-paper\` after the round
|
|
207
|
+
to strip what it deposited. Full rule, with the cost of skipping it:
|
|
208
|
+
\`paper-pipeline/references/review-ratchet.md\`.
|
|
209
|
+
|
|
210
|
+
## Venue-fit mode (single-lens spot-check — the merged \`venue-review-sim\`)
|
|
211
|
+
When a full panel is overkill and you just want to predict **"given THIS venue's criteria, bar, and
|
|
212
|
+
culture, would a real reviewer accept it, and what scores?"**, run **one** venue-fit reviewer scoring
|
|
213
|
+
the specific CFP rubric — the fast fit-prediction lens (this absorbs the former standalone
|
|
214
|
+
\`venue-review-sim\` skill). It answers a different question than the defect hunt: fit-to-CFP and matching
|
|
215
|
+
the venue's expectation (a WIP workshop short paper vs a top-tier full paper) decide most workshop
|
|
216
|
+
outcomes, often more than raw defect count.
|
|
217
|
+
|
|
218
|
+
Run it:
|
|
219
|
+
1. **Fetch the venue's actual review criteria** — the CFP topics, paper types, stated review criteria,
|
|
220
|
+
workshop goals, blind model. Quote the exact criteria. If none stated, use the venue-class default
|
|
221
|
+
(workshops: relevance / originality / technical quality / clarity; security venues add threat-model
|
|
222
|
+
soundness + ethics). Do NOT fabricate the CFP — fetch it; say when you fall back to the default.
|
|
223
|
+
2. **Adopt the reviewer persona for THAT venue** — a PC member who knows the sub-field, cares about the
|
|
224
|
+
workshop's goals (e.g. "foster discussion", "bridge research–practice"), and calibrates to the paper
|
|
225
|
+
TYPE (a short/WIP/position paper is judged for promise and discussion value, NOT the completeness
|
|
226
|
+
demanded of a full paper — don't reject a WIP for "only 2 tools" if the venue invites WIP).
|
|
227
|
+
3. **Use a separate model as the reviewer** (e.g. Fable via subagent) so the author isn't grading
|
|
228
|
+
itself. Give it the full paper text + the quoted CFP criteria.
|
|
229
|
+
|
|
230
|
+
Output — fill the venue's review form:
|
|
231
|
+
- **Per-criterion score** on the venue's scale (map to 1–5 if unstated), one for EACH stated criterion,
|
|
232
|
+
each with a one-line justification grounded in the paper.
|
|
233
|
+
- **Overall recommendation** (Accept / Weak Accept / Borderline / Weak Reject / Reject) + **reviewer
|
|
234
|
+
confidence** (low/med/high).
|
|
235
|
+
- **Fit-to-CFP**: which listed topics it hits, and whether the framing foregrounds them (a
|
|
236
|
+
trustworthiness/verification workshop wants that word in the abstract).
|
|
237
|
+
- **Type-appropriateness**: is it pitched right for its track (WIP/vision/position/short vs full)? Call
|
|
238
|
+
out over-reach for a short paper or an under-sold real contribution.
|
|
239
|
+
- **What the PC discussion would say** — the 2–3 meta-review sentences that decide it, as a PC chair
|
|
240
|
+
would summarize.
|
|
241
|
+
- **Minimum changes to flip a borderline to accept** — the specific, venue-relevant edits (often:
|
|
242
|
+
foreground the on-topic framing, right-size claims to the track, add the one obviously-expected
|
|
243
|
+
citation) — NOT a full defect list.
|
|
244
|
+
- **Predicted outcome + probability**, calibrated to the venue's selectivity and edition (first-edition
|
|
245
|
+
workshops are more welcoming; established ones more competitive).
|
|
246
|
+
|
|
247
|
+
Rules: judge against the venue's ACTUAL bar and goals, not an abstract ideal; reward fit and
|
|
248
|
+
discussion-value for workshops, reserve full-paper rigor for full-paper tracks; be honest about
|
|
249
|
+
reject-risk without rubber-stamping. For the real decision, escalate to the full panel below.
|
|
250
|
+
|
|
251
|
+
## Compose with
|
|
252
|
+
- \`paper-adversarial-review\` (defect hunt) as an *input* — this skill is the layer above it (and above
|
|
253
|
+
its own Venue-fit mode). Run the full panel LAST, after a hardening pass, as the final pre-submission
|
|
254
|
+
gate.
|
|
255
|
+
- Append each panel's result to the paper's review ledger
|
|
256
|
+
(\`papers/research/<date>-fable-rereview-<venue>.md\`) so accept-probability progression is tracked
|
|
257
|
+
across rounds.
|
|
258
|
+
|
|
259
|
+
## The reproduce loop (why this beats a prose review)
|
|
260
|
+
This is "double-blind review as a real venue does it, WITH reproduction." The artifact-runner reviewer
|
|
261
|
+
is the difference-maker on both runs so far: it doesn't judge the paper's numbers, it re-derives them.
|
|
262
|
+
Run the panel, apply the consensus fixes, then **re-run the panel** (same lenses, told what changed) to
|
|
263
|
+
confirm the fixes landed and nothing regressed — the artifact-runner re-executes every harness each
|
|
264
|
+
round. Track the accept-probability progression in the review ledger. Stop when the panel converges to
|
|
265
|
+
accept and the artifact reproduces clean; don't loop past one confirming round.
|
|
266
|
+
|
|
267
|
+
## Provenance (two papers, battle-tested)
|
|
268
|
+
- **AgenticDev 2026** (2026-07-12): 3 lenses (stats / practitioner / artifact-runner) + chair. Caught
|
|
269
|
+
what two prior single-lens rounds missed — a paper↔artifact mismatch ("raw 140-run data" vs shipped
|
|
270
|
+
per-arm aggregates) and a mislabeled "paired" test, both flagged by ≥2 reviewers. Round→confirm:
|
|
271
|
+
~85%→90%.
|
|
272
|
+
- **AISec 2026 @ CCS** (2026-07-13): same shape at a harder (top-tier security) bar. The artifact-runner
|
|
273
|
+
ran all four harnesses AND stress-tested the release gate (planted a leak, confirmed it caught it);
|
|
274
|
+
the ethics reviewer caught the release-vs-paper anonymization contradiction (the released artifact
|
|
275
|
+
named 46 maintainers) — the single highest reject-vector, invisible to a paper-only read. After the
|
|
276
|
+
fixes + a confirm round, all three moved to Accept (~87–88%, from Weak-Accept/conditional/leak-risk).
|
|
277
|
+
Lesson: the two things that most move a security paper — an ethics/disclosure contradiction and a
|
|
278
|
+
reproduction/anonymization leak — live in the ARTIFACT, so the artifact-runner + an ethics lens are
|
|
279
|
+
non-optional at security venues.`,
|
|
280
|
+
});
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* pc-panel-review — the PAID tier: does this skill's description actually fire?
|
|
3
|
+
*
|
|
4
|
+
* COLOCATED ON PURPOSE (vigiles decides coverage by placement as of 2026-08-11).
|
|
5
|
+
* The prompts live in `.claude/lib/skill-trigger-cases.mjs` so all 21 cases
|
|
6
|
+
* are reviewed as one table where collisions between siblings are visible;
|
|
7
|
+
* copying them here would recreate the drift that rule exists to prevent.
|
|
8
|
+
*
|
|
9
|
+
* Measures recall (fires on its own territory) AND precision (stays quiet on a
|
|
10
|
+
* colliding sibling's territory), against the REAL `.claude` harness so the skill
|
|
11
|
+
* competes with every other installed description — an isolated run overstates
|
|
12
|
+
* recall and understates false positives.
|
|
13
|
+
*
|
|
14
|
+
* Costs money; not CI.
|
|
15
|
+
* node .claude/skills/pc-panel-review/pc-panel-review.eval.mjs [trials]
|
|
16
|
+
*/
|
|
17
|
+
import { runSkillTriggerEval } from "../../lib/skill-eval-kit.mjs";
|
|
18
|
+
|
|
19
|
+
await runSkillTriggerEval("pc-panel-review");
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* pc-panel-review — the free, deterministic tier. No model, no network.
|
|
3
|
+
*
|
|
4
|
+
* COLOCATED ON PURPOSE. vigiles decides coverage by PLACEMENT as of 2026-08-11:
|
|
5
|
+
* a test that merely names a surface no longer counts, because that tier was
|
|
6
|
+
* crediting surfaces nothing touched. So each skill needs a file inside its own
|
|
7
|
+
* directory — this one.
|
|
8
|
+
*
|
|
9
|
+
* The assertions live in `.claude/lib/skill-checks.mjs` and are CALLED here with
|
|
10
|
+
* this skill's name. They are not copied: 22 copies of the same checks is the drift that
|
|
11
|
+
* module exists to avoid. (Until 2026-08-11 this was an env-var side channel into a
|
|
12
|
+
* 614-line file named after no surface; it is a function call now.)
|
|
13
|
+
*
|
|
14
|
+
* What this proves: this skill's frontmatter parses as strict YAML, its declared
|
|
15
|
+
* tool contract is sane, its pipeline wiring points at scripts that exist, and it
|
|
16
|
+
* announces/records under ITS OWN identity rather than a sibling's.
|
|
17
|
+
*
|
|
18
|
+
* What it does NOT prove: that the skill fires, or that its guidance produces a
|
|
19
|
+
* good result. Those need a real model — see `pc-panel-review.eval.mjs`.
|
|
20
|
+
*/
|
|
21
|
+
import { checkSkill } from "../../lib/skill-checks.mjs";
|
|
22
|
+
|
|
23
|
+
await checkSkill("pc-panel-review");
|