dsh-aris-panel 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +98 -0
- package/README_CN.md +87 -0
- package/dsh/checkout.patch.yml +38 -0
- package/dsh/client.js +634 -0
- package/dsh/cordis.patch.yml +44 -0
- package/dsh/index.mjs +76 -0
- package/dsh/run-status.mjs +182 -0
- package/dsh/scope-limits.mjs +50 -0
- package/dsh/workbench.mjs +291 -0
- package/mcp-servers/claude-review/README.md +93 -0
- package/mcp-servers/claude-review/run_with_claude_aws.sh +49 -0
- package/mcp-servers/claude-review/server.py +718 -0
- package/mcp-servers/codex-image2/README.md +65 -0
- package/mcp-servers/codex-image2/server.py +893 -0
- package/mcp-servers/feishu-bridge/requirements.txt +1 -0
- package/mcp-servers/feishu-bridge/server.py +240 -0
- package/mcp-servers/gemini-review/README.md +171 -0
- package/mcp-servers/gemini-review/server.py +1856 -0
- package/mcp-servers/llm-chat/requirements.txt +1 -0
- package/mcp-servers/llm-chat/server.py +664 -0
- package/mcp-servers/manual-review/README.md +133 -0
- package/mcp-servers/manual-review/server.py +910 -0
- package/mcp-servers/manual-review/ui.html +279 -0
- package/mcp-servers/minimax-chat/requirements.txt +1 -0
- package/mcp-servers/minimax-chat/server.py +381 -0
- package/package.json +51 -0
- package/skills/ablation-planner/SKILL.md +123 -0
- package/skills/alphaxiv/SKILL.md +196 -0
- package/skills/analyze-results/SKILL.md +46 -0
- package/skills/arxiv/SKILL.md +248 -0
- package/skills/auto-paper-improvement-loop/SKILL.md +651 -0
- package/skills/auto-review-loop/SKILL.md +1137 -0
- package/skills/auto-review-loop-llm/SKILL.md +259 -0
- package/skills/auto-review-loop-minimax/SKILL.md +302 -0
- package/skills/citation-audit/SKILL.md +502 -0
- package/skills/claims-drafting/SKILL.md +227 -0
- package/skills/comm-lit-review/SKILL.md +297 -0
- package/skills/deepxiv/SKILL.md +263 -0
- package/skills/dse-loop/SKILL.md +296 -0
- package/skills/embodiment-description/SKILL.md +129 -0
- package/skills/exa-search/SKILL.md +205 -0
- package/skills/experiment-audit/SKILL.md +311 -0
- package/skills/experiment-bridge/SKILL.md +376 -0
- package/skills/experiment-plan/SKILL.md +249 -0
- package/skills/experiment-queue/SKILL.md +431 -0
- package/skills/experiment-queue/scripts/build_manifest.py +142 -0
- package/skills/experiment-queue/scripts/queue_manager.py +433 -0
- package/skills/feishu-notify/SKILL.md +156 -0
- package/skills/figure-description/SKILL.md +138 -0
- package/skills/figure-spec/SKILL.md +262 -0
- package/skills/figure-spec/scripts/figure_renderer.py +799 -0
- package/skills/formula-derivation/SKILL.md +280 -0
- package/skills/gemini-search/SKILL.md +231 -0
- package/skills/grant-proposal/SKILL.md +698 -0
- package/skills/idea-creator/SKILL.md +542 -0
- package/skills/idea-discovery/SKILL.md +521 -0
- package/skills/idea-discovery-robot/SKILL.md +363 -0
- package/skills/integrity-forensics/SKILL.md +284 -0
- package/skills/interview-cheatsheet/SKILL.md +245 -0
- package/skills/invention-structuring/SKILL.md +188 -0
- package/skills/jurisdiction-format/SKILL.md +192 -0
- package/skills/kill-argument/SKILL.md +437 -0
- package/skills/mermaid-diagram/SKILL.md +419 -0
- package/skills/meta-apply/SKILL.md +141 -0
- package/skills/meta-optimize/SKILL.md +437 -0
- package/skills/monitor-experiment/SKILL.md +140 -0
- package/skills/novelty-check/SKILL.md +101 -0
- package/skills/openalex/SKILL.md +237 -0
- package/skills/overleaf-sync/SKILL.md +220 -0
- package/skills/paper-claim-audit/SKILL.md +348 -0
- package/skills/paper-compile/SKILL.md +266 -0
- package/skills/paper-figure/SKILL.md +312 -0
- package/skills/paper-illustration/SKILL.md +736 -0
- package/skills/paper-illustration-image2/SKILL.md +391 -0
- package/skills/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
- package/skills/paper-plan/SKILL.md +386 -0
- package/skills/paper-poster/SKILL.md +19 -0
- package/skills/paper-poster-html/DESIGN_FINAL.md +176 -0
- package/skills/paper-poster-html/IMPLEMENTATION_CONVENTIONS.md +161 -0
- package/skills/paper-poster-html/LICENSES/posterly-MIT.txt +21 -0
- package/skills/paper-poster-html/NOTICE.md +57 -0
- package/skills/paper-poster-html/SKILL.md +323 -0
- package/skills/paper-poster-html/scripts/_posterly/__init__.py +0 -0
- package/skills/paper-poster-html/scripts/_posterly/canvas.py +200 -0
- package/skills/paper-poster-html/scripts/_posterly/measure.py +588 -0
- package/skills/paper-poster-html/scripts/_posterly/polish.py +498 -0
- package/skills/paper-poster-html/scripts/_posterly/preflight.py +489 -0
- package/skills/paper-poster-html/scripts/_posterly/render.py +215 -0
- package/skills/paper-poster-html/scripts/_posterly/textutil.py +16 -0
- package/skills/paper-poster-html/scripts/_posterly/verify_final.py +171 -0
- package/skills/paper-poster-html/scripts/asset_check.py +897 -0
- package/skills/paper-poster-html/scripts/extract_pdf_figures.py +666 -0
- package/skills/paper-poster-html/scripts/poster_check.py +251 -0
- package/skills/paper-poster-html/scripts/preprocess_figures.py +238 -0
- package/skills/paper-poster-html/scripts/render_preview.py +217 -0
- package/skills/paper-poster-html/scripts/run_gates.py +556 -0
- package/skills/paper-poster-html/scripts/style_check.py +1324 -0
- package/skills/paper-poster-html/templates/COMPONENTS.md +462 -0
- package/skills/paper-poster-html/templates/README.md +170 -0
- package/skills/paper-poster-html/templates/landscape_4col.html +1032 -0
- package/skills/paper-poster-html/templates/landscape_hero.html +1046 -0
- package/skills/paper-poster-html/templates/portrait_2col.html +947 -0
- package/skills/paper-poster-html/templates/tokens/acl.json +9 -0
- package/skills/paper-poster-html/templates/tokens/cvpr.json +9 -0
- package/skills/paper-poster-html/templates/tokens/generic.json +9 -0
- package/skills/paper-poster-html/templates/tokens/iclr.json +9 -0
- package/skills/paper-poster-html/templates/tokens/icml.json +9 -0
- package/skills/paper-poster-html/templates/tokens/neurips.json +9 -0
- package/skills/paper-slides/SKILL.md +635 -0
- package/skills/paper-talk/SKILL.md +381 -0
- package/skills/paper-write/SKILL.md +604 -0
- package/skills/paper-write/templates/IEEEtran.bst +2409 -0
- package/skills/paper-write/templates/IEEEtran.cls +6347 -0
- package/skills/paper-write/templates/iclr2026.tex +84 -0
- package/skills/paper-write/templates/icml2025.tex +87 -0
- package/skills/paper-write/templates/ieee_conference.tex +89 -0
- package/skills/paper-write/templates/ieee_journal.tex +93 -0
- package/skills/paper-write/templates/math_commands.tex +48 -0
- package/skills/paper-write/templates/neurips2025.tex +80 -0
- package/skills/paper-writing/SKILL.md +916 -0
- package/skills/patent-novelty-check/SKILL.md +153 -0
- package/skills/patent-pipeline/SKILL.md +344 -0
- package/skills/patent-review/SKILL.md +203 -0
- package/skills/pixel-art/SKILL.md +137 -0
- package/skills/prior-art-search/SKILL.md +146 -0
- package/skills/proof-checker/SKILL.md +866 -0
- package/skills/proof-orchestrator/NOTICE.md +24 -0
- package/skills/proof-orchestrator/SKILL.md +254 -0
- package/skills/proof-orchestrator/references/audit-output-contract.md +126 -0
- package/skills/proof-orchestrator/references/deepseek-routing.md +74 -0
- package/skills/proof-orchestrator/references/dispatch-prompts.md +227 -0
- package/skills/proof-orchestrator/references/notation-audit.md +135 -0
- package/skills/proof-orchestrator/references/proof-audit-rubric.md +70 -0
- package/skills/proof-orchestrator/references/stress-tests.md +38 -0
- package/skills/proof-writer/SKILL.md +223 -0
- package/skills/qzcli/SKILL.md +324 -0
- package/skills/rebuttal/SKILL.md +376 -0
- package/skills/render-html/SKILL.md +316 -0
- package/skills/render-html/scripts/render_html.py +1006 -0
- package/skills/render-html/scripts/templates/academic.html +703 -0
- package/skills/render-html/scripts/templates/dashboard.html +333 -0
- package/skills/research-lit/SKILL.md +756 -0
- package/skills/research-pipeline/SKILL.md +384 -0
- package/skills/research-refine/SKILL.md +770 -0
- package/skills/research-refine-pipeline/SKILL.md +186 -0
- package/skills/research-review/SKILL.md +198 -0
- package/skills/research-wiki/SKILL.md +461 -0
- package/skills/resubmit-pipeline/SKILL.md +447 -0
- package/skills/result-to-claim/SKILL.md +311 -0
- package/skills/run-experiment/SKILL.md +313 -0
- package/skills/semantic-scholar/SKILL.md +236 -0
- package/skills/serverless-modal/SKILL.md +335 -0
- package/skills/shared-references/acceptance-gate.md +324 -0
- package/skills/shared-references/assurance-contract.md +248 -0
- package/skills/shared-references/capture-antipatterns.md +78 -0
- package/skills/shared-references/citation-discipline.md +583 -0
- package/skills/shared-references/compute-env-contract.md +163 -0
- package/skills/shared-references/effort-contract.md +183 -0
- package/skills/shared-references/evidence-precheck.md +65 -0
- package/skills/shared-references/experiment-integrity.md +49 -0
- package/skills/shared-references/external-cadence.md +326 -0
- package/skills/shared-references/fan-out-pattern.md +366 -0
- package/skills/shared-references/injection-hygiene.md +127 -0
- package/skills/shared-references/integration-contract.md +461 -0
- package/skills/shared-references/output-composition.md +93 -0
- package/skills/shared-references/output-language.md +45 -0
- package/skills/shared-references/output-manifest.md +49 -0
- package/skills/shared-references/output-versioning.md +111 -0
- package/skills/shared-references/patent-format-cn.md +199 -0
- package/skills/shared-references/patent-format-ep.md +173 -0
- package/skills/shared-references/patent-format-us.md +161 -0
- package/skills/shared-references/patent-writing-principles.md +197 -0
- package/skills/shared-references/prior-art-databases.md +141 -0
- package/skills/shared-references/resumable-runs.md +109 -0
- package/skills/shared-references/review-scope-limits.md +81 -0
- package/skills/shared-references/review-tracing.md +391 -0
- package/skills/shared-references/reviewer-independence.md +79 -0
- package/skills/shared-references/reviewer-routing.md +852 -0
- package/skills/shared-references/skill-governance.md +104 -0
- package/skills/shared-references/taste-calibration.md +85 -0
- package/skills/shared-references/venue-checklists.md +114 -0
- package/skills/shared-references/wiki-helper-resolution.md +134 -0
- package/skills/shared-references/writing-principles.md +525 -0
- package/skills/skills-codex/README.md +102 -0
- package/skills/skills-codex/README_CN.md +100 -0
- package/skills/skills-codex/ablation-planner/SKILL.md +126 -0
- package/skills/skills-codex/alphaxiv/SKILL.md +186 -0
- package/skills/skills-codex/analyze-results/SKILL.md +45 -0
- package/skills/skills-codex/arxiv/SKILL.md +210 -0
- package/skills/skills-codex/auto-paper-improvement-loop/SKILL.md +574 -0
- package/skills/skills-codex/auto-review-loop/SKILL.md +500 -0
- package/skills/skills-codex/auto-review-loop-llm/SKILL.md +247 -0
- package/skills/skills-codex/auto-review-loop-minimax/SKILL.md +290 -0
- package/skills/skills-codex/citation-audit/SKILL.md +504 -0
- package/skills/skills-codex/claims-drafting/SKILL.md +239 -0
- package/skills/skills-codex/comm-lit-review/SKILL.md +299 -0
- package/skills/skills-codex/comm-lit-review/references/domain-taxonomy.md +57 -0
- package/skills/skills-codex/comm-lit-review/references/output-template.md +37 -0
- package/skills/skills-codex/comm-lit-review/references/source-policy.md +99 -0
- package/skills/skills-codex/comm-lit-review/references/venue-tiering.md +112 -0
- package/skills/skills-codex/deepxiv/SKILL.md +142 -0
- package/skills/skills-codex/dse-loop/SKILL.md +285 -0
- package/skills/skills-codex/embodiment-description/SKILL.md +129 -0
- package/skills/skills-codex/exa-search/SKILL.md +192 -0
- package/skills/skills-codex/experiment-audit/SKILL.md +286 -0
- package/skills/skills-codex/experiment-bridge/SKILL.md +356 -0
- package/skills/skills-codex/experiment-plan/SKILL.md +249 -0
- package/skills/skills-codex/experiment-queue/SKILL.md +401 -0
- package/skills/skills-codex/feishu-notify/SKILL.md +155 -0
- package/skills/skills-codex/figure-description/SKILL.md +138 -0
- package/skills/skills-codex/figure-spec/SKILL.md +252 -0
- package/skills/skills-codex/formula-derivation/SKILL.md +280 -0
- package/skills/skills-codex/gemini-search/SKILL.md +205 -0
- package/skills/skills-codex/grant-proposal/SKILL.md +626 -0
- package/skills/skills-codex/idea-creator/SKILL.md +405 -0
- package/skills/skills-codex/idea-discovery/SKILL.md +475 -0
- package/skills/skills-codex/idea-discovery-robot/SKILL.md +362 -0
- package/skills/skills-codex/integrity-forensics/SKILL.md +106 -0
- package/skills/skills-codex/interview-cheatsheet/SKILL.md +245 -0
- package/skills/skills-codex/invention-structuring/SKILL.md +188 -0
- package/skills/skills-codex/jurisdiction-format/SKILL.md +192 -0
- package/skills/skills-codex/kill-argument/SKILL.md +403 -0
- package/skills/skills-codex/mermaid-diagram/SKILL.md +379 -0
- package/skills/skills-codex/meta-apply/SKILL.md +154 -0
- package/skills/skills-codex/meta-optimize/SKILL.md +348 -0
- package/skills/skills-codex/monitor-experiment/SKILL.md +98 -0
- package/skills/skills-codex/novelty-check/SKILL.md +89 -0
- package/skills/skills-codex/openalex/SKILL.md +228 -0
- package/skills/skills-codex/overleaf-sync/SKILL.md +220 -0
- package/skills/skills-codex/paper-claim-audit/SKILL.md +350 -0
- package/skills/skills-codex/paper-compile/SKILL.md +253 -0
- package/skills/skills-codex/paper-figure/SKILL.md +311 -0
- package/skills/skills-codex/paper-illustration/SKILL.md +690 -0
- package/skills/skills-codex/paper-illustration-image2/SKILL.md +383 -0
- package/skills/skills-codex/paper-illustration-image2/scripts/paper_illustration_image2.py +255 -0
- package/skills/skills-codex/paper-plan/SKILL.md +278 -0
- package/skills/skills-codex/paper-poster/SKILL.md +19 -0
- package/skills/skills-codex/paper-poster-html/SKILL.md +377 -0
- package/skills/skills-codex/paper-slides/SKILL.md +571 -0
- package/skills/skills-codex/paper-talk/SKILL.md +381 -0
- package/skills/skills-codex/paper-write/SKILL.md +411 -0
- package/skills/skills-codex/paper-write/templates/IEEEtran.bst +2409 -0
- package/skills/skills-codex/paper-write/templates/IEEEtran.cls +6347 -0
- package/skills/skills-codex/paper-write/templates/aaai2026.bst +1493 -0
- package/skills/skills-codex/paper-write/templates/aaai2026.sty +315 -0
- package/skills/skills-codex/paper-write/templates/aaai2026.tex +952 -0
- package/skills/skills-codex/paper-write/templates/acl.sty +312 -0
- package/skills/skills-codex/paper-write/templates/acl2026.tex +377 -0
- package/skills/skills-codex/paper-write/templates/acl_natbib.bst +1940 -0
- package/skills/skills-codex/paper-write/templates/acm.bst +3081 -0
- package/skills/skills-codex/paper-write/templates/acm_mm2026.tex +204 -0
- package/skills/skills-codex/paper-write/templates/acmart.cls +3520 -0
- package/skills/skills-codex/paper-write/templates/cvpr.bst +1448 -0
- package/skills/skills-codex/paper-write/templates/cvpr.sty +508 -0
- package/skills/skills-codex/paper-write/templates/cvpr2026.tex +63 -0
- package/skills/skills-codex/paper-write/templates/iclr2026.tex +84 -0
- package/skills/skills-codex/paper-write/templates/iclr2026_conference.bst +1440 -0
- package/skills/skills-codex/paper-write/templates/iclr2026_conference.sty +246 -0
- package/skills/skills-codex/paper-write/templates/icml2026.sty +767 -0
- package/skills/skills-codex/paper-write/templates/icml2026.tex +662 -0
- package/skills/skills-codex/paper-write/templates/ieee_conference.tex +89 -0
- package/skills/skills-codex/paper-write/templates/ieee_journal.tex +93 -0
- package/skills/skills-codex/paper-write/templates/math_commands.tex +48 -0
- package/skills/skills-codex/paper-write/templates/neurips2026.tex +493 -0
- package/skills/skills-codex/paper-write/templates/neurips_2026.sty +437 -0
- package/skills/skills-codex/paper-writing/SKILL.md +731 -0
- package/skills/skills-codex/patent-novelty-check/SKILL.md +153 -0
- package/skills/skills-codex/patent-pipeline/SKILL.md +344 -0
- package/skills/skills-codex/patent-review/SKILL.md +202 -0
- package/skills/skills-codex/pixel-art/SKILL.md +139 -0
- package/skills/skills-codex/prior-art-search/SKILL.md +146 -0
- package/skills/skills-codex/proof-checker/SKILL.md +554 -0
- package/skills/skills-codex/proof-orchestrator/SKILL.md +260 -0
- package/skills/skills-codex/proof-orchestrator/references/audit-output-contract.md +126 -0
- package/skills/skills-codex/proof-orchestrator/references/deepseek-routing.md +76 -0
- package/skills/skills-codex/proof-orchestrator/references/dispatch-prompts.md +227 -0
- package/skills/skills-codex/proof-orchestrator/references/notation-audit.md +135 -0
- package/skills/skills-codex/proof-orchestrator/references/proof-audit-rubric.md +70 -0
- package/skills/skills-codex/proof-orchestrator/references/stress-tests.md +38 -0
- package/skills/skills-codex/proof-writer/SKILL.md +222 -0
- package/skills/skills-codex/qzcli/SKILL.md +324 -0
- package/skills/skills-codex/rebuttal/SKILL.md +305 -0
- package/skills/skills-codex/render-html/SKILL.md +305 -0
- package/skills/skills-codex/render-html/scripts/__pycache__/render_html.cpython-314.pyc +0 -0
- package/skills/skills-codex/render-html/scripts/render_html.py +909 -0
- package/skills/skills-codex/render-html/scripts/templates/academic.html +342 -0
- package/skills/skills-codex/render-html/scripts/templates/dashboard.html +333 -0
- package/skills/skills-codex/research-lit/SKILL.md +464 -0
- package/skills/skills-codex/research-pipeline/SKILL.md +340 -0
- package/skills/skills-codex/research-refine/SKILL.md +721 -0
- package/skills/skills-codex/research-refine-pipeline/SKILL.md +186 -0
- package/skills/skills-codex/research-review/SKILL.md +135 -0
- package/skills/skills-codex/research-wiki/SKILL.md +421 -0
- package/skills/skills-codex/resubmit-pipeline/SKILL.md +444 -0
- package/skills/skills-codex/result-to-claim/SKILL.md +246 -0
- package/skills/skills-codex/run-experiment/SKILL.md +236 -0
- package/skills/skills-codex/semantic-scholar/SKILL.md +219 -0
- package/skills/skills-codex/serverless-modal/SKILL.md +335 -0
- package/skills/skills-codex/shared-references/acceptance-gate.md +336 -0
- package/skills/skills-codex/shared-references/assurance-contract.md +139 -0
- package/skills/skills-codex/shared-references/capture-antipatterns.md +84 -0
- package/skills/skills-codex/shared-references/citation-discipline.md +452 -0
- package/skills/skills-codex/shared-references/compute-env-contract.md +163 -0
- package/skills/skills-codex/shared-references/effort-contract.md +143 -0
- package/skills/skills-codex/shared-references/evidence-precheck.md +73 -0
- package/skills/skills-codex/shared-references/experiment-integrity.md +49 -0
- package/skills/skills-codex/shared-references/external-cadence.md +334 -0
- package/skills/skills-codex/shared-references/fan-out-pattern.md +375 -0
- package/skills/skills-codex/shared-references/injection-hygiene.md +132 -0
- package/skills/skills-codex/shared-references/integration-contract.md +372 -0
- package/skills/skills-codex/shared-references/output-composition.md +98 -0
- package/skills/skills-codex/shared-references/output-language.md +45 -0
- package/skills/skills-codex/shared-references/output-manifest.md +40 -0
- package/skills/skills-codex/shared-references/output-versioning.md +111 -0
- package/skills/skills-codex/shared-references/patent-format-cn.md +199 -0
- package/skills/skills-codex/shared-references/patent-format-ep.md +173 -0
- package/skills/skills-codex/shared-references/patent-format-us.md +161 -0
- package/skills/skills-codex/shared-references/patent-writing-principles.md +197 -0
- package/skills/skills-codex/shared-references/prior-art-databases.md +141 -0
- package/skills/skills-codex/shared-references/resumable-runs.md +125 -0
- package/skills/skills-codex/shared-references/review-scope-limits.md +81 -0
- package/skills/skills-codex/shared-references/review-tracing.md +144 -0
- package/skills/skills-codex/shared-references/reviewer-independence.md +66 -0
- package/skills/skills-codex/shared-references/reviewer-routing.md +128 -0
- package/skills/skills-codex/shared-references/skill-governance.md +119 -0
- package/skills/skills-codex/shared-references/taste-calibration.md +90 -0
- package/skills/skills-codex/shared-references/venue-checklists.md +73 -0
- package/skills/skills-codex/shared-references/wiki-helper-resolution.md +69 -0
- package/skills/skills-codex/shared-references/writing-principles.md +525 -0
- package/skills/skills-codex/slides-polish/SKILL.md +563 -0
- package/skills/skills-codex/specification-writing/SKILL.md +211 -0
- package/skills/skills-codex/system-profile/SKILL.md +103 -0
- package/skills/skills-codex/training-check/SKILL.md +83 -0
- package/skills/skills-codex/vast-gpu/SKILL.md +394 -0
- package/skills/skills-codex/web-debug-search/SKILL.md +334 -0
- package/skills/skills-codex/wiki-enrich/SKILL.md +255 -0
- package/skills/skills-codex/writing-systems-papers/SKILL.md +184 -0
- package/skills/skills-codex-claude-review/README.md +79 -0
- package/skills/skills-codex-claude-review/README_CN.md +78 -0
- package/skills/skills-codex-claude-review/auto-paper-improvement-loop/SKILL.md +581 -0
- package/skills/skills-codex-claude-review/auto-review-loop/SKILL.md +510 -0
- package/skills/skills-codex-claude-review/novelty-check/SKILL.md +102 -0
- package/skills/skills-codex-claude-review/paper-figure/SKILL.md +319 -0
- package/skills/skills-codex-claude-review/paper-plan/SKILL.md +287 -0
- package/skills/skills-codex-claude-review/paper-write/SKILL.md +420 -0
- package/skills/skills-codex-claude-review/research-refine/SKILL.md +732 -0
- package/skills/skills-codex-claude-review/research-review/SKILL.md +149 -0
- package/skills/skills-codex-gemini-review/README.md +176 -0
- package/skills/skills-codex-gemini-review/README_CN.md +175 -0
- package/skills/skills-codex-gemini-review/auto-paper-improvement-loop/SKILL.md +331 -0
- package/skills/skills-codex-gemini-review/auto-review-loop/SKILL.md +304 -0
- package/skills/skills-codex-gemini-review/grant-proposal/SKILL.md +630 -0
- package/skills/skills-codex-gemini-review/idea-creator/SKILL.md +263 -0
- package/skills/skills-codex-gemini-review/idea-discovery/SKILL.md +275 -0
- package/skills/skills-codex-gemini-review/idea-discovery-robot/SKILL.md +365 -0
- package/skills/skills-codex-gemini-review/novelty-check/SKILL.md +92 -0
- package/skills/skills-codex-gemini-review/paper-figure/SKILL.md +289 -0
- package/skills/skills-codex-gemini-review/paper-plan/SKILL.md +265 -0
- package/skills/skills-codex-gemini-review/paper-poster-html/SKILL.md +102 -0
- package/skills/skills-codex-gemini-review/paper-slides/SKILL.md +582 -0
- package/skills/skills-codex-gemini-review/paper-write/SKILL.md +346 -0
- package/skills/skills-codex-gemini-review/paper-writing/SKILL.md +312 -0
- package/skills/skills-codex-gemini-review/research-refine/SKILL.md +674 -0
- package/skills/skills-codex-gemini-review/research-review/SKILL.md +112 -0
- package/skills/slides-polish/SKILL.md +565 -0
- package/skills/specification-writing/SKILL.md +211 -0
- package/skills/system-profile/SKILL.md +103 -0
- package/skills/training-check/SKILL.md +132 -0
- package/skills/vast-gpu/SKILL.md +394 -0
- package/skills/web-debug-search/SKILL.md +334 -0
- package/skills/wiki-enrich/SKILL.md +257 -0
- package/skills/writing-systems-papers/SKILL.md +184 -0
- package/templates/CLAUDE_MD_TEMPLATE.md +29 -0
- package/templates/EXPERIMENT_LOG_TEMPLATE.md +47 -0
- package/templates/EXPERIMENT_PLAN_TEMPLATE.md +51 -0
- package/templates/EXPERIMENT_PLAN_TEMPLATE_CN.md +53 -0
- package/templates/FINDINGS_TEMPLATE.md +52 -0
- package/templates/IDEA_CANDIDATES_TEMPLATE.md +47 -0
- package/templates/IDEA_CANDIDATES_TEMPLATE_CN.md +47 -0
- package/templates/INVENTION_BRIEF_TEMPLATE.md +87 -0
- package/templates/MANIFEST_TEMPLATE.md +7 -0
- package/templates/NARRATIVE_REPORT_TEMPLATE.md +49 -0
- package/templates/PAPER_PLAN_TEMPLATE.md +47 -0
- package/templates/PATENT_CLAIMS_TEMPLATE.md +78 -0
- package/templates/PATENT_SPECIFICATION_TEMPLATE.md +68 -0
- package/templates/README.md +57 -0
- package/templates/RESEARCH_BRIEF_TEMPLATE.md +35 -0
- package/templates/RESEARCH_BRIEF_TEMPLATE_CN.md +41 -0
- package/templates/RESEARCH_CONTRACT_TEMPLATE.md +60 -0
- package/templates/claude-hooks/corpus_write_guard.json +16 -0
- package/templates/claude-hooks/corpus_write_guard.py +85 -0
- package/templates/claude-hooks/meta_logging.json +74 -0
- package/templates/gitignore-trace.txt +3 -0
- package/tools/__pycache__/check_skills_inventory.cpython-314.pyc +0 -0
- package/tools/arxiv_fetch.py +311 -0
- package/tools/capture_filter.py +126 -0
- package/tools/check_skills_inventory.py +273 -0
- package/tools/convert_skills_to_llm_chat.py +282 -0
- package/tools/copilot_native_evidence.py +818 -0
- package/tools/deepxiv_fetch.py +213 -0
- package/tools/evidence_check.py +212 -0
- package/tools/exa_search.py +425 -0
- package/tools/experiment_queue/README.md +118 -0
- package/tools/experiment_queue/build_manifest.py +44 -0
- package/tools/experiment_queue/queue_manager.py +44 -0
- package/tools/extract_paper_style.py +560 -0
- package/tools/figure_renderer.py +69 -0
- package/tools/forensics_gate.py +669 -0
- package/tools/generate_codex_claude_review_overrides.py +299 -0
- package/tools/idea_discovery_gate.py +256 -0
- package/tools/install_aris.ps1 +1372 -0
- package/tools/install_aris.sh +1370 -0
- package/tools/install_aris_codex.sh +1023 -0
- package/tools/install_aris_copilot.sh +1052 -0
- package/tools/iteration_log.py +143 -0
- package/tools/lint_skills_helpers.sh +84 -0
- package/tools/meta_opt/check_ready.sh +80 -0
- package/tools/meta_opt/log_event.sh +91 -0
- package/tools/meta_opt/trigger_eval.py +280 -0
- package/tools/meta_opt/trigger_evals.sample.json +28 -0
- package/tools/openalex_fetch.py +326 -0
- package/tools/overleaf_audit.sh +104 -0
- package/tools/overleaf_setup.sh +150 -0
- package/tools/paper_illustration_image2.py +62 -0
- package/tools/provenance.py +294 -0
- package/tools/research_wiki.py +1720 -0
- package/tools/review_gate.py +502 -0
- package/tools/run_state.py +399 -0
- package/tools/save_trace.sh +477 -0
- package/tools/semantic_scholar_fetch.py +438 -0
- package/tools/skill-groups.tsv +116 -0
- package/tools/skill_picker.py +238 -0
- package/tools/smart_update.ps1 +521 -0
- package/tools/smart_update.sh +591 -0
- package/tools/smart_update_codex.sh +419 -0
- package/tools/smart_update_copilot.sh +605 -0
- package/tools/threat_scan.py +222 -0
- package/tools/verify_paper_audits.sh +487 -0
- package/tools/verify_papers.py +613 -0
- package/tools/verify_wiki_coverage.sh +176 -0
- package/tools/watchdog.py +485 -0
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""iteration_log.py — overnight-loop stall detection → forced structural pivot.
|
|
3
|
+
|
|
4
|
+
Append-only per-iteration ledger for an unattended research loop. Each tick the
|
|
5
|
+
orchestrator records how many NEW findings the iteration produced — where a "finding"
|
|
6
|
+
is a concrete added entry (new evidence, a falsified hypothesis, a candidate direction),
|
|
7
|
+
NOT a subjective "valuable result". Consecutive zero-finding iterations accumulate a
|
|
8
|
+
stale_count, which drives a forced pivot:
|
|
9
|
+
|
|
10
|
+
stale_count >= 2 → pivot = "structural" (change a STRUCTURAL constraint, not tactical params)
|
|
11
|
+
stale_count >= 4 → pivot = "human" (flag for human attention)
|
|
12
|
+
|
|
13
|
+
This is a **Type-A signal**: it COUNTS entries and changes *direction*; it does NOT judge
|
|
14
|
+
quality — quality/correctness stays with the cross-model jury (shared-references/
|
|
15
|
+
acceptance-gate.md). It only ever says "keep going / change direction," never "good enough".
|
|
16
|
+
|
|
17
|
+
The ledger is a sidecar at `.aris/runs/<run_id>.iterations.jsonl`; it deliberately does
|
|
18
|
+
NOT import or touch run_state.py's done/accepted state machine (only shares the `.aris/runs/`
|
|
19
|
+
dir, with a distinct `.iterations.jsonl` suffix). An optional `direction` per record lets
|
|
20
|
+
the loop's re-generation step reject candidates too close to a tried direction. See
|
|
21
|
+
shared-references/external-cadence.md → "Stall detection & forced structural pivot".
|
|
22
|
+
|
|
23
|
+
Usage:
|
|
24
|
+
python3 iteration_log.py note <root> <run_id> <phase> <new_findings> [--direction "..."]
|
|
25
|
+
python3 iteration_log.py show <root> <run_id>
|
|
26
|
+
"""
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import json
|
|
31
|
+
import sys
|
|
32
|
+
from contextlib import contextmanager
|
|
33
|
+
from datetime import datetime, timezone
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Iterator, Optional
|
|
36
|
+
|
|
37
|
+
try:
|
|
38
|
+
import fcntl
|
|
39
|
+
except ImportError: # pragma: no cover - non-POSIX
|
|
40
|
+
fcntl = None # type: ignore
|
|
41
|
+
|
|
42
|
+
PIVOT_STRUCTURAL_AT = 2 # consecutive zero-finding iterations → force a structural pivot
|
|
43
|
+
ESCALATE_HUMAN_AT = 4 # still stalled → flag for human attention
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _now() -> str:
|
|
47
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _log_path(root: str, run_id: str) -> Path:
|
|
51
|
+
# Same run_id discipline as run_state.py: no path escape.
|
|
52
|
+
safe = "".join(c for c in run_id if c.isalnum() or c in "-_.")
|
|
53
|
+
if not safe or safe != run_id or run_id in (".", ".."):
|
|
54
|
+
raise ValueError(f"invalid run_id {run_id!r} (use [A-Za-z0-9-_.])")
|
|
55
|
+
return Path(root) / ".aris" / "runs" / f"{run_id}.iterations.jsonl"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@contextmanager
|
|
59
|
+
def _lock(path: Path) -> Iterator[None]:
|
|
60
|
+
"""Best-effort advisory lock (single-orchestrator contract; guards a stray resumer)."""
|
|
61
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
62
|
+
if fcntl is None:
|
|
63
|
+
yield
|
|
64
|
+
return
|
|
65
|
+
fh = open(path.with_suffix(".jsonl.lock"), "w")
|
|
66
|
+
try:
|
|
67
|
+
fcntl.flock(fh, fcntl.LOCK_EX)
|
|
68
|
+
yield
|
|
69
|
+
finally:
|
|
70
|
+
try:
|
|
71
|
+
fcntl.flock(fh, fcntl.LOCK_UN)
|
|
72
|
+
finally:
|
|
73
|
+
fh.close()
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _last_stale(path: Path) -> int:
|
|
77
|
+
"""Read the most recent stale_count from the append-only ledger (0 if none)."""
|
|
78
|
+
if not path.is_file():
|
|
79
|
+
return 0
|
|
80
|
+
last = 0
|
|
81
|
+
try:
|
|
82
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
83
|
+
line = line.strip()
|
|
84
|
+
if not line:
|
|
85
|
+
continue
|
|
86
|
+
try:
|
|
87
|
+
last = int(json.loads(line).get("stale_count", last))
|
|
88
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
89
|
+
continue # tolerate a partial/garbled line, keep the last good count
|
|
90
|
+
except OSError:
|
|
91
|
+
return 0
|
|
92
|
+
return last
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def pivot_for(stale_count: int) -> str:
|
|
96
|
+
if stale_count >= ESCALATE_HUMAN_AT:
|
|
97
|
+
return "human"
|
|
98
|
+
if stale_count >= PIVOT_STRUCTURAL_AT:
|
|
99
|
+
return "structural"
|
|
100
|
+
return "none"
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def note(root: str, run_id: str, phase: str, new_findings: int,
|
|
104
|
+
direction: Optional[str] = None) -> dict:
|
|
105
|
+
"""Record one iteration; return {stale_count, pivot}. Append-only; never blocks work."""
|
|
106
|
+
new_findings = int(new_findings)
|
|
107
|
+
if new_findings < 0:
|
|
108
|
+
raise ValueError(f"new_findings must be >= 0, got {new_findings}")
|
|
109
|
+
path = _log_path(root, run_id)
|
|
110
|
+
with _lock(path):
|
|
111
|
+
stale_count = 0 if new_findings > 0 else _last_stale(path) + 1
|
|
112
|
+
pivot = pivot_for(stale_count)
|
|
113
|
+
rec = {"ts": _now(), "phase": phase, "new_findings": new_findings,
|
|
114
|
+
"stale_count": stale_count, "pivot": pivot}
|
|
115
|
+
if direction is not None:
|
|
116
|
+
rec["direction"] = direction
|
|
117
|
+
with open(path, "a", encoding="utf-8") as f:
|
|
118
|
+
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
|
119
|
+
return {"stale_count": stale_count, "pivot": pivot}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def show(root: str, run_id: str) -> str:
|
|
123
|
+
path = _log_path(root, run_id)
|
|
124
|
+
return path.read_text(encoding="utf-8") if path.is_file() else ""
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def main() -> int:
|
|
128
|
+
ap = argparse.ArgumentParser(description="overnight-loop stall detection → forced structural pivot")
|
|
129
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
130
|
+
n = sub.add_parser("note")
|
|
131
|
+
n.add_argument("root"); n.add_argument("run_id"); n.add_argument("phase")
|
|
132
|
+
n.add_argument("new_findings", type=int); n.add_argument("--direction", default=None)
|
|
133
|
+
s = sub.add_parser("show"); s.add_argument("root"); s.add_argument("run_id")
|
|
134
|
+
a = ap.parse_args()
|
|
135
|
+
if a.cmd == "note":
|
|
136
|
+
print(json.dumps(note(a.root, a.run_id, a.phase, a.new_findings, a.direction)))
|
|
137
|
+
elif a.cmd == "show":
|
|
138
|
+
sys.stdout.write(show(a.root, a.run_id))
|
|
139
|
+
return 0
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
if __name__ == "__main__":
|
|
143
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# lint_skills_helpers.sh — Advisory lint for hardcoded `tools/<helper>` references.
|
|
3
|
+
#
|
|
4
|
+
# Per shared-references/integration-contract.md §2, SKILL.md files must
|
|
5
|
+
# resolve helpers via the canonical strict-safe chain
|
|
6
|
+
# .aris/tools/<helper> → tools/<helper> → $ARIS_REPO/tools/<helper>
|
|
7
|
+
# → $ARIS_REPO/tools/<helper> via the global pointer file ~/.aris/repo (#366)
|
|
8
|
+
# (Codex mirror uses the mirror-side chain), NOT hardcode `python3 tools/foo.py`
|
|
9
|
+
# or `bash tools/foo.sh` directly.
|
|
10
|
+
#
|
|
11
|
+
# This script is ADVISORY: it always exits 0 and only prints findings.
|
|
12
|
+
# A future enforcement layer (issue #178) may fail CI on new violations,
|
|
13
|
+
# but Phase 2 keeps the contract gentle so the maintainer is not blocked.
|
|
14
|
+
#
|
|
15
|
+
# Run from the ARIS repo root:
|
|
16
|
+
# bash tools/lint_skills_helpers.sh
|
|
17
|
+
|
|
18
|
+
set -u
|
|
19
|
+
|
|
20
|
+
# Patterns that indicate hardcoded helper invocation (no resolver).
|
|
21
|
+
INVOCATION_PY='python3 tools/(verify_papers|extract_paper_style|paper_illustration_image2|figure_renderer|arxiv_fetch|semantic_scholar_fetch|deepxiv_fetch|exa_search|openalex_fetch|research_wiki|iteration_log)\.py'
|
|
22
|
+
INVOCATION_SH='bash tools/(verify_paper_audits|save_trace|verify_wiki_coverage|overleaf_audit)\.sh'
|
|
23
|
+
|
|
24
|
+
# Files exempted from the lint:
|
|
25
|
+
# - integration-contract.md (canonical docs include ❌ anti-pattern examples)
|
|
26
|
+
# - wiki-helper-resolution.md (defines the chain; layer-2 reference is intentional)
|
|
27
|
+
# - skills-codex/paper-writing/SKILL.md L525 hook JSON example (placeholder for user
|
|
28
|
+
# ~/.claude/settings.json or ~/.codex/config hook, not a SKILL bash block)
|
|
29
|
+
EXEMPTIONS="\
|
|
30
|
+
skills/shared-references/integration-contract.md
|
|
31
|
+
skills/skills-codex/shared-references/integration-contract.md
|
|
32
|
+
skills/shared-references/wiki-helper-resolution.md
|
|
33
|
+
skills/skills-codex/shared-references/wiki-helper-resolution.md
|
|
34
|
+
skills/skills-codex/paper-writing/SKILL.md"
|
|
35
|
+
|
|
36
|
+
is_exempt() {
|
|
37
|
+
case "$EXEMPTIONS" in
|
|
38
|
+
*"$1"*) return 0 ;;
|
|
39
|
+
esac
|
|
40
|
+
return 1
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
violation_count=0
|
|
44
|
+
violation_report=""
|
|
45
|
+
|
|
46
|
+
while IFS= read -r f; do
|
|
47
|
+
if is_exempt "$f"; then
|
|
48
|
+
continue
|
|
49
|
+
fi
|
|
50
|
+
py_hits=$(grep -nE "$INVOCATION_PY" "$f" 2>/dev/null || true)
|
|
51
|
+
sh_hits=$(grep -nE "$INVOCATION_SH" "$f" 2>/dev/null || true)
|
|
52
|
+
if [ -n "$py_hits" ] || [ -n "$sh_hits" ]; then
|
|
53
|
+
violation_count=$((violation_count + 1))
|
|
54
|
+
violation_report="${violation_report}
|
|
55
|
+
=== $f ==="
|
|
56
|
+
[ -n "$py_hits" ] && violation_report="${violation_report}
|
|
57
|
+
${py_hits}"
|
|
58
|
+
[ -n "$sh_hits" ] && violation_report="${violation_report}
|
|
59
|
+
${sh_hits}"
|
|
60
|
+
fi
|
|
61
|
+
done < <(find skills -name '*.md' -type f 2>/dev/null)
|
|
62
|
+
|
|
63
|
+
echo "ARIS helper-resolution lint (advisory)"
|
|
64
|
+
echo "======================================="
|
|
65
|
+
echo "Files with hardcoded \`tools/<helper>\` references: $violation_count"
|
|
66
|
+
|
|
67
|
+
if [ "$violation_count" -gt 0 ]; then
|
|
68
|
+
printf '%s\n\n' "$violation_report"
|
|
69
|
+
echo "Resolution:"
|
|
70
|
+
echo " Migrate each violating SKILL.md to the canonical strict-safe resolver"
|
|
71
|
+
echo " per shared-references/integration-contract.md §2 (assign a semantic"
|
|
72
|
+
echo " variable like \$AUDIT_VERIFIER / \$TRACE_HELPER / \$<NAME>_FETCHER from"
|
|
73
|
+
echo " the four-layer chain, then invoke as \`python3 \"\$VAR\" ...\` or"
|
|
74
|
+
echo " \`bash \"\$VAR\" ...\`)."
|
|
75
|
+
echo ""
|
|
76
|
+
echo " Per-helper policy (Policy A gate / B side-effect / C forensic /"
|
|
77
|
+
echo " D1 cascade / D2 multi-source / E diagnostic) is documented in the"
|
|
78
|
+
echo " \"Per-helper policy assignments\" table of integration-contract.md §2."
|
|
79
|
+
fi
|
|
80
|
+
|
|
81
|
+
echo ""
|
|
82
|
+
echo "Status: advisory (this script never fails CI; warnings only)."
|
|
83
|
+
|
|
84
|
+
exit 0
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# ARIS Meta-Optimize: Readiness Check
|
|
3
|
+
# Called by SessionEnd hook. If enough data has accumulated,
|
|
4
|
+
# outputs a reminder to stdout (injected into Claude's context).
|
|
5
|
+
#
|
|
6
|
+
# Trigger: ≥5 skill invocations since last /meta-optimize run
|
|
7
|
+
|
|
8
|
+
set -euo pipefail
|
|
9
|
+
|
|
10
|
+
ARIS_META_DIR="${CLAUDE_PROJECT_DIR:-.}/.aris/meta"
|
|
11
|
+
EVENTS_FILE="$ARIS_META_DIR/events.jsonl"
|
|
12
|
+
LAST_RUN_FILE="$ARIS_META_DIR/.last_optimize"
|
|
13
|
+
|
|
14
|
+
# No log = nothing to check
|
|
15
|
+
[ -f "$EVENTS_FILE" ] || exit 0
|
|
16
|
+
|
|
17
|
+
# Count skill invocations. Match the full event-field key so a stray
|
|
18
|
+
# "skill_invoke" substring inside args/prompt values can't false-match.
|
|
19
|
+
# Note: `grep -c` prints "0" then exits 1 on no match, so `... || echo 0`
|
|
20
|
+
# would produce a two-line string "0\n0" that breaks the later integer
|
|
21
|
+
# comparison. Use `|| true` to absorb the non-zero exit and keep the
|
|
22
|
+
# captured "0" alone.
|
|
23
|
+
TOTAL_SKILLS=$(grep -cE '"event": *"skill_invoke"' "$EVENTS_FILE" 2>/dev/null || true)
|
|
24
|
+
TOTAL_SKILLS=${TOTAL_SKILLS:-0}
|
|
25
|
+
|
|
26
|
+
# Check when meta-optimize was last run
|
|
27
|
+
if [ -f "$LAST_RUN_FILE" ]; then
|
|
28
|
+
LAST_TS=$(cat "$LAST_RUN_FILE")
|
|
29
|
+
# Count skill invocations AFTER last run.
|
|
30
|
+
# WARNING: do NOT use `$0 > ts` here — every JSONL row starts with `{`
|
|
31
|
+
# (ASCII 0x7B), which sorts greater than every digit in an ISO 8601
|
|
32
|
+
# timestamp, so the comparison would degenerate to "always true" and
|
|
33
|
+
# SINCE_LAST would equal TOTAL_SKILLS. Extract the embedded "ts" value
|
|
34
|
+
# and compare that instead. log_event.sh emits Python json.dumps default
|
|
35
|
+
# format (`"ts": "..."` with a space after the colon), so the regex
|
|
36
|
+
# tolerates 0+ spaces between key and value to stay compatible if that
|
|
37
|
+
# ever changes.
|
|
38
|
+
SINCE_LAST=$(awk -v ts="$LAST_TS" '
|
|
39
|
+
/"event": *"skill_invoke"/ {
|
|
40
|
+
if (match($0, /"ts": *"[^"]+"/)) {
|
|
41
|
+
event_ts = substr($0, RSTART, RLENGTH)
|
|
42
|
+
sub(/^"ts": *"/, "", event_ts)
|
|
43
|
+
sub(/"$/, "", event_ts)
|
|
44
|
+
if (event_ts > ts) count++
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
END { print count + 0 }
|
|
48
|
+
' "$EVENTS_FILE")
|
|
49
|
+
else
|
|
50
|
+
SINCE_LAST=$TOTAL_SKILLS
|
|
51
|
+
fi
|
|
52
|
+
|
|
53
|
+
# Threshold: 5 skill invocations since last optimize
|
|
54
|
+
if [ "$SINCE_LAST" -ge 5 ]; then
|
|
55
|
+
echo "📊 ARIS has logged $SINCE_LAST skill runs since last optimization. Run /meta-optimize to check for improvement opportunities."
|
|
56
|
+
fi
|
|
57
|
+
|
|
58
|
+
# Model-delta trigger (harness diet): a model bump makes existing scaffolding a
|
|
59
|
+
# deletion candidate, independent of usage volume. /meta-optimize records the
|
|
60
|
+
# session model at each run in .last_optimize_model; compare against the LATEST
|
|
61
|
+
# session_start event's model. Absent files (older installs, no session_start
|
|
62
|
+
# yet) silently skip.
|
|
63
|
+
LAST_MODEL_FILE="$ARIS_META_DIR/.last_optimize_model"
|
|
64
|
+
if [ -f "$LAST_MODEL_FILE" ]; then
|
|
65
|
+
CURRENT_MODEL=$(awk '
|
|
66
|
+
/"event": *"session_start"/ {
|
|
67
|
+
if (match($0, /"model": *"[^"]+"/)) {
|
|
68
|
+
m = substr($0, RSTART, RLENGTH)
|
|
69
|
+
sub(/^"model": *"/, "", m)
|
|
70
|
+
sub(/"$/, "", m)
|
|
71
|
+
latest = m
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
END { if (latest != "") print latest }
|
|
75
|
+
' "$EVENTS_FILE")
|
|
76
|
+
LAST_MODEL=$(cat "$LAST_MODEL_FILE")
|
|
77
|
+
if [ -n "$CURRENT_MODEL" ] && [ -n "$LAST_MODEL" ] && [ "$CURRENT_MODEL" != "$LAST_MODEL" ]; then
|
|
78
|
+
echo "🔁 Model changed since last optimization ($LAST_MODEL → $CURRENT_MODEL). Run /meta-optimize — a model bump makes existing scaffolding a deletion candidate (harness diet)."
|
|
79
|
+
fi
|
|
80
|
+
fi
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# ARIS Meta-Optimize: Event Logger
|
|
3
|
+
# Reads Claude Code hook JSON from stdin, extracts key fields,
|
|
4
|
+
# appends structured event to BOTH project-level and global logs.
|
|
5
|
+
#
|
|
6
|
+
# Called automatically by Claude Code hooks (PostToolUse, UserPromptSubmit, etc.)
|
|
7
|
+
# Input: JSON via stdin (Claude Code hook payload)
|
|
8
|
+
# Output:
|
|
9
|
+
# Project: $CLAUDE_PROJECT_DIR/.aris/meta/events.jsonl (project-specific details)
|
|
10
|
+
# Global: ~/.aris/meta/events.jsonl (cross-project trends)
|
|
11
|
+
|
|
12
|
+
set -euo pipefail
|
|
13
|
+
|
|
14
|
+
PROJECT_META="${CLAUDE_PROJECT_DIR:-.}/.aris/meta"
|
|
15
|
+
GLOBAL_META="$HOME/.aris/meta"
|
|
16
|
+
mkdir -p "$PROJECT_META" "$GLOBAL_META"
|
|
17
|
+
|
|
18
|
+
# Read stdin payload into env var (cannot use heredoc + herestring simultaneously)
|
|
19
|
+
export ARIS_HOOK_PAYLOAD="$(cat)"
|
|
20
|
+
|
|
21
|
+
python3 - "$PROJECT_META/events.jsonl" "$GLOBAL_META/events.jsonl" << 'PYEOF'
|
|
22
|
+
import json, sys, os
|
|
23
|
+
from datetime import datetime, timezone
|
|
24
|
+
|
|
25
|
+
project_log = sys.argv[1]
|
|
26
|
+
global_log = sys.argv[2]
|
|
27
|
+
|
|
28
|
+
raw = os.environ.get("ARIS_HOOK_PAYLOAD", "").strip()
|
|
29
|
+
if not raw:
|
|
30
|
+
sys.exit(0)
|
|
31
|
+
try:
|
|
32
|
+
p = json.loads(raw)
|
|
33
|
+
except json.JSONDecodeError:
|
|
34
|
+
sys.exit(0)
|
|
35
|
+
|
|
36
|
+
event_name = p.get("hook_event_name", "unknown")
|
|
37
|
+
session_id = p.get("session_id", "")
|
|
38
|
+
project_dir = os.environ.get("CLAUDE_PROJECT_DIR", "")
|
|
39
|
+
ts = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
40
|
+
|
|
41
|
+
record = {"ts": ts, "session": session_id, "event": event_name}
|
|
42
|
+
|
|
43
|
+
if event_name in ("PostToolUse", "PostToolUseFailure"):
|
|
44
|
+
tool_name = p.get("tool_name", "")
|
|
45
|
+
tool_input = p.get("tool_input", {})
|
|
46
|
+
record["tool"] = tool_name
|
|
47
|
+
|
|
48
|
+
if event_name == "PostToolUseFailure":
|
|
49
|
+
record["event"] = "tool_failure"
|
|
50
|
+
|
|
51
|
+
if tool_name == "Skill":
|
|
52
|
+
record["event"] = "skill_invoke"
|
|
53
|
+
record["skill"] = tool_input.get("skill", "")
|
|
54
|
+
record["args"] = tool_input.get("args", "")
|
|
55
|
+
elif tool_name == "Bash":
|
|
56
|
+
record["input_summary"] = tool_input.get("command", "")[:200]
|
|
57
|
+
elif tool_name in ("Edit", "Write", "Read"):
|
|
58
|
+
record["input_summary"] = tool_input.get("file_path", "")
|
|
59
|
+
elif tool_name.startswith("mcp__codex__"):
|
|
60
|
+
record["event"] = "codex_call"
|
|
61
|
+
record["input_summary"] = tool_input.get("prompt", "")[:150]
|
|
62
|
+
|
|
63
|
+
elif event_name == "UserPromptSubmit":
|
|
64
|
+
prompt = p.get("prompt", "")
|
|
65
|
+
if prompt.startswith("/"):
|
|
66
|
+
parts = prompt.split(None, 1)
|
|
67
|
+
record["event"] = "slash_command"
|
|
68
|
+
record["command"] = parts[0]
|
|
69
|
+
record["args"] = parts[1] if len(parts) > 1 else ""
|
|
70
|
+
else:
|
|
71
|
+
record["event"] = "user_prompt"
|
|
72
|
+
record["prompt_preview"] = prompt[:100]
|
|
73
|
+
|
|
74
|
+
elif event_name == "SessionStart":
|
|
75
|
+
record["event"] = "session_start"
|
|
76
|
+
record["source"] = p.get("source", "")
|
|
77
|
+
record["model"] = p.get("model", "")
|
|
78
|
+
|
|
79
|
+
elif event_name == "SessionEnd":
|
|
80
|
+
record["event"] = "session_end"
|
|
81
|
+
|
|
82
|
+
# Write to project-level log
|
|
83
|
+
with open(project_log, "a") as f:
|
|
84
|
+
f.write(json.dumps(record, ensure_ascii=False) + "\n")
|
|
85
|
+
|
|
86
|
+
# Write to global log with project tag
|
|
87
|
+
global_record = record.copy()
|
|
88
|
+
global_record["project"] = os.path.basename(project_dir) if project_dir else "unknown"
|
|
89
|
+
with open(global_log, "a") as f:
|
|
90
|
+
f.write(json.dumps(global_record, ensure_ascii=False) + "\n")
|
|
91
|
+
PYEOF
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""trigger_eval.py — measure whether a skill's `description` actually triggers.
|
|
3
|
+
|
|
4
|
+
ARIS's known pain: with 80+ skills installed, Claude Code sometimes fails to
|
|
5
|
+
invoke the right skill for a query it should handle — and until now the only
|
|
6
|
+
lever (the frontmatter `description`) was tuned by pure judgment, with zero
|
|
7
|
+
measurement. This tool turns trigger behavior into a number.
|
|
8
|
+
|
|
9
|
+
MEASURE-ONLY BY DESIGN. It never rewrites a description. Its report is
|
|
10
|
+
EVIDENCE for a /meta-optimize proposal (which lands only via the human-gated
|
|
11
|
+
/meta-apply) — "a loop can drive, never acquit" applies to description tuning
|
|
12
|
+
too.
|
|
13
|
+
|
|
14
|
+
How it works (adapted from Anthropic's Claude Science `skill-creator`
|
|
15
|
+
run_eval.py — Apache-2.0; ported off its host.* runtime onto plain `claude -p`):
|
|
16
|
+
- For each (skill, query), run `claude -p <query> --output-format stream-json
|
|
17
|
+
--max-turns 1 --permission-mode plan --disallowed-tools Bash Write Edit …`
|
|
18
|
+
as a subprocess FROM A NEUTRAL TEMP CWD. The user-level ~/.claude/skills
|
|
19
|
+
corpus is loaded as usual, so the measurement happens under the REALISTIC
|
|
20
|
+
long installed list — the exact condition under which omission happens (an
|
|
21
|
+
isolated one-skill sandbox would trivially inflate trigger rates).
|
|
22
|
+
- Parse the stream for the first assistant turn's tool_use blocks. A `Skill`
|
|
23
|
+
tool call with input.skill == target counts as a TRIGGER; a Skill call for a
|
|
24
|
+
different skill is a CONFUSION (recorded by name — the confusion matrix is
|
|
25
|
+
the interesting part for the long-list problem); no Skill call is a MISS.
|
|
26
|
+
Reading the target's SKILL.md via the Read tool counts as a trigger too
|
|
27
|
+
(secondary signal).
|
|
28
|
+
- SAFETY: `--permission-mode plan` blocks every side-effecting tool (Bash,
|
|
29
|
+
Write, Edit, …) from executing, so a probed skill's own commands (e.g.
|
|
30
|
+
check-gpu's ssh, vast-gpu's rentals) do NOT run — we observe only which tool
|
|
31
|
+
the model REACHED FOR. The read-only tools we score on (a `Skill` load, a
|
|
32
|
+
`Read` of a SKILL.md) may execute, and both are side-effect-free.
|
|
33
|
+
`--disallowed-tools` denies the stateful tools explicitly as belt-and-braces,
|
|
34
|
+
and `--no-session-persistence` avoids leaving session artifacts. This is a
|
|
35
|
+
measurement, not a sandbox — it does not stop the user's own SessionStart
|
|
36
|
+
hooks (their normal per-session behavior), it stops the PROBED WORK.
|
|
37
|
+
|
|
38
|
+
Query-set methodology (see trigger_evals.sample.json): queries must PARAPHRASE
|
|
39
|
+
user intent, never quote the description's own trigger phrases verbatim — a
|
|
40
|
+
query containing the literal trigger string is trivially positive and measures
|
|
41
|
+
nothing. Optional negative queries (expect: none) measure false-triggering.
|
|
42
|
+
|
|
43
|
+
Usage:
|
|
44
|
+
python3 tools/meta_opt/trigger_eval.py --eval-file tools/meta_opt/trigger_evals.sample.json \\
|
|
45
|
+
[--skills check-gpu,research-lit] [--samples 2] [--model haiku] \\
|
|
46
|
+
[--out .aris/meta/trigger_report.json] [--timeout 120]
|
|
47
|
+
|
|
48
|
+
Exit code: 0 on completed run (regardless of rates), 2 on setup error.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
import argparse
|
|
52
|
+
import json
|
|
53
|
+
import os
|
|
54
|
+
import subprocess
|
|
55
|
+
import sys
|
|
56
|
+
import tempfile
|
|
57
|
+
from pathlib import Path
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# ---------------------------------------------------------------- pure logic
|
|
61
|
+
|
|
62
|
+
def parse_stream_tool_uses(stream_text: str):
|
|
63
|
+
"""Extract (tool_name, tool_input) pairs from `claude -p` stream-json output.
|
|
64
|
+
|
|
65
|
+
Each line is a JSON event; assistant events carry message.content lists in
|
|
66
|
+
which tool_use blocks appear. Malformed lines are skipped (the stream can
|
|
67
|
+
interleave non-JSON stderr noise when things go wrong).
|
|
68
|
+
"""
|
|
69
|
+
uses = []
|
|
70
|
+
for line in stream_text.splitlines():
|
|
71
|
+
line = line.strip()
|
|
72
|
+
if not line or not line.startswith("{"):
|
|
73
|
+
continue
|
|
74
|
+
try:
|
|
75
|
+
ev = json.loads(line)
|
|
76
|
+
except json.JSONDecodeError:
|
|
77
|
+
continue
|
|
78
|
+
if ev.get("type") != "assistant":
|
|
79
|
+
continue
|
|
80
|
+
content = (ev.get("message") or {}).get("content") or []
|
|
81
|
+
for block in content:
|
|
82
|
+
if isinstance(block, dict) and block.get("type") == "tool_use":
|
|
83
|
+
uses.append((block.get("name") or "", block.get("input") or {}))
|
|
84
|
+
return uses
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def classify(tool_uses, target_skill: str):
|
|
88
|
+
"""Classify one probe run: ('trigger'|'confusion'|'miss', detail).
|
|
89
|
+
|
|
90
|
+
Trigger: a Skill call for the target (exact id, or a `plugin:target`
|
|
91
|
+
namespaced form — the latter tagged in detail so a namespaced match is
|
|
92
|
+
never silently indistinguishable from an exact one), or a Read of the
|
|
93
|
+
target's SKILL.md.
|
|
94
|
+
Confusion: the FIRST Skill call named a different skill (detail = its name).
|
|
95
|
+
Miss: no skill engagement at all.
|
|
96
|
+
"""
|
|
97
|
+
for name, inp in tool_uses:
|
|
98
|
+
if name == "Skill":
|
|
99
|
+
invoked = (inp.get("skill") or "").strip()
|
|
100
|
+
if invoked == target_skill:
|
|
101
|
+
return "trigger", invoked
|
|
102
|
+
# plugin-namespaced form "plugin:skill": a trigger only if the tail
|
|
103
|
+
# equals the target AND the target itself is bare (not namespaced),
|
|
104
|
+
# surfaced distinctly so a human can spot a plugin/bare collision.
|
|
105
|
+
if ":" in invoked and invoked.split(":")[-1] == target_skill \
|
|
106
|
+
and ":" not in target_skill:
|
|
107
|
+
return "trigger", f"{invoked} (namespaced→{target_skill})"
|
|
108
|
+
return "confusion", invoked
|
|
109
|
+
if name == "Read":
|
|
110
|
+
path = str(inp.get("file_path") or "")
|
|
111
|
+
if f"/skills/{target_skill}/SKILL.md" in path:
|
|
112
|
+
return "trigger", path
|
|
113
|
+
return "miss", ""
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def aggregate(records):
|
|
117
|
+
"""records: list of {skill, query, outcome, detail} → per-skill summary."""
|
|
118
|
+
out = {}
|
|
119
|
+
for r in records:
|
|
120
|
+
s = out.setdefault(r["skill"], {
|
|
121
|
+
"probes": 0, "triggers": 0, "misses": 0, "errors": 0,
|
|
122
|
+
"confusions": {}, "queries": {},
|
|
123
|
+
})
|
|
124
|
+
s["probes"] += 1
|
|
125
|
+
q = s["queries"].setdefault(r["query"], {"trigger": 0, "confusion": 0,
|
|
126
|
+
"miss": 0, "error": 0})
|
|
127
|
+
q[r["outcome"]] += 1
|
|
128
|
+
if r["outcome"] == "trigger":
|
|
129
|
+
s["triggers"] += 1
|
|
130
|
+
elif r["outcome"] == "miss":
|
|
131
|
+
s["misses"] += 1
|
|
132
|
+
elif r["outcome"] == "error":
|
|
133
|
+
s["errors"] += 1
|
|
134
|
+
elif r["outcome"] == "confusion":
|
|
135
|
+
s["confusions"][r["detail"]] = s["confusions"].get(r["detail"], 0) + 1
|
|
136
|
+
for s in out.values():
|
|
137
|
+
graded = s["probes"] - s["errors"]
|
|
138
|
+
s["trigger_rate"] = round(s["triggers"] / graded, 3) if graded else None
|
|
139
|
+
return out
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
# ------------------------------------------------------------------- probing
|
|
143
|
+
|
|
144
|
+
# Stateful tools that must never execute during a probe (belt-and-braces on top
|
|
145
|
+
# of --permission-mode plan, which already blocks side-effecting tools).
|
|
146
|
+
_DENY_TOOLS = ["Bash", "Write", "Edit", "NotebookEdit", "WebFetch"]
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
# `--max-turns 1` deliberately caps the probe at one turn, so the CLI ends with
|
|
150
|
+
# result subtype `error_max_turns` and a NONZERO exit — that is the EXPECTED,
|
|
151
|
+
# successful termination for a probe, NOT a failure. Only other errors (auth,
|
|
152
|
+
# startup/hook failure, execution error) count as a real error.
|
|
153
|
+
_EXPECTED_TERMINATION = "error_max_turns"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _stream_real_error(stream_text: str) -> bool:
|
|
157
|
+
"""True iff the stream carries a genuine terminal error — an `is_error`
|
|
158
|
+
result whose subtype is NOT the expected max-turns cap. Auth/hook failures
|
|
159
|
+
that still emit JSON are caught here so they are graded `error`, never a
|
|
160
|
+
`miss` that would silently corrupt the trigger rate."""
|
|
161
|
+
for line in stream_text.splitlines():
|
|
162
|
+
line = line.strip()
|
|
163
|
+
if not line.startswith("{"):
|
|
164
|
+
continue
|
|
165
|
+
try:
|
|
166
|
+
ev = json.loads(line)
|
|
167
|
+
except json.JSONDecodeError:
|
|
168
|
+
continue
|
|
169
|
+
if ev.get("type") == "result" and ev.get("is_error") \
|
|
170
|
+
and ev.get("subtype") != _EXPECTED_TERMINATION:
|
|
171
|
+
return True
|
|
172
|
+
return False
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _stream_has_assistant(stream_text: str) -> bool:
|
|
176
|
+
"""True iff the model produced at least one assistant turn — i.e. the probe
|
|
177
|
+
ran far enough to be gradeable (even if it then hit the max-turns cap)."""
|
|
178
|
+
for line in stream_text.splitlines():
|
|
179
|
+
line = line.strip()
|
|
180
|
+
if not line.startswith("{"):
|
|
181
|
+
continue
|
|
182
|
+
try:
|
|
183
|
+
if json.loads(line).get("type") == "assistant":
|
|
184
|
+
return True
|
|
185
|
+
except json.JSONDecodeError:
|
|
186
|
+
continue
|
|
187
|
+
return False
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def run_probe(query: str, model: str | None, timeout: int, cwd: str) -> str:
|
|
191
|
+
"""One `claude -p` probe; returns raw stream-json text. Raises RuntimeError
|
|
192
|
+
on a REAL failure (genuine error event, or nonzero exit with no assistant
|
|
193
|
+
turn at all) so the caller records `error` rather than a rate-corrupting
|
|
194
|
+
`miss`. The expected max-turns termination (nonzero exit + assistant turn
|
|
195
|
+
present) is a normal, gradeable result."""
|
|
196
|
+
cmd = ["claude", "-p", "--output-format", "stream-json", "--verbose",
|
|
197
|
+
"--max-turns", "1", "--permission-mode", "plan",
|
|
198
|
+
"--no-session-persistence", "--disallowed-tools", *_DENY_TOOLS]
|
|
199
|
+
if model:
|
|
200
|
+
cmd += ["--model", model]
|
|
201
|
+
# Allow nesting claude -p inside a Claude Code session (same pattern as the
|
|
202
|
+
# Apache-2.0 source): the CLAUDECODE guard is for interactive terminals.
|
|
203
|
+
env = {k: v for k, v in os.environ.items() if k != "CLAUDECODE"}
|
|
204
|
+
result = subprocess.run(cmd, input=query, capture_output=True, text=True,
|
|
205
|
+
env=env, timeout=timeout, cwd=cwd)
|
|
206
|
+
if _stream_real_error(result.stdout):
|
|
207
|
+
raise RuntimeError("claude -p stream carried a terminal error result event")
|
|
208
|
+
if _stream_has_assistant(result.stdout):
|
|
209
|
+
return result.stdout # gradeable (max-turns cap is fine)
|
|
210
|
+
if result.returncode != 0: # no assistant turn AND failed = real
|
|
211
|
+
raise RuntimeError(f"claude -p exited {result.returncode} with no assistant "
|
|
212
|
+
f"turn: {result.stderr.strip()[:300]}")
|
|
213
|
+
return result.stdout # clean, no tool call → graded miss
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def main(argv=None) -> int:
|
|
217
|
+
ap = argparse.ArgumentParser(description="Measure skill-description trigger rates.")
|
|
218
|
+
ap.add_argument("--eval-file", required=True,
|
|
219
|
+
help='JSON: {"<skill>": ["query", ...], ...}')
|
|
220
|
+
ap.add_argument("--skills", default="",
|
|
221
|
+
help="comma-separated subset of skills to probe (default: all in file)")
|
|
222
|
+
ap.add_argument("--samples", type=int, default=3,
|
|
223
|
+
help="probes per query (trigger behavior is stochastic; the "
|
|
224
|
+
"default 3 matches the upstream eval — samples=1 is too "
|
|
225
|
+
"noisy to act on)")
|
|
226
|
+
ap.add_argument("--model", default=None,
|
|
227
|
+
help="model override for probes (default: claude CLI default). "
|
|
228
|
+
"NB: trigger behavior is model-dependent — compare like with like.")
|
|
229
|
+
ap.add_argument("--timeout", type=int, default=120)
|
|
230
|
+
ap.add_argument("--out", default=".aris/meta/trigger_report.json")
|
|
231
|
+
args = ap.parse_args(argv)
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
evals = json.loads(Path(args.eval_file).read_text(encoding="utf-8"))
|
|
235
|
+
except (OSError, json.JSONDecodeError) as e:
|
|
236
|
+
print(f"ERROR: cannot read eval file: {e}", file=sys.stderr)
|
|
237
|
+
return 2
|
|
238
|
+
subset = {s.strip() for s in args.skills.split(",") if s.strip()}
|
|
239
|
+
targets = {k: v for k, v in evals.items()
|
|
240
|
+
if (not subset or k in subset) and not k.startswith("_")}
|
|
241
|
+
if not targets:
|
|
242
|
+
print("ERROR: no skills selected", file=sys.stderr)
|
|
243
|
+
return 2
|
|
244
|
+
|
|
245
|
+
records = []
|
|
246
|
+
# Neutral cwd: no project-level .claude/, so probes see exactly the
|
|
247
|
+
# user-level installed corpus — the realistic long list.
|
|
248
|
+
with tempfile.TemporaryDirectory(prefix="trigger-eval-") as neutral_cwd:
|
|
249
|
+
for skill, queries in targets.items():
|
|
250
|
+
for query in queries:
|
|
251
|
+
for _ in range(args.samples):
|
|
252
|
+
try:
|
|
253
|
+
stream = run_probe(query, args.model, args.timeout, neutral_cwd)
|
|
254
|
+
outcome, detail = classify(parse_stream_tool_uses(stream), skill)
|
|
255
|
+
except (RuntimeError, subprocess.TimeoutExpired) as e:
|
|
256
|
+
outcome, detail = "error", str(e)[:200]
|
|
257
|
+
records.append({"skill": skill, "query": query,
|
|
258
|
+
"outcome": outcome, "detail": detail})
|
|
259
|
+
print(f" [{outcome:9}] {skill} ← {query[:60]!r}"
|
|
260
|
+
+ (f" → {detail}" if outcome == "confusion" else ""))
|
|
261
|
+
|
|
262
|
+
summary = aggregate(records)
|
|
263
|
+
report = {"model": args.model or "cli-default", "samples": args.samples,
|
|
264
|
+
"skills": summary, "records": records}
|
|
265
|
+
out = Path(args.out)
|
|
266
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
267
|
+
out.write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
|
|
268
|
+
|
|
269
|
+
print("\nskill rate probes confusions")
|
|
270
|
+
for name, s in sorted(summary.items()):
|
|
271
|
+
conf = ", ".join(f"{k}×{v}" for k, v in
|
|
272
|
+
sorted(s["confusions"].items(), key=lambda kv: -kv[1])) or "-"
|
|
273
|
+
rate = "n/a " if s["trigger_rate"] is None else f"{s['trigger_rate']:.2f}"
|
|
274
|
+
print(f"{name:30} {rate} {s['probes']:4} {conf}")
|
|
275
|
+
print(f"\nreport → {out}")
|
|
276
|
+
return 0
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
if __name__ == "__main__":
|
|
280
|
+
sys.exit(main())
|