@iowarp/clio-coder 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/CONTRIBUTING.md +142 -52
- package/README.md +434 -473
- package/SECURITY.md +2 -1
- package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
- package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
- package/dist/builtins-IA7V7FUC.js +22 -0
- package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
- package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
- package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
- package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
- package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
- package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
- package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
- package/dist/chunk-3EBYEESD.js +314 -0
- package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
- package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
- package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
- package/dist/chunk-4UVU7BJ5.js +39 -0
- package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
- package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
- package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
- package/dist/chunk-5MEZN6CB.js +1334 -0
- package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
- package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
- package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
- package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
- package/dist/chunk-7DRAWPTZ.js +360 -0
- package/dist/chunk-7E7I3WLS.js +3762 -0
- package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
- package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
- package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
- package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
- package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
- package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
- package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
- package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
- package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
- package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
- package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
- package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
- package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
- package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
- package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
- package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
- package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
- package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
- package/dist/chunk-GYV6VZOC.js +26 -0
- package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
- package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
- package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
- package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
- package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
- package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
- package/dist/chunk-IJNZMHLA.js +101 -0
- package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
- package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
- package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
- package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
- package/dist/chunk-JEQ3XTHC.js +42 -0
- package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
- package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
- package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
- package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
- package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
- package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
- package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
- package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
- package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
- package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
- package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
- package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
- package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
- package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
- package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
- package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
- package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
- package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
- package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
- package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
- package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
- package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
- package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
- package/dist/chunk-R6Q67RJH.js +134 -0
- package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
- package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
- package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
- package/dist/chunk-SKHCAU7K.js +385 -0
- package/dist/chunk-SZAA6XDG.js +30 -0
- package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
- package/dist/chunk-UOIZ7DA4.js +41 -0
- package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
- package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
- package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
- package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
- package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
- package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
- package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
- package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
- package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
- package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
- package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
- package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
- package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
- package/dist/chunk-XRZT5WY5.js +47 -0
- package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
- package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
- package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
- package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
- package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
- package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
- package/dist/cli/index.js +41 -39
- package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
- package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
- package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
- package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
- package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
- package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
- package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
- package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
- package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
- package/dist/context-map-COB37XXN.js +505 -0
- package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
- package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
- package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
- package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
- package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
- package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
- package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
- package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
- package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
- package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
- package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
- package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
- package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
- package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
- package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
- package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
- package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
- package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
- package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
- package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
- package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
- package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
- package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
- package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
- package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
- package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
- package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
- package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
- package/dist/reset-EAJFFJVB.js +344 -0
- package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
- package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
- package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
- package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
- package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
- package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
- package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
- package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
- package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
- package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
- package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
- package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
- package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
- package/dist/uninstall-5PEVOE5B.js +408 -0
- package/dist/upgrade-M4WXY6KN.js +303 -0
- package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
- package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
- package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
- package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
- package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
- package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
- package/dist/worker/entry.js +90 -74
- package/docs/README.md +176 -81
- package/docs/{acp.md → architecture/acp.md} +36 -20
- package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
- package/docs/{architecture.md → architecture/architecture.md} +43 -22
- package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
- package/docs/architecture/artifact-versions.md +90 -0
- package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
- package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
- package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
- package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
- package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
- package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
- package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
- package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
- package/docs/{observability.md → architecture/observability.md} +38 -14
- package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
- package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
- package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
- package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
- package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
- package/docs/architecture/time-conventions.md +125 -0
- package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
- package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
- package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
- package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
- package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
- package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
- package/docs/guide/configuration-reference.md +1163 -0
- package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
- package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
- package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
- package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
- package/docs/{glossary.md → guide/glossary.md} +14 -11
- package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
- package/docs/guide/panes-and-files.md +290 -0
- package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
- package/docs/{resource-library.md → guide/resource-library.md} +13 -4
- package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
- package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
- package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
- package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
- package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
- package/docs/process/development-pipeline.md +152 -0
- package/docs/process/documentation-coverage.md +100 -0
- package/docs/process/documentation-guide.md +187 -0
- package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
- package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
- package/docs/{evolution.md → process/evolution.md} +2 -2
- package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
- package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
- package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
- package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
- package/evals/README.md +2 -2
- package/evals/behavioral-model.yaml +3 -2
- package/package.json +10 -8
- package/skills/README.md +52 -41
- package/skills/coding/ast-grep/SKILL.md +102 -31
- package/skills/coding/ast-grep/evals.md +26 -0
- package/skills/coding/coding-standards/SKILL.md +41 -6
- package/skills/coding/coding-standards/evals.md +23 -0
- package/skills/coding/prototype/SKILL.md +88 -29
- package/skills/coding/prototype/evals.md +19 -0
- package/skills/coding/tdd/SKILL.md +81 -54
- package/skills/coding/tdd/evals.md +20 -0
- package/skills/context/context-handoff/SKILL.md +44 -3
- package/skills/context/context-handoff/evals.md +44 -0
- package/skills/context/context-prime/SKILL.md +46 -16
- package/skills/context/context-prime/evals.md +45 -0
- package/skills/git/branch-closeout/SKILL.md +132 -0
- package/skills/git/branch-closeout/evals.md +133 -0
- package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
- package/skills/git/file-ticket/SKILL.md +78 -64
- package/skills/git/file-ticket/assets/issue-template.md +22 -0
- package/skills/git/file-ticket/evals.md +31 -26
- package/skills/git/file-ticket/references/issue-discovery.md +49 -0
- package/skills/git/fix-issue/SKILL.md +88 -65
- package/skills/git/fix-issue/evals.md +35 -31
- package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
- package/skills/git/resolve-merge-conflicts/evals.md +52 -25
- package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
- package/skills/git/ship/SKILL.md +103 -67
- package/skills/git/ship/assets/pr-template.md +21 -0
- package/skills/git/ship/evals.md +44 -28
- package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
- package/skills/git/worktree-create/SKILL.md +80 -50
- package/skills/git/worktree-create/evals.md +40 -33
- package/skills/git/worktree-create/references/worktree-setup.md +62 -66
- package/skills/git/worktree-merge/SKILL.md +112 -65
- package/skills/git/worktree-merge/evals.md +42 -34
- package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
- package/skills/meta/clio-coder-dev/SKILL.md +9 -5
- package/skills/meta/clio-coder-dev/evals.md +3 -2
- package/skills/meta/clio-coder-test/SKILL.md +102 -95
- package/skills/meta/clio-coder-test/evals.md +9 -4
- package/skills/meta/clio-coder-test/references/harness.md +100 -124
- package/skills/meta/clio-coder-test/references/test-map.md +77 -50
- package/skills/meta/credentials/SKILL.md +2 -2
- package/skills/meta/find-skills/SKILL.md +2 -2
- package/skills/meta/herdr/SKILL.md +2 -2
- package/skills/meta/skill-craft/SKILL.md +22 -16
- package/skills/planning/archify/SKILL.md +196 -0
- package/skills/planning/archify/evals.md +65 -0
- package/skills/planning/architecture/SKILL.md +62 -13
- package/skills/planning/architecture/evals.md +65 -0
- package/skills/planning/backlog/SKILL.md +131 -15
- package/skills/planning/backlog/evals.md +142 -0
- package/skills/planning/prd/SKILL.md +47 -7
- package/skills/planning/prd/evals.md +54 -0
- package/skills/planning/product-intent/SKILL.md +58 -3
- package/skills/planning/product-intent/evals.md +70 -0
- package/skills/planning/tech-spec/SKILL.md +54 -3
- package/skills/planning/tech-spec/evals.md +73 -0
- package/skills/registry.yaml +70 -62
- package/skills/remote.yaml +13 -0
- package/skills/research/arxiv-literature/SKILL.md +77 -19
- package/skills/research/arxiv-literature/evals.md +50 -0
- package/skills/research/experiment-protocol/SKILL.md +21 -2
- package/skills/research/experiment-protocol/evals.md +23 -0
- package/skills/research/scientific-debugging/SKILL.md +24 -2
- package/skills/research/scientific-debugging/evals.md +18 -0
- package/skills/research/scientific-modernization/SKILL.md +27 -2
- package/skills/research/scientific-modernization/evals.md +27 -0
- package/skills/skill-marketplace.json +97 -62
- package/skills/workflow/cut-it/SKILL.md +66 -6
- package/skills/workflow/cut-it/evals.md +101 -0
- package/skills/workflow/design-council/SKILL.md +118 -28
- package/skills/workflow/design-council/evals.md +161 -0
- package/skills/workflow/grill-me/SKILL.md +87 -11
- package/skills/workflow/grill-me/evals.md +153 -0
- package/skills/workflow/workflow-distiller/SKILL.md +77 -18
- package/skills/workflow/workflow-distiller/evals.md +118 -0
- package/src/cli/args.ts +2 -2
- package/src/cli/bootstrap-generate.ts +1 -1
- package/src/cli/config-inspect.ts +65 -12
- package/src/cli/configure-interop.ts +105 -13
- package/src/cli/configure-oauth.ts +57 -0
- package/src/cli/configure-onboarding.ts +980 -0
- package/src/cli/configure-target.ts +594 -0
- package/src/cli/configure.ts +1082 -532
- package/src/cli/context-map.ts +114 -0
- package/src/cli/context.ts +4 -0
- package/src/cli/docs.ts +22 -14
- package/src/cli/doctor-naming.ts +5 -5
- package/src/cli/doctor-toolchain.ts +3 -3
- package/src/cli/eval.ts +1 -2
- package/src/cli/extensions.ts +2 -1
- package/src/cli/fleet.ts +1 -1
- package/src/cli/index.ts +3 -1
- package/src/cli/internal-dispatch.ts +3 -4
- package/src/cli/lifecycle-presenter.ts +436 -0
- package/src/cli/models.ts +10 -2
- package/src/cli/modes/print.ts +5 -1
- package/src/cli/panes.ts +19 -5
- package/src/cli/reset.ts +228 -106
- package/src/cli/run.ts +9 -4
- package/src/cli/select.ts +664 -0
- package/src/cli/share.ts +5 -1
- package/src/cli/skills-eval.ts +3 -3
- package/src/cli/skills.ts +9 -2
- package/src/cli/targets.ts +5 -6
- package/src/cli/trace.ts +55 -4
- package/src/cli/uninstall.ts +233 -165
- package/src/cli/upgrade.ts +204 -149
- package/src/cli/usage.ts +86 -27
- package/src/cli/validate-model.ts +3 -3
- package/src/cli/wiki-generate.ts +1 -1
- package/src/core/artifact-paths.ts +1 -1
- package/src/core/bash-exec.ts +131 -86
- package/src/core/bus-events.ts +51 -6
- package/src/core/config.ts +61 -1
- package/src/core/defaults.ts +7 -4
- package/src/core/dispatch-outcome.ts +16 -0
- package/src/core/external-diagnostic.ts +44 -0
- package/src/core/gateway-routing.ts +157 -0
- package/src/core/guardrails.ts +10 -49
- package/src/core/prompt-hint.ts +9 -0
- package/src/core/safe-exec.ts +17 -2
- package/src/core/skill-activation.ts +89 -2
- package/src/domains/agents/builtins/architect.md +2 -3
- package/src/domains/agents/builtins/coder.md +3 -2
- package/src/domains/agents/builtins/debugger.md +2 -2
- package/src/domains/agents/builtins/documenter.md +2 -2
- package/src/domains/agents/builtins/git-master.md +1 -1
- package/src/domains/agents/builtins/oracle.md +1 -1
- package/src/domains/agents/builtins/provenance.md +1 -1
- package/src/domains/agents/builtins/researcher.md +1 -1
- package/src/domains/agents/builtins/scout.md +1 -1
- package/src/domains/agents/builtins/tester.md +2 -2
- package/src/domains/agents/builtins/verifier.md +2 -2
- package/src/domains/agents/builtins/wiki-writer.md +1 -1
- package/src/domains/agents/builtins/world-knowledge.md +31 -0
- package/src/domains/agents/catalog.ts +13 -15
- package/src/domains/agents/contract.ts +2 -0
- package/src/domains/agents/extension.ts +23 -1
- package/src/domains/agents/result-contract.ts +70 -0
- package/src/domains/config/keybindings.ts +8 -0
- package/src/domains/context/extension.ts +0 -3
- package/src/domains/context/wiki/map-seed.ts +589 -0
- package/src/domains/context/wiki/plan.ts +2 -2
- package/src/domains/context/working-set/path-index.ts +1 -0
- package/src/domains/dispatch/admission.ts +29 -0
- package/src/domains/dispatch/agent-candidates.ts +10 -0
- package/src/domains/dispatch/budget-envelope.ts +86 -1
- package/src/domains/dispatch/capability-match.ts +11 -0
- package/src/domains/dispatch/capacity-lease.ts +17 -0
- package/src/domains/dispatch/contract.ts +11 -1
- package/src/domains/dispatch/extension.ts +237 -49
- package/src/domains/dispatch/host-verification.ts +435 -39
- package/src/domains/dispatch/intent-requirements.ts +10 -0
- package/src/domains/dispatch/intent.ts +18 -1
- package/src/domains/dispatch/path-scope.ts +235 -24
- package/src/domains/dispatch/run-event-journal.ts +4 -15
- package/src/domains/dispatch/state.ts +2 -3
- package/src/domains/dispatch/transport.ts +45 -21
- package/src/domains/dispatch/types.ts +58 -3
- package/src/domains/dispatch/worker-model-metadata.ts +38 -0
- package/src/domains/eval/artifacts/store.ts +5 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
- package/src/domains/eval/metrics/token-stream.ts +201 -31
- package/src/domains/eval/metrics/tracked.ts +40 -4
- package/src/domains/eval/runners/clio-run.ts +5 -2
- package/src/domains/eval/schema/suite.ts +28 -0
- package/src/domains/eval/schema/verdict.ts +2 -2
- package/src/domains/eval/store.ts +8 -1
- package/src/domains/eval/suites/resolve.ts +13 -1
- package/src/domains/eval/suites/run.ts +24 -3
- package/src/domains/evidence/trust-status.ts +10 -1
- package/src/domains/extensions/contract.ts +15 -1
- package/src/domains/extensions/discovery.ts +238 -41
- package/src/domains/extensions/extension.ts +105 -6
- package/src/domains/extensions/index.ts +24 -0
- package/src/domains/extensions/integrity.ts +189 -0
- package/src/domains/extensions/manager.ts +17 -1
- package/src/domains/extensions/resource-path.ts +27 -0
- package/src/domains/extensions/resources.ts +18 -38
- package/src/domains/extensions/snapshot-store.ts +39 -0
- package/src/domains/extensions/snapshot.ts +180 -0
- package/src/domains/extensions/state.ts +385 -57
- package/src/domains/extensions/types.ts +118 -1
- package/src/domains/interop/registry.ts +6 -2
- package/src/domains/interop/types.ts +4 -0
- package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
- package/src/domains/lifecycle/migrations/index.ts +6 -0
- package/src/domains/lifecycle/naming-resources.ts +19 -4
- package/src/domains/lifecycle/naming-yazi.ts +10 -5
- package/src/domains/memory/task-memory-policy.ts +70 -26
- package/src/domains/memory/task-memory-telemetry.ts +1 -0
- package/src/domains/middleware/contract.ts +26 -0
- package/src/domains/middleware/extension.ts +24 -24
- package/src/domains/middleware/hook-receipts.ts +27 -4
- package/src/domains/middleware/hooks-io.ts +65 -32
- package/src/domains/middleware/hooks.ts +64 -0
- package/src/domains/middleware/index.ts +28 -5
- package/src/domains/middleware/marketplace-offer.ts +3 -35
- package/src/domains/middleware/memory-intervention.ts +127 -32
- package/src/domains/middleware/memory-step-endpoint.ts +3 -2
- package/src/domains/middleware/registrations.ts +326 -0
- package/src/domains/middleware/runtime.ts +28 -0
- package/src/domains/middleware/skills-reminder.ts +31 -2
- package/src/domains/middleware/snapshot.ts +20 -7
- package/src/domains/mux/contract.ts +38 -0
- package/src/domains/mux/detect.ts +6 -13
- package/src/domains/mux/index.ts +1 -1
- package/src/domains/mux/operations.ts +44 -5
- package/src/domains/mux/yazi/assets/yazi.toml +2 -2
- package/src/domains/mux/yazi/session.ts +53 -4
- package/src/domains/mux/yazi/theme.ts +117 -17
- package/src/domains/observability/compaction-usage.ts +118 -0
- package/src/domains/observability/contract.ts +10 -11
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/extension.ts +17 -4
- package/src/domains/observability/out-of-turn-usage.ts +52 -21
- package/src/domains/observability/projection.ts +14 -90
- package/src/domains/observability/trace-store.ts +43 -7
- package/src/domains/prompts/compiler.ts +73 -53
- package/src/domains/prompts/contract.ts +15 -3
- package/src/domains/prompts/extension.ts +97 -9
- package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
- package/src/domains/prompts/fragments/identity/clio.md +6 -12
- package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
- package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
- package/src/domains/prompts/fragments/operating/contract.md +7 -15
- package/src/domains/prompts/fragments/operating/delegation.md +32 -34
- package/src/domains/prompts/fragments/operating/skills.md +10 -24
- package/src/domains/prompts/fragments/operating/worker.md +1 -8
- package/src/domains/providers/contract.ts +4 -1
- package/src/domains/providers/extension.ts +40 -9
- package/src/domains/providers/index.ts +1 -1
- package/src/domains/providers/model-capabilities.ts +9 -0
- package/src/domains/providers/model-discovery.ts +2 -0
- package/src/domains/providers/model-runtime-capabilities.ts +99 -25
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
- package/src/domains/providers/runtime-resolution.ts +31 -0
- package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
- package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
- package/src/domains/providers/support.ts +11 -5
- package/src/domains/providers/target-model-cache.ts +25 -2
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/cost-provenance.ts +19 -0
- package/src/domains/providers/types/local-model-quirks.ts +85 -37
- package/src/domains/providers/types/runtime-descriptor.ts +20 -1
- package/src/domains/providers/types/target-descriptor.ts +19 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/skills/install.ts +72 -7
- package/src/domains/resources/skills/loader.ts +23 -19
- package/src/domains/resources/skills/marketplace.ts +63 -11
- package/src/domains/safety/autonomy.ts +15 -0
- package/src/domains/safety/call-target.ts +1 -1
- package/src/domains/safety/index.ts +1 -0
- package/src/domains/safety/loop-detector.ts +7 -4
- package/src/domains/safety/path-policy.ts +1 -1
- package/src/domains/safety/policy-engine.ts +34 -11
- package/src/domains/safety/protected-artifacts.ts +191 -88
- package/src/domains/safety/run-effects.ts +2 -22
- package/src/domains/safety/skill-authority.ts +55 -0
- package/src/domains/session/compaction/compact.ts +72 -22
- package/src/domains/session/entries.ts +6 -0
- package/src/domains/session/task-board.ts +10 -9
- package/src/domains/session/usage.ts +3 -3
- package/src/domains/share/archive.ts +164 -7
- package/src/engine/acp/server.ts +62 -9
- package/src/engine/agent.ts +13 -3
- package/src/engine/ai.ts +26 -8
- package/src/engine/antigravity/subprocess-runtime.ts +386 -120
- package/src/engine/api-registry.ts +3 -0
- package/src/engine/apis/llamacpp-residency.ts +3 -4
- package/src/engine/apis/lmstudio.ts +3 -3
- package/src/engine/apis/ollama-native.ts +6 -6
- package/src/engine/apis/openai-completions.ts +145 -39
- package/src/engine/apis/output-budget.ts +8 -18
- package/src/engine/apis/residency.ts +8 -27
- package/src/engine/external-subprocess.ts +114 -6
- package/src/engine/gemma-channel-filter.ts +19 -0
- package/src/engine/loop-guard.ts +92 -12
- package/src/engine/worker-runtime.ts +40 -11
- package/src/engine/worker-tools.ts +3 -1
- package/src/entry/background-model-metadata.ts +18 -0
- package/src/entry/compaction-prompt.ts +57 -0
- package/src/entry/extension-hook-sources.ts +28 -0
- package/src/entry/extension-reload.ts +309 -0
- package/src/entry/orchestrator.ts +464 -251
- package/src/entry/task-memory-lifecycle.ts +35 -0
- package/src/interactive/application-controller.ts +2 -1
- package/src/interactive/bus-notices.ts +8 -1
- package/src/interactive/chat-loop-messages.ts +16 -17
- package/src/interactive/chat-loop.ts +75 -3
- package/src/interactive/chat-panel.ts +36 -13
- package/src/interactive/chat-renderer.ts +72 -7
- package/src/interactive/cost-overlay.ts +26 -2
- package/src/interactive/dispatch-board.ts +6 -11
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +39 -4
- package/src/interactive/interactive-input-runtime.ts +4 -0
- package/src/interactive/interactive-presentation.ts +2 -2
- package/src/interactive/interactive-slash-runtime.ts +4 -1
- package/src/interactive/overlays/extensions.ts +9 -1
- package/src/interactive/overlays/help-reference.ts +13 -0
- package/src/interactive/overlays/settings.ts +27 -16
- package/src/interactive/panes-runtime.ts +111 -35
- package/src/interactive/prompt-cache-identity.ts +88 -0
- package/src/interactive/renderers/worker-entry.ts +32 -0
- package/src/interactive/slash-commands.ts +153 -20
- package/src/interactive/stream-pacing-policy.ts +0 -23
- package/src/interactive/theme/labels.ts +19 -13
- package/src/interactive/turn-context.ts +39 -20
- package/src/interactive/turn-recovery.ts +8 -0
- package/src/interactive/turn-runtime.ts +27 -11
- package/src/interactive/turn-state.ts +7 -0
- package/src/interactive/worker-receipts.ts +1 -0
- package/src/interactive/worker-stream.ts +6 -1
- package/src/interactive/yazi-bridge.ts +60 -6
- package/src/tools/agent-tools.ts +30 -1
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +3 -3
- package/src/tools/bash.ts +1 -1
- package/src/tools/bootstrap.ts +4 -0
- package/src/tools/builtin-tool-catalog.ts +52 -22
- package/src/tools/codewiki/code-nav-surface.ts +6 -0
- package/src/tools/codewiki/code-nav.ts +99 -13
- package/src/tools/context/docs-engine.ts +20 -7
- package/src/tools/context/index.ts +59 -21
- package/src/tools/core-bootstrap.ts +28 -6
- package/src/tools/credential-present.ts +1 -2
- package/src/tools/dispatch-arguments.ts +6 -1
- package/src/tools/dispatch-event-text.ts +10 -0
- package/src/tools/dispatch-plan.ts +49 -4
- package/src/tools/dispatch-run-events.ts +1 -1
- package/src/tools/dispatch-runner.ts +12 -0
- package/src/tools/dispatch-schema.ts +338 -0
- package/src/tools/dispatch-types.ts +3 -0
- package/src/tools/dispatch.ts +9 -254
- package/src/tools/ledger.ts +3 -5
- package/src/tools/monitor-surface.ts +5 -13
- package/src/tools/observation.ts +4 -5
- package/src/tools/panes-surface.ts +4 -11
- package/src/tools/panes.ts +4 -2
- package/src/tools/policy.ts +15 -2
- package/src/tools/read.ts +5 -6
- package/src/tools/registry.ts +41 -12
- package/src/tools/result-shaping.ts +18 -14
- package/src/tools/steer-surface.ts +1 -1
- package/src/tools/tasks.ts +1 -1
- package/src/tools/truncate.ts +6 -5
- package/src/tools/verify/surface.ts +6 -12
- package/src/tools/web-fetch-surface.ts +1 -3
- package/src/tools/worker-evidence.ts +3 -1
- package/src/worker/spec-contract.ts +4 -0
- package/dist/builtins-UJLMOVOV.js +0 -17
- package/dist/chunk-5QIAJV2D.js +0 -48
- package/dist/chunk-JZWT5J3Y.js +0 -814
- package/dist/chunk-K7VKOLQQ.js +0 -15
- package/dist/chunk-PMZCIOCJ.js +0 -25
- package/dist/chunk-SUW5DORT.js +0 -819
- package/dist/chunk-UOV2BYIW.js +0 -107
- package/dist/chunk-WR6U3OVP.js +0 -45
- package/dist/chunk-Y45G3AXC.js +0 -1558
- package/dist/reset-EOLM7GVE.js +0 -230
- package/dist/uninstall-N34PCTGJ.js +0 -331
- package/dist/upgrade-H7TOM7YL.js +0 -323
- package/docs/artifact-versions.md +0 -67
- package/docs/development-pipeline.md +0 -121
- package/docs/documentation-coverage.md +0 -46
- package/docs/documentation-guide.md +0 -167
- package/docs/time-conventions.md +0 -101
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: arxiv-literature
|
|
3
|
-
description:
|
|
3
|
+
description: Searches arXiv, summarizes or compares papers, and builds a compact literature survey as citation-ready, source-linked paper cards. Prefer the Researcher shadow agent for noisy multi-paper retrieval.
|
|
4
4
|
triggers:
|
|
5
5
|
- search arXiv
|
|
6
6
|
- summarize an arXiv paper
|
|
7
7
|
- compare these papers
|
|
8
8
|
- find recent research papers
|
|
9
9
|
- build a literature survey
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.5.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
allowed-tools:
|
|
13
13
|
- web_fetch
|
|
@@ -16,7 +16,6 @@ allowed-tools:
|
|
|
16
16
|
- grep
|
|
17
17
|
- find
|
|
18
18
|
- ls
|
|
19
|
-
- artifact
|
|
20
19
|
clio-coder:
|
|
21
20
|
registry-id: iowarp/clio-coder
|
|
22
21
|
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/arxiv-literature
|
|
@@ -34,6 +33,20 @@ Find, summarize, or compare academic papers without flooding the main context
|
|
|
34
33
|
window. Raw search results and paper text stay in a worker or get compressed
|
|
35
34
|
immediately; only paper cards reach the user.
|
|
36
35
|
|
|
36
|
+
## Arguments
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
/skill arxiv-literature <request>
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
The request is free text: a paper URL/ID, a topic, or two or more IDs to
|
|
43
|
+
compare. There is no operator in a headless run — `ask_user` is not in this
|
|
44
|
+
skill's tool surface and nothing answers it. If the request is ambiguous (no
|
|
45
|
+
clear topic, an ID that doesn't resolve), state your best reading and
|
|
46
|
+
proceed; never stall waiting for clarification. `tasks` sits outside this
|
|
47
|
+
skill's tool surface and is refused; the steps below are the whole plan, do
|
|
48
|
+
not open a task list for them.
|
|
49
|
+
|
|
37
50
|
## Step 1 — Classify the request
|
|
38
51
|
|
|
39
52
|
Pick exactly one:
|
|
@@ -45,29 +58,49 @@ Pick exactly one:
|
|
|
45
58
|
|
|
46
59
|
## Step 2 — Pick the vehicle
|
|
47
60
|
|
|
48
|
-
|
|
49
|
-
|
|
61
|
+
Default every request — single paper, search, compare, and survey alike — to
|
|
62
|
+
`web_fetch` directly against arXiv:
|
|
50
63
|
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
```
|
|
56
|
-
|
|
57
|
-
- **Single paper, or dispatch unavailable**: use `web_fetch` directly.
|
|
58
|
-
- Paper URL/ID: fetch the arXiv page. Clio normalizes it into structured
|
|
59
|
-
metadata plus AlphaXiv enrichment when available.
|
|
60
|
-
- Search query: fetch the arXiv Atom API; Clio compacts the XML into paper
|
|
61
|
-
cards:
|
|
64
|
+
- Paper URL/ID: fetch the arXiv page. Clio normalizes it into structured
|
|
65
|
+
metadata plus AlphaXiv enrichment when available.
|
|
66
|
+
- Topic, compare, or survey: fetch the arXiv Atom API; Clio compacts the XML
|
|
67
|
+
into paper cards:
|
|
62
68
|
|
|
63
69
|
```text
|
|
64
70
|
https://export.arxiv.org/api/query?search_query=all:QUERY&sortBy=submittedDate&sortOrder=descending&start=0&max_results=10
|
|
65
71
|
```
|
|
66
72
|
|
|
73
|
+
For a compare request, run one query per paper ID (`id_list=ID` instead of
|
|
74
|
+
`search_query`) or one broader query covering all of them — whichever stays
|
|
75
|
+
inside the fetch cap in Step 3.
|
|
76
|
+
|
|
67
77
|
Useful categories for `search_query`: `cs.AI` (AI), `cs.LG` (ML), `cs.CL`
|
|
68
78
|
(NLP/LLMs), `cs.CR` (security), `cs.SE` (software engineering), `cs.MA`
|
|
69
79
|
(multi-agent), `cs.IR` (retrieval/RAG), `cs.CV` (vision), `cs.RO` (robotics).
|
|
70
80
|
|
|
81
|
+
**Only dispatch the `researcher` shadow agent when the user explicitly asks
|
|
82
|
+
for a deep or broad survey** ("survey the field", "don't just skim arXiv, go
|
|
83
|
+
wide") — never as the default for an ordinary search or compare. Left
|
|
84
|
+
unbounded, a dispatched worker has no arXiv-only restriction and no fetch
|
|
85
|
+
budget of its own, and will wander into Semantic Scholar, DBLP, OpenAlex, and
|
|
86
|
+
general web search, taking several minutes to return nothing useful. When you
|
|
87
|
+
do dispatch, state the same bound this skill uses directly, in the task
|
|
88
|
+
prompt itself:
|
|
89
|
+
|
|
90
|
+
```text
|
|
91
|
+
Research arXiv literature for: <user goal>.
|
|
92
|
+
Search arXiv only (export.arxiv.org Atom API or arxiv.org paper pages) — do
|
|
93
|
+
not query Semantic Scholar, DBLP, OpenAlex, or general web search. One
|
|
94
|
+
round, at most two fetch attempts total (successes and failures both count).
|
|
95
|
+
On a timeout or HTTP error, do not retry with a different host, scheme, or
|
|
96
|
+
protocol — retry the identical URL at most once, then stop and report the
|
|
97
|
+
failure. Build cards from whatever you have; do not keep escalating.
|
|
98
|
+
Return only compact source-linked paper cards, comparison/synthesis,
|
|
99
|
+
caveats, and read/skim/skip recommendations.
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
If dispatch is unavailable, fall back to the direct `web_fetch` path above.
|
|
103
|
+
|
|
71
104
|
## Step 3 — Return paper cards only
|
|
72
105
|
|
|
73
106
|
Never paste raw Atom XML or full paper text into the response. Output format:
|
|
@@ -96,9 +129,14 @@ Never paste raw Atom XML or full paper text into the response. Output format:
|
|
|
96
129
|
Done when every returned paper has a card with a working link and the
|
|
97
130
|
recommendation section is filled in. Stop after one search round unless the
|
|
98
131
|
user asks to go deeper; do not keep fetching to "be thorough". One round
|
|
99
|
-
means at most two Atom API
|
|
100
|
-
|
|
101
|
-
|
|
132
|
+
means at most two Atom API fetch attempts total: the initial query plus one
|
|
133
|
+
refinement, or one retry of a failed fetch. Failed attempts (timeout, 429,
|
|
134
|
+
connection error) count against this cap the same as successful ones — a
|
|
135
|
+
timeout is not a free retry. Rewording the same query a third time, or
|
|
136
|
+
retrying through a different host/scheme (`http` vs `https`,
|
|
137
|
+
`export.arxiv.org` vs `arxiv.org/search`, an unofficial JSON mirror) after a
|
|
138
|
+
failure, is thrash: build cards from whatever the attempts returned, or
|
|
139
|
+
report the network failure plainly and stop.
|
|
102
140
|
|
|
103
141
|
## Gotchas
|
|
104
142
|
|
|
@@ -108,3 +146,23 @@ first two returned.
|
|
|
108
146
|
- AlphaXiv is AI-generated enrichment: useful for scanning, never citable as
|
|
109
147
|
authoritative.
|
|
110
148
|
- Fetch or enrich only the top candidates, not every result.
|
|
149
|
+
- `export.arxiv.org` rate-limits (HTTP 429) under repeated hits; that is a
|
|
150
|
+
reason to stop at the fetch cap, not a reason to retry against a mirror.
|
|
151
|
+
- This skill's tool surface has no file-writing tool and no `artifact`. The
|
|
152
|
+
paper cards are the chat reply, never a document; do not go looking for a
|
|
153
|
+
way to save one.
|
|
154
|
+
|
|
155
|
+
## Red flags
|
|
156
|
+
|
|
157
|
+
- Dispatching `researcher` for an ordinary search or compare "just in case"
|
|
158
|
+
it does a better job than a direct fetch — it is slower and unbounded by
|
|
159
|
+
default; reserve it for an explicit deep-survey ask.
|
|
160
|
+
- A dispatch task prompt without the arXiv-only, fetch-capped instruction —
|
|
161
|
+
that omission is what let a prior run wander into Semantic Scholar, DBLP,
|
|
162
|
+
and OpenAlex for minutes with nothing to show for it.
|
|
163
|
+
- Treating a timeout or 429 as free of the fetch cap and retrying against a
|
|
164
|
+
different host, scheme, or unofficial mirror instead of stopping.
|
|
165
|
+
- Raw Atom XML, a full abstract dump, or more than the top few candidates
|
|
166
|
+
reaching the final reply.
|
|
167
|
+
- Reaching for `artifact` or any write tool to "save" the result — it is not
|
|
168
|
+
in this skill's tool surface; the reply is the deliverable.
|
|
@@ -56,3 +56,53 @@ Expected:
|
|
|
56
56
|
|
|
57
57
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
58
58
|
(30B local, llamacpp on mini), full-auto sandbox. PASS (driver FAIL overturned on transcript review): skill loaded, skill_surface blocked two curl attempts, retrieval ran through web_fetch on the Atom API per the designed dispatch-unavailable fallback. Judge misread the OR in bullet 1. Query thrash (8 fetches) led to the two-fetch cap now in the body.
|
|
59
|
+
|
|
60
|
+
## Battletest record (2026-09-03)
|
|
61
|
+
|
|
62
|
+
S1 (topic search, speculative decoding) and S2 (single known paper,
|
|
63
|
+
`arxiv.org/abs/1706.03762`), `ornith1.5-35b-moe` on mini (llamacpp), `clio-coder
|
|
64
|
+
run --autonomy full-auto --json`, headless, real network access (no fixture).
|
|
65
|
+
|
|
66
|
+
| run | model | wall | turns | in / out tokens | safety blocks | outcome |
|
|
67
|
+
|---|---|---|---|---|---|---|
|
|
68
|
+
| baseline S1 (no skill) | ornith1.5-35b-moe | 286s (killed at timeout) | 9 | n/a (killed before `agent_end`) | 5 | never finished: fired raw `curl` via `bash` (permission-gated, refused) before falling back to `web_fetch`; hit real timeouts against `export.arxiv.org`; killed by the harness's 300s cap mid-turn with no cards produced |
|
|
69
|
+
| baseline S2 (no skill) | ornith1.5-35b-moe | 24s | 2 | 3.6k / 1.0k | 0 | fetched the real paper directly and wrote a good prose summary, but never labeled problem/method/evidence/limitation as distinct fields — the un-skilled gap S2 expects |
|
|
70
|
+
| v1 S1 (frozen v0.4.0, dispatch runaway) | ornith1.5-35b-moe | 286s (killed at timeout) | 6 | n/a (killed before `agent_end`) | 2 | opened a `tasks` plan (refused, outside surface), dispatched `researcher` twice (first dispatch flagged an absolute-path token in the briefing), then started re-fetching directly itself; killed by the harness's 300s cap before producing cards — the dispatch-runaway/no-cap bug reproduced live |
|
|
71
|
+
| v2 S1 (hardened v0.5.0) | ornith1.5-35b-moe | 109s | 4 | 2.9k / 3.9k | 2 (real `export.arxiv.org` 429 + timeout) | stayed on `web_fetch` only (no dispatch — topic search didn't ask for a "deep survey"), made exactly two fetch attempts against the identical URL per the tightened cap, hit a genuine rate limit then a timeout, **stopped at the cap**, explicitly refused to fabricate paper cards from invented IDs, and returned an honestly-labeled "established knowledge, not a fresh fetch" orientation instead — terminated cleanly on its own, no runaway |
|
|
72
|
+
| v2 S2 (hardened v0.5.0) | ornith1.5-35b-moe | 28s | 3 | 6.2k / 1.5k | 0 | direct `web_fetch` on the real paper page, full problem/method/evidence/limitation/relevance card, explicit read/skim/skip recommendation, AlphaXiv linked and labeled as enrichment, no `artifact` call — 5/5 on the S2 rubric |
|
|
73
|
+
|
|
74
|
+
Changes: removed `artifact` from `allowed-tools` — nothing in the procedure
|
|
75
|
+
ever called it, and it is a terminal tool that would end the run the moment
|
|
76
|
+
the model reached for it to "save" a result. Reworked Step 2 so `web_fetch`
|
|
77
|
+
directly against arXiv is the default vehicle for every request class
|
|
78
|
+
(single paper, search, compare, survey), and `dispatch` is reserved for an
|
|
79
|
+
explicitly requested deep/broad survey — the live probe that motivated this
|
|
80
|
+
pass showed a dispatched `researcher` wandering into Semantic Scholar, DBLP,
|
|
81
|
+
and OpenAlex with no arXiv-only restriction and no fetch budget, never
|
|
82
|
+
returning. When dispatch is used, its task-prompt template now states the
|
|
83
|
+
same arXiv-only, fetch-capped discipline the direct path uses, verbatim.
|
|
84
|
+
Tightened the Step 3 fetch cap to count failed attempts (timeout, 429,
|
|
85
|
+
connection error) against the same two-attempt budget as successes, and
|
|
86
|
+
banned escalating to a different host/scheme/mirror on failure — this closed
|
|
87
|
+
a real gap the v2 S1 run against a genuinely rate-limited `export.arxiv.org`
|
|
88
|
+
would otherwise have exploited (the same tightened wording is what let it
|
|
89
|
+
stop cleanly at 109s instead of retrying indefinitely). Added `## Arguments`
|
|
90
|
+
(free-text request, no operator headlessly, `tasks` refused) and a `##
|
|
91
|
+
Red flags` section naming the dispatch-runaway, uncapped-retry, and
|
|
92
|
+
artifact-reach failure modes actually observed.
|
|
93
|
+
|
|
94
|
+
Still weak: S3 (three-paper comparison) was not run against a real fixture
|
|
95
|
+
this pass — budget went to confirming the dispatch-runaway fix and the
|
|
96
|
+
fetch-cap fix on S1/S2, both of which reproduced live. The compare path's
|
|
97
|
+
"one query per ID or one broader query" guidance in Step 2 is new and
|
|
98
|
+
untested end-to-end. Both baseline and v1 runs for S1 were killed by the
|
|
99
|
+
harness's outer timeout rather than allowed to run to their own natural
|
|
100
|
+
(bad) conclusion — a longer timeout might show the old skill eventually
|
|
101
|
+
recovering, or might show it running further off scope; the fix (bounding
|
|
102
|
+
the fetch cap and defaulting off dispatch) is validated by the v2 behavior,
|
|
103
|
+
not by watching v1 fail for longer. `export.arxiv.org` rate-limited several
|
|
104
|
+
runs in this session from repeated hits in short succession; the 429s in the
|
|
105
|
+
v2 S1 run are a real external condition this pass ran into, not a fixture
|
|
106
|
+
simulation, but a quieter network day could produce a fully-populated
|
|
107
|
+
card set on the same prompt instead of the honest-failure path exercised
|
|
108
|
+
here — both are now handled, but only the failure path got a live rep.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: experiment-protocol
|
|
3
|
-
description:
|
|
3
|
+
description: "Pre-registers a performance study, numerical comparison, parameter sweep, or benchmark: thresholds, tolerances, environment pins, and verdict conditions locked into the validation contract before any measurement. Not for diagnosing a stalled bug; use scientific-debugging."
|
|
4
4
|
triggers:
|
|
5
5
|
- benchmark these implementations
|
|
6
6
|
- pre-register a performance experiment
|
|
@@ -8,7 +8,7 @@ triggers:
|
|
|
8
8
|
- define numerical tolerances
|
|
9
9
|
- reproduce these results
|
|
10
10
|
- compare solver accuracy
|
|
11
|
-
version: 0.
|
|
11
|
+
version: 0.3.0
|
|
12
12
|
license: Apache-2.0
|
|
13
13
|
allowed-tools:
|
|
14
14
|
- read
|
|
@@ -39,6 +39,25 @@ mode this protocol exists to prevent.
|
|
|
39
39
|
Anti-trigger: if the question is "why is this output wrong", that is a
|
|
40
40
|
diagnosis, not an experiment; use scientific-debugging.
|
|
41
41
|
|
|
42
|
+
## Arguments
|
|
43
|
+
|
|
44
|
+
```text
|
|
45
|
+
/skill experiment-protocol <what to benchmark, compare, or sweep>
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
There is no operator in a headless run — `ask_user` is not in this skill's
|
|
49
|
+
tool surface. If a threshold, tolerance, or environment detail is unstated,
|
|
50
|
+
pick the most defensible default, record it as an explicit assumption in the
|
|
51
|
+
pre-registration, and proceed; never stall Phase 0 waiting for confirmation.
|
|
52
|
+
|
|
53
|
+
The phases below are the plan; do not open a task list for them — `tasks`
|
|
54
|
+
sits outside this skill's tool surface and any call to it is refused.
|
|
55
|
+
|
|
56
|
+
Shell rules for every `bash` call: one command per call, plain and direct.
|
|
57
|
+
Never use `$(...)` or backticks; they trigger an approval gate that ends a
|
|
58
|
+
headless run. Capture checksums and environment facts with direct calls
|
|
59
|
+
(`sha256sum data/mesh.h5`), never command substitution.
|
|
60
|
+
|
|
42
61
|
## Phase 0 - Pre-register
|
|
43
62
|
|
|
44
63
|
Before any measurement, write the protocol into the repository validation
|
|
@@ -89,3 +89,26 @@ Prompt: "Make this kernel faster."
|
|
|
89
89
|
|
|
90
90
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
91
91
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Pre-registration written before touching the seeded kernel; judge 5/5.
|
|
92
|
+
|
|
93
|
+
## Battletest record (2026-09-03)
|
|
94
|
+
|
|
95
|
+
Real `clio-coder run --json` against dynamo (LM Studio, `qwen3.8-27b`), the S1
|
|
96
|
+
kernel.py fixture above, git repo under
|
|
97
|
+
`/home/akougkas/eval-temp/expprotocol-fixture/`. Same gap as every other
|
|
98
|
+
research skill going in: no `## Arguments`, no `tasks`-refusal, no shell-rules
|
|
99
|
+
paragraph, no no-operator statement. Added all four.
|
|
100
|
+
|
|
101
|
+
| run | model | outcome |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| v0.3.0 (hardened) | qwen3.8-27b | environment capture (`python3 --version`, `uname -a`, CPU model), `numpy` version check, `sha256sum` on the input array and the frozen baseline copy, `.clio-coder/validation.yaml` written with thresholds/tolerance semantics before any benchmark, a real 20-rep baseline vs. a vectorized candidate, a bit-exact accuracy check between them, then a correctness spot-check on a second slice-based candidate before timing it — 34 tool calls, zero safety blocks, zero `$(...)`, zero `tasks`, `write`/`edit` used correctly (in this skill's surface, unlike scientific-debugging) instead of a heredoc |
|
|
104
|
+
|
|
105
|
+
The run did not reach a final Phase 3 verdict/report inside the 280s box used
|
|
106
|
+
this pass (31 API calls, ~750k cumulative input tokens — the box closes on
|
|
107
|
+
context-reprocessing volume, not model slowness); everything observed up to
|
|
108
|
+
that point followed Phase 0-2 exactly as specified, including registering a
|
|
109
|
+
100x stretch target, measuring ~13x, and correctly continuing to iterate
|
|
110
|
+
rather than declaring victory early.
|
|
111
|
+
|
|
112
|
+
**Still weak**: no observed run reaching a written Phase 3 verdict this pass
|
|
113
|
+
(same time-box cause as scientific-debugging's record above). No cross-model
|
|
114
|
+
confirmation this pass.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: scientific-debugging
|
|
3
|
-
description:
|
|
3
|
+
description: Diagnoses stalled or cross-system failures and wrong, NaN, nondeterministic, or unexpectedly slow scientific results through falsifiable hypotheses across distinct fault classes, with evidence-cited verdicts before any fix. Not for designing benchmarks or pre-registered experiments; use experiment-protocol.
|
|
4
4
|
triggers:
|
|
5
5
|
- wrong scientific results
|
|
6
6
|
- nondeterministic HPC code
|
|
@@ -8,7 +8,7 @@ triggers:
|
|
|
8
8
|
- diagnose NaNs
|
|
9
9
|
- debug with falsifiable hypotheses
|
|
10
10
|
- scientific root cause
|
|
11
|
-
version: 0.
|
|
11
|
+
version: 0.3.0
|
|
12
12
|
license: Apache-2.0
|
|
13
13
|
allowed-tools:
|
|
14
14
|
- read
|
|
@@ -38,6 +38,28 @@ Anti-trigger: if the failure is a typo, a missing import, or an error message
|
|
|
38
38
|
that names its own cause, fix it directly and skip this workflow. The loop
|
|
39
39
|
below is for failures that survived the first obvious fix.
|
|
40
40
|
|
|
41
|
+
## Arguments
|
|
42
|
+
|
|
43
|
+
```text
|
|
44
|
+
/skill scientific-debugging <failure description>
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Everything after the skill name is the failure report: the observed wrong
|
|
48
|
+
behavior and whatever has already been tried. There is no operator in a
|
|
49
|
+
headless run — `ask_user` is not in this skill's tool surface. If the goal,
|
|
50
|
+
a fault-class split, or a ranking call is ambiguous, state your best reading
|
|
51
|
+
in Step 1 or Step 3 and proceed; never stall a step waiting for confirmation.
|
|
52
|
+
|
|
53
|
+
The Loop below is the plan; do not open a task list for it — `tasks` sits
|
|
54
|
+
outside this skill's tool surface and any call to it is refused.
|
|
55
|
+
|
|
56
|
+
Shell rules for every `bash` call: one command per call, plain and direct.
|
|
57
|
+
Never use `$(...)` or backticks; they trigger an approval gate that ends a
|
|
58
|
+
headless run. This skill has no `write`/`edit` tool — the structured-
|
|
59
|
+
investigation file in the Tiers section below is written with a `bash`
|
|
60
|
+
heredoc (`cat > file <<'EOF' ... EOF`), never through an edit tool that
|
|
61
|
+
isn't in this skill's surface.
|
|
62
|
+
|
|
41
63
|
## The Loop
|
|
42
64
|
|
|
43
65
|
1. **Goal.** One sentence stating the observable "fixed" state. "The regression
|
|
@@ -136,3 +136,21 @@ accumulation loop; regression check fails by 1.474e-4).
|
|
|
136
136
|
|
|
137
137
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
138
138
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Judge 6/6 on the seeded numerical fixture; cleanest research run.
|
|
139
|
+
|
|
140
|
+
## Battletest record (2026-09-03)
|
|
141
|
+
|
|
142
|
+
Real `clio-coder run --json` against dynamo (LM Studio, `qwen3.8-27b`), the S1
|
|
143
|
+
diffusion fixture above (order-sensitive summation regression, git repo under
|
|
144
|
+
`/home/akougkas/eval-temp/scidebug-fixture/`). No `## Arguments` section,
|
|
145
|
+
`tasks`-refusal, shell-rules paragraph, or no-operator statement existed in
|
|
146
|
+
the skill body going in — same gap every other hardened category this session
|
|
147
|
+
found. Added all four, matching skills/coding/prototype/SKILL.md's pattern,
|
|
148
|
+
and made explicit that the structured-investigation file goes through a
|
|
149
|
+
`bash` heredoc because this skill has no `write`/`edit` tool.
|
|
150
|
+
|
|
151
|
+
| run | model | outcome |
|
|
152
|
+
|---|---|---|
|
|
153
|
+
| baseline (no skill) | qwen3.8-27b | ran 9 API calls / ~180k tokens without converging inside a 200s box; genuinely still reasoning, not stalled |
|
|
154
|
+
| v0.3.0 (hardened) | qwen3.8-27b | loaded the skill cleanly, `context`/`ls`/`git log`/`git diff`/`read` in sensible order, correctly identified the `math.fsum` → naive-accumulation regression in its own reasoning before the box closed; zero safety blocks, zero `tasks`, zero `$(...)` |
|
|
155
|
+
|
|
156
|
+
**Still weak**: this session's harness runs are token-heavy (each tool round-trip reprocesses the full growing context) and neither the baseline nor the hardened run reached a written goal/hypotheses/verdict block inside the time box used this pass — the trajectory is correct and clean, but full-loop completion on this model under this box is unconfirmed, only strongly suggested. No cross-model confirmation this pass (time-boxed session). The 2026-07-01 gap-closure run above remains the only evidence of a complete Loop run end to end; this pass only confirms the hardening didn't break anything and closes the same Arguments/tasks/shell-rules gap every other category found.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: scientific-modernization
|
|
3
|
-
description:
|
|
3
|
+
description: Modernizes, ports, rewrites, packages, or replaces established scientific software in independently validated stages against an external scientific oracle, settling compatibility and upstream stewardship before calling it complete. Not for an isolated benchmark; use experiment-protocol. Not for diagnosing wrong results; use scientific-debugging.
|
|
4
4
|
triggers:
|
|
5
5
|
- modernize this scientific code
|
|
6
6
|
- rewrite this scientific software in Rust
|
|
@@ -8,7 +8,7 @@ triggers:
|
|
|
8
8
|
- migrate the scientific build system
|
|
9
9
|
- create a maintained fork
|
|
10
10
|
- preserve scientific parity
|
|
11
|
-
version: 0.
|
|
11
|
+
version: 0.4.0
|
|
12
12
|
license: Apache-2.0
|
|
13
13
|
allowed-tools:
|
|
14
14
|
- read
|
|
@@ -36,6 +36,31 @@ translation. Faster code, a clean build, and passing self-authored unit tests
|
|
|
36
36
|
do not establish scientific equivalence. Work the stages below in order; each
|
|
37
37
|
has an explicit exit condition.
|
|
38
38
|
|
|
39
|
+
## Arguments
|
|
40
|
+
|
|
41
|
+
```text
|
|
42
|
+
/skill scientific-modernization <what to modernize, port, rewrite, or replace, and why>
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
There is no operator in a headless run — `ask_user` is not in this skill's
|
|
46
|
+
tool surface. Stage 1's "the user has seen them" exit condition means,
|
|
47
|
+
headlessly: state the four bullets in your reply and proceed, never stall
|
|
48
|
+
waiting for acknowledgment. Every other stage gate below works the same way
|
|
49
|
+
— state the decision and its reasoning, then continue.
|
|
50
|
+
|
|
51
|
+
The six stages are the plan; do not open a task list for them — `tasks` sits
|
|
52
|
+
outside this skill's tool surface and any call to it is refused.
|
|
53
|
+
|
|
54
|
+
Shell rules for every `bash` call: one command per call, plain and direct.
|
|
55
|
+
Never use `$(...)` or backticks; they trigger an approval gate that ends a
|
|
56
|
+
headless run.
|
|
57
|
+
|
|
58
|
+
This is a long, multi-stage process on a small model or a time-boxed run. If
|
|
59
|
+
you are approaching your tool-call or time budget before reaching Stage 6,
|
|
60
|
+
stop at the current stage, state exactly which stage you reached and why you
|
|
61
|
+
stopped, and report the work as an incomplete prototype — never fabricate
|
|
62
|
+
completion of stages you did not actually reach.
|
|
63
|
+
|
|
39
64
|
## Stage 1 — Decide whether this work should exist
|
|
40
65
|
|
|
41
66
|
Identify the upstream project: active maintainers, license, release cadence,
|
|
@@ -82,3 +82,30 @@ Expected:
|
|
|
82
82
|
|
|
83
83
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
84
84
|
(30B local, llamacpp on mini), full-auto sandbox. NOT COMPLETED: loop guard at 75 tool calls. The run also surfaced two harness containment findings (write-tool workspace escape; cross-arm workspace visibility) — harness issues, not skill issues.
|
|
85
|
+
|
|
86
|
+
## Battletest record (2026-09-03)
|
|
87
|
+
|
|
88
|
+
Same structural gap as the other three research skills: no `## Arguments`,
|
|
89
|
+
no `tasks`-refusal, no shell-rules paragraph, no no-operator statement. Added
|
|
90
|
+
all four, plus an explicit budget-awareness line (state which stage you
|
|
91
|
+
reached and report the work as an incomplete prototype rather than
|
|
92
|
+
fabricate completion) directly answering the 2026-08-13 record's loop-guard
|
|
93
|
+
finding above.
|
|
94
|
+
|
|
95
|
+
This mission's plan called for a small legacy-C fixture and a live A/B pass
|
|
96
|
+
on `mini`; that track was killed mid-run this session for taking materially
|
|
97
|
+
longer than the wall-clock budget available (this is a six-stage,
|
|
98
|
+
`model-size: large` skill — the prior smoke record already didn't complete
|
|
99
|
+
on a 30B model, and a fresh fixture never got built before the track was
|
|
100
|
+
stopped). **This pass is a static hardening pass only** — the four additions
|
|
101
|
+
above were applied and `npm run skills:pin && npm run skills:check && npm
|
|
102
|
+
run lint` verified green, but no fresh live run against this skill's
|
|
103
|
+
hardened body exists yet. Treat 0.4.0 as unverified beyond the structural
|
|
104
|
+
fix; it needs the small legacy-C fixture (a single-file numeric routine with
|
|
105
|
+
a captured reference output as the oracle) and a real baseline/hardened A/B
|
|
106
|
+
pass before its `eval-status` can move past `scenarios-recorded`.
|
|
107
|
+
|
|
108
|
+
**Still weak**: everything — no fixture built this pass, no run executed
|
|
109
|
+
against 0.4.0, no cross-model confirmation. This is the one skill in the
|
|
110
|
+
category that did not get real battletest evidence this session; say so
|
|
111
|
+
plainly rather than implying otherwise from the version bump.
|