@iowarp/clio-coder 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/CONTRIBUTING.md +142 -52
- package/README.md +434 -473
- package/SECURITY.md +2 -1
- package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
- package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
- package/dist/builtins-IA7V7FUC.js +22 -0
- package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
- package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
- package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
- package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
- package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
- package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
- package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
- package/dist/chunk-3EBYEESD.js +314 -0
- package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
- package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
- package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
- package/dist/chunk-4UVU7BJ5.js +39 -0
- package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
- package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
- package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
- package/dist/chunk-5MEZN6CB.js +1334 -0
- package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
- package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
- package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
- package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
- package/dist/chunk-7DRAWPTZ.js +360 -0
- package/dist/chunk-7E7I3WLS.js +3762 -0
- package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
- package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
- package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
- package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
- package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
- package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
- package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
- package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
- package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
- package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
- package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
- package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
- package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
- package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
- package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
- package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
- package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
- package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
- package/dist/chunk-GYV6VZOC.js +26 -0
- package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
- package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
- package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
- package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
- package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
- package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
- package/dist/chunk-IJNZMHLA.js +101 -0
- package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
- package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
- package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
- package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
- package/dist/chunk-JEQ3XTHC.js +42 -0
- package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
- package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
- package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
- package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
- package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
- package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
- package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
- package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
- package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
- package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
- package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
- package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
- package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
- package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
- package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
- package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
- package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
- package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
- package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
- package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
- package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
- package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
- package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
- package/dist/chunk-R6Q67RJH.js +134 -0
- package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
- package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
- package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
- package/dist/chunk-SKHCAU7K.js +385 -0
- package/dist/chunk-SZAA6XDG.js +30 -0
- package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
- package/dist/chunk-UOIZ7DA4.js +41 -0
- package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
- package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
- package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
- package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
- package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
- package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
- package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
- package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
- package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
- package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
- package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
- package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
- package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
- package/dist/chunk-XRZT5WY5.js +47 -0
- package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
- package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
- package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
- package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
- package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
- package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
- package/dist/cli/index.js +41 -39
- package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
- package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
- package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
- package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
- package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
- package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
- package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
- package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
- package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
- package/dist/context-map-COB37XXN.js +505 -0
- package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
- package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
- package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
- package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
- package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
- package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
- package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
- package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
- package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
- package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
- package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
- package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
- package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
- package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
- package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
- package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
- package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
- package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
- package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
- package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
- package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
- package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
- package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
- package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
- package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
- package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
- package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
- package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
- package/dist/reset-EAJFFJVB.js +344 -0
- package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
- package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
- package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
- package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
- package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
- package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
- package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
- package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
- package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
- package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
- package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
- package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
- package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
- package/dist/uninstall-5PEVOE5B.js +408 -0
- package/dist/upgrade-M4WXY6KN.js +303 -0
- package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
- package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
- package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
- package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
- package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
- package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
- package/dist/worker/entry.js +90 -74
- package/docs/README.md +176 -81
- package/docs/{acp.md → architecture/acp.md} +36 -20
- package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
- package/docs/{architecture.md → architecture/architecture.md} +43 -22
- package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
- package/docs/architecture/artifact-versions.md +90 -0
- package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
- package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
- package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
- package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
- package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
- package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
- package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
- package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
- package/docs/{observability.md → architecture/observability.md} +38 -14
- package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
- package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
- package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
- package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
- package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
- package/docs/architecture/time-conventions.md +125 -0
- package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
- package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
- package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
- package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
- package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
- package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
- package/docs/guide/configuration-reference.md +1163 -0
- package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
- package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
- package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
- package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
- package/docs/{glossary.md → guide/glossary.md} +14 -11
- package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
- package/docs/guide/panes-and-files.md +290 -0
- package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
- package/docs/{resource-library.md → guide/resource-library.md} +13 -4
- package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
- package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
- package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
- package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
- package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
- package/docs/process/development-pipeline.md +152 -0
- package/docs/process/documentation-coverage.md +100 -0
- package/docs/process/documentation-guide.md +187 -0
- package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
- package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
- package/docs/{evolution.md → process/evolution.md} +2 -2
- package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
- package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
- package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
- package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
- package/evals/README.md +2 -2
- package/evals/behavioral-model.yaml +3 -2
- package/package.json +10 -8
- package/skills/README.md +52 -41
- package/skills/coding/ast-grep/SKILL.md +102 -31
- package/skills/coding/ast-grep/evals.md +26 -0
- package/skills/coding/coding-standards/SKILL.md +41 -6
- package/skills/coding/coding-standards/evals.md +23 -0
- package/skills/coding/prototype/SKILL.md +88 -29
- package/skills/coding/prototype/evals.md +19 -0
- package/skills/coding/tdd/SKILL.md +81 -54
- package/skills/coding/tdd/evals.md +20 -0
- package/skills/context/context-handoff/SKILL.md +44 -3
- package/skills/context/context-handoff/evals.md +44 -0
- package/skills/context/context-prime/SKILL.md +46 -16
- package/skills/context/context-prime/evals.md +45 -0
- package/skills/git/branch-closeout/SKILL.md +132 -0
- package/skills/git/branch-closeout/evals.md +133 -0
- package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
- package/skills/git/file-ticket/SKILL.md +78 -64
- package/skills/git/file-ticket/assets/issue-template.md +22 -0
- package/skills/git/file-ticket/evals.md +31 -26
- package/skills/git/file-ticket/references/issue-discovery.md +49 -0
- package/skills/git/fix-issue/SKILL.md +88 -65
- package/skills/git/fix-issue/evals.md +35 -31
- package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
- package/skills/git/resolve-merge-conflicts/evals.md +52 -25
- package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
- package/skills/git/ship/SKILL.md +103 -67
- package/skills/git/ship/assets/pr-template.md +21 -0
- package/skills/git/ship/evals.md +44 -28
- package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
- package/skills/git/worktree-create/SKILL.md +80 -50
- package/skills/git/worktree-create/evals.md +40 -33
- package/skills/git/worktree-create/references/worktree-setup.md +62 -66
- package/skills/git/worktree-merge/SKILL.md +112 -65
- package/skills/git/worktree-merge/evals.md +42 -34
- package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
- package/skills/meta/clio-coder-dev/SKILL.md +9 -5
- package/skills/meta/clio-coder-dev/evals.md +3 -2
- package/skills/meta/clio-coder-test/SKILL.md +102 -95
- package/skills/meta/clio-coder-test/evals.md +9 -4
- package/skills/meta/clio-coder-test/references/harness.md +100 -124
- package/skills/meta/clio-coder-test/references/test-map.md +77 -50
- package/skills/meta/credentials/SKILL.md +2 -2
- package/skills/meta/find-skills/SKILL.md +2 -2
- package/skills/meta/herdr/SKILL.md +2 -2
- package/skills/meta/skill-craft/SKILL.md +22 -16
- package/skills/planning/archify/SKILL.md +196 -0
- package/skills/planning/archify/evals.md +65 -0
- package/skills/planning/architecture/SKILL.md +62 -13
- package/skills/planning/architecture/evals.md +65 -0
- package/skills/planning/backlog/SKILL.md +131 -15
- package/skills/planning/backlog/evals.md +142 -0
- package/skills/planning/prd/SKILL.md +47 -7
- package/skills/planning/prd/evals.md +54 -0
- package/skills/planning/product-intent/SKILL.md +58 -3
- package/skills/planning/product-intent/evals.md +70 -0
- package/skills/planning/tech-spec/SKILL.md +54 -3
- package/skills/planning/tech-spec/evals.md +73 -0
- package/skills/registry.yaml +70 -62
- package/skills/remote.yaml +13 -0
- package/skills/research/arxiv-literature/SKILL.md +77 -19
- package/skills/research/arxiv-literature/evals.md +50 -0
- package/skills/research/experiment-protocol/SKILL.md +21 -2
- package/skills/research/experiment-protocol/evals.md +23 -0
- package/skills/research/scientific-debugging/SKILL.md +24 -2
- package/skills/research/scientific-debugging/evals.md +18 -0
- package/skills/research/scientific-modernization/SKILL.md +27 -2
- package/skills/research/scientific-modernization/evals.md +27 -0
- package/skills/skill-marketplace.json +97 -62
- package/skills/workflow/cut-it/SKILL.md +66 -6
- package/skills/workflow/cut-it/evals.md +101 -0
- package/skills/workflow/design-council/SKILL.md +118 -28
- package/skills/workflow/design-council/evals.md +161 -0
- package/skills/workflow/grill-me/SKILL.md +87 -11
- package/skills/workflow/grill-me/evals.md +153 -0
- package/skills/workflow/workflow-distiller/SKILL.md +77 -18
- package/skills/workflow/workflow-distiller/evals.md +118 -0
- package/src/cli/args.ts +2 -2
- package/src/cli/bootstrap-generate.ts +1 -1
- package/src/cli/config-inspect.ts +65 -12
- package/src/cli/configure-interop.ts +105 -13
- package/src/cli/configure-oauth.ts +57 -0
- package/src/cli/configure-onboarding.ts +980 -0
- package/src/cli/configure-target.ts +594 -0
- package/src/cli/configure.ts +1082 -532
- package/src/cli/context-map.ts +114 -0
- package/src/cli/context.ts +4 -0
- package/src/cli/docs.ts +22 -14
- package/src/cli/doctor-naming.ts +5 -5
- package/src/cli/doctor-toolchain.ts +3 -3
- package/src/cli/eval.ts +1 -2
- package/src/cli/extensions.ts +2 -1
- package/src/cli/fleet.ts +1 -1
- package/src/cli/index.ts +3 -1
- package/src/cli/internal-dispatch.ts +3 -4
- package/src/cli/lifecycle-presenter.ts +436 -0
- package/src/cli/models.ts +10 -2
- package/src/cli/modes/print.ts +5 -1
- package/src/cli/panes.ts +19 -5
- package/src/cli/reset.ts +228 -106
- package/src/cli/run.ts +9 -4
- package/src/cli/select.ts +664 -0
- package/src/cli/share.ts +5 -1
- package/src/cli/skills-eval.ts +3 -3
- package/src/cli/skills.ts +9 -2
- package/src/cli/targets.ts +5 -6
- package/src/cli/trace.ts +55 -4
- package/src/cli/uninstall.ts +233 -165
- package/src/cli/upgrade.ts +204 -149
- package/src/cli/usage.ts +86 -27
- package/src/cli/validate-model.ts +3 -3
- package/src/cli/wiki-generate.ts +1 -1
- package/src/core/artifact-paths.ts +1 -1
- package/src/core/bash-exec.ts +131 -86
- package/src/core/bus-events.ts +51 -6
- package/src/core/config.ts +61 -1
- package/src/core/defaults.ts +7 -4
- package/src/core/dispatch-outcome.ts +16 -0
- package/src/core/external-diagnostic.ts +44 -0
- package/src/core/gateway-routing.ts +157 -0
- package/src/core/guardrails.ts +10 -49
- package/src/core/prompt-hint.ts +9 -0
- package/src/core/safe-exec.ts +17 -2
- package/src/core/skill-activation.ts +89 -2
- package/src/domains/agents/builtins/architect.md +2 -3
- package/src/domains/agents/builtins/coder.md +3 -2
- package/src/domains/agents/builtins/debugger.md +2 -2
- package/src/domains/agents/builtins/documenter.md +2 -2
- package/src/domains/agents/builtins/git-master.md +1 -1
- package/src/domains/agents/builtins/oracle.md +1 -1
- package/src/domains/agents/builtins/provenance.md +1 -1
- package/src/domains/agents/builtins/researcher.md +1 -1
- package/src/domains/agents/builtins/scout.md +1 -1
- package/src/domains/agents/builtins/tester.md +2 -2
- package/src/domains/agents/builtins/verifier.md +2 -2
- package/src/domains/agents/builtins/wiki-writer.md +1 -1
- package/src/domains/agents/builtins/world-knowledge.md +31 -0
- package/src/domains/agents/catalog.ts +13 -15
- package/src/domains/agents/contract.ts +2 -0
- package/src/domains/agents/extension.ts +23 -1
- package/src/domains/agents/result-contract.ts +70 -0
- package/src/domains/config/keybindings.ts +8 -0
- package/src/domains/context/extension.ts +0 -3
- package/src/domains/context/wiki/map-seed.ts +589 -0
- package/src/domains/context/wiki/plan.ts +2 -2
- package/src/domains/context/working-set/path-index.ts +1 -0
- package/src/domains/dispatch/admission.ts +29 -0
- package/src/domains/dispatch/agent-candidates.ts +10 -0
- package/src/domains/dispatch/budget-envelope.ts +86 -1
- package/src/domains/dispatch/capability-match.ts +11 -0
- package/src/domains/dispatch/capacity-lease.ts +17 -0
- package/src/domains/dispatch/contract.ts +11 -1
- package/src/domains/dispatch/extension.ts +237 -49
- package/src/domains/dispatch/host-verification.ts +435 -39
- package/src/domains/dispatch/intent-requirements.ts +10 -0
- package/src/domains/dispatch/intent.ts +18 -1
- package/src/domains/dispatch/path-scope.ts +235 -24
- package/src/domains/dispatch/run-event-journal.ts +4 -15
- package/src/domains/dispatch/state.ts +2 -3
- package/src/domains/dispatch/transport.ts +45 -21
- package/src/domains/dispatch/types.ts +58 -3
- package/src/domains/dispatch/worker-model-metadata.ts +38 -0
- package/src/domains/eval/artifacts/store.ts +5 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
- package/src/domains/eval/metrics/token-stream.ts +201 -31
- package/src/domains/eval/metrics/tracked.ts +40 -4
- package/src/domains/eval/runners/clio-run.ts +5 -2
- package/src/domains/eval/schema/suite.ts +28 -0
- package/src/domains/eval/schema/verdict.ts +2 -2
- package/src/domains/eval/store.ts +8 -1
- package/src/domains/eval/suites/resolve.ts +13 -1
- package/src/domains/eval/suites/run.ts +24 -3
- package/src/domains/evidence/trust-status.ts +10 -1
- package/src/domains/extensions/contract.ts +15 -1
- package/src/domains/extensions/discovery.ts +238 -41
- package/src/domains/extensions/extension.ts +105 -6
- package/src/domains/extensions/index.ts +24 -0
- package/src/domains/extensions/integrity.ts +189 -0
- package/src/domains/extensions/manager.ts +17 -1
- package/src/domains/extensions/resource-path.ts +27 -0
- package/src/domains/extensions/resources.ts +18 -38
- package/src/domains/extensions/snapshot-store.ts +39 -0
- package/src/domains/extensions/snapshot.ts +180 -0
- package/src/domains/extensions/state.ts +385 -57
- package/src/domains/extensions/types.ts +118 -1
- package/src/domains/interop/registry.ts +6 -2
- package/src/domains/interop/types.ts +4 -0
- package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
- package/src/domains/lifecycle/migrations/index.ts +6 -0
- package/src/domains/lifecycle/naming-resources.ts +19 -4
- package/src/domains/lifecycle/naming-yazi.ts +10 -5
- package/src/domains/memory/task-memory-policy.ts +70 -26
- package/src/domains/memory/task-memory-telemetry.ts +1 -0
- package/src/domains/middleware/contract.ts +26 -0
- package/src/domains/middleware/extension.ts +24 -24
- package/src/domains/middleware/hook-receipts.ts +27 -4
- package/src/domains/middleware/hooks-io.ts +65 -32
- package/src/domains/middleware/hooks.ts +64 -0
- package/src/domains/middleware/index.ts +28 -5
- package/src/domains/middleware/marketplace-offer.ts +3 -35
- package/src/domains/middleware/memory-intervention.ts +127 -32
- package/src/domains/middleware/memory-step-endpoint.ts +3 -2
- package/src/domains/middleware/registrations.ts +326 -0
- package/src/domains/middleware/runtime.ts +28 -0
- package/src/domains/middleware/skills-reminder.ts +31 -2
- package/src/domains/middleware/snapshot.ts +20 -7
- package/src/domains/mux/contract.ts +38 -0
- package/src/domains/mux/detect.ts +6 -13
- package/src/domains/mux/index.ts +1 -1
- package/src/domains/mux/operations.ts +44 -5
- package/src/domains/mux/yazi/assets/yazi.toml +2 -2
- package/src/domains/mux/yazi/session.ts +53 -4
- package/src/domains/mux/yazi/theme.ts +117 -17
- package/src/domains/observability/compaction-usage.ts +118 -0
- package/src/domains/observability/contract.ts +10 -11
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/extension.ts +17 -4
- package/src/domains/observability/out-of-turn-usage.ts +52 -21
- package/src/domains/observability/projection.ts +14 -90
- package/src/domains/observability/trace-store.ts +43 -7
- package/src/domains/prompts/compiler.ts +73 -53
- package/src/domains/prompts/contract.ts +15 -3
- package/src/domains/prompts/extension.ts +97 -9
- package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
- package/src/domains/prompts/fragments/identity/clio.md +6 -12
- package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
- package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
- package/src/domains/prompts/fragments/operating/contract.md +7 -15
- package/src/domains/prompts/fragments/operating/delegation.md +32 -34
- package/src/domains/prompts/fragments/operating/skills.md +10 -24
- package/src/domains/prompts/fragments/operating/worker.md +1 -8
- package/src/domains/providers/contract.ts +4 -1
- package/src/domains/providers/extension.ts +40 -9
- package/src/domains/providers/index.ts +1 -1
- package/src/domains/providers/model-capabilities.ts +9 -0
- package/src/domains/providers/model-discovery.ts +2 -0
- package/src/domains/providers/model-runtime-capabilities.ts +99 -25
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
- package/src/domains/providers/runtime-resolution.ts +31 -0
- package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
- package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
- package/src/domains/providers/support.ts +11 -5
- package/src/domains/providers/target-model-cache.ts +25 -2
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/cost-provenance.ts +19 -0
- package/src/domains/providers/types/local-model-quirks.ts +85 -37
- package/src/domains/providers/types/runtime-descriptor.ts +20 -1
- package/src/domains/providers/types/target-descriptor.ts +19 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/skills/install.ts +72 -7
- package/src/domains/resources/skills/loader.ts +23 -19
- package/src/domains/resources/skills/marketplace.ts +63 -11
- package/src/domains/safety/autonomy.ts +15 -0
- package/src/domains/safety/call-target.ts +1 -1
- package/src/domains/safety/index.ts +1 -0
- package/src/domains/safety/loop-detector.ts +7 -4
- package/src/domains/safety/path-policy.ts +1 -1
- package/src/domains/safety/policy-engine.ts +34 -11
- package/src/domains/safety/protected-artifacts.ts +191 -88
- package/src/domains/safety/run-effects.ts +2 -22
- package/src/domains/safety/skill-authority.ts +55 -0
- package/src/domains/session/compaction/compact.ts +72 -22
- package/src/domains/session/entries.ts +6 -0
- package/src/domains/session/task-board.ts +10 -9
- package/src/domains/session/usage.ts +3 -3
- package/src/domains/share/archive.ts +164 -7
- package/src/engine/acp/server.ts +62 -9
- package/src/engine/agent.ts +13 -3
- package/src/engine/ai.ts +26 -8
- package/src/engine/antigravity/subprocess-runtime.ts +386 -120
- package/src/engine/api-registry.ts +3 -0
- package/src/engine/apis/llamacpp-residency.ts +3 -4
- package/src/engine/apis/lmstudio.ts +3 -3
- package/src/engine/apis/ollama-native.ts +6 -6
- package/src/engine/apis/openai-completions.ts +145 -39
- package/src/engine/apis/output-budget.ts +8 -18
- package/src/engine/apis/residency.ts +8 -27
- package/src/engine/external-subprocess.ts +114 -6
- package/src/engine/gemma-channel-filter.ts +19 -0
- package/src/engine/loop-guard.ts +92 -12
- package/src/engine/worker-runtime.ts +40 -11
- package/src/engine/worker-tools.ts +3 -1
- package/src/entry/background-model-metadata.ts +18 -0
- package/src/entry/compaction-prompt.ts +57 -0
- package/src/entry/extension-hook-sources.ts +28 -0
- package/src/entry/extension-reload.ts +309 -0
- package/src/entry/orchestrator.ts +464 -251
- package/src/entry/task-memory-lifecycle.ts +35 -0
- package/src/interactive/application-controller.ts +2 -1
- package/src/interactive/bus-notices.ts +8 -1
- package/src/interactive/chat-loop-messages.ts +16 -17
- package/src/interactive/chat-loop.ts +75 -3
- package/src/interactive/chat-panel.ts +36 -13
- package/src/interactive/chat-renderer.ts +72 -7
- package/src/interactive/cost-overlay.ts +26 -2
- package/src/interactive/dispatch-board.ts +6 -11
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +39 -4
- package/src/interactive/interactive-input-runtime.ts +4 -0
- package/src/interactive/interactive-presentation.ts +2 -2
- package/src/interactive/interactive-slash-runtime.ts +4 -1
- package/src/interactive/overlays/extensions.ts +9 -1
- package/src/interactive/overlays/help-reference.ts +13 -0
- package/src/interactive/overlays/settings.ts +27 -16
- package/src/interactive/panes-runtime.ts +111 -35
- package/src/interactive/prompt-cache-identity.ts +88 -0
- package/src/interactive/renderers/worker-entry.ts +32 -0
- package/src/interactive/slash-commands.ts +153 -20
- package/src/interactive/stream-pacing-policy.ts +0 -23
- package/src/interactive/theme/labels.ts +19 -13
- package/src/interactive/turn-context.ts +39 -20
- package/src/interactive/turn-recovery.ts +8 -0
- package/src/interactive/turn-runtime.ts +27 -11
- package/src/interactive/turn-state.ts +7 -0
- package/src/interactive/worker-receipts.ts +1 -0
- package/src/interactive/worker-stream.ts +6 -1
- package/src/interactive/yazi-bridge.ts +60 -6
- package/src/tools/agent-tools.ts +30 -1
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +3 -3
- package/src/tools/bash.ts +1 -1
- package/src/tools/bootstrap.ts +4 -0
- package/src/tools/builtin-tool-catalog.ts +52 -22
- package/src/tools/codewiki/code-nav-surface.ts +6 -0
- package/src/tools/codewiki/code-nav.ts +99 -13
- package/src/tools/context/docs-engine.ts +20 -7
- package/src/tools/context/index.ts +59 -21
- package/src/tools/core-bootstrap.ts +28 -6
- package/src/tools/credential-present.ts +1 -2
- package/src/tools/dispatch-arguments.ts +6 -1
- package/src/tools/dispatch-event-text.ts +10 -0
- package/src/tools/dispatch-plan.ts +49 -4
- package/src/tools/dispatch-run-events.ts +1 -1
- package/src/tools/dispatch-runner.ts +12 -0
- package/src/tools/dispatch-schema.ts +338 -0
- package/src/tools/dispatch-types.ts +3 -0
- package/src/tools/dispatch.ts +9 -254
- package/src/tools/ledger.ts +3 -5
- package/src/tools/monitor-surface.ts +5 -13
- package/src/tools/observation.ts +4 -5
- package/src/tools/panes-surface.ts +4 -11
- package/src/tools/panes.ts +4 -2
- package/src/tools/policy.ts +15 -2
- package/src/tools/read.ts +5 -6
- package/src/tools/registry.ts +41 -12
- package/src/tools/result-shaping.ts +18 -14
- package/src/tools/steer-surface.ts +1 -1
- package/src/tools/tasks.ts +1 -1
- package/src/tools/truncate.ts +6 -5
- package/src/tools/verify/surface.ts +6 -12
- package/src/tools/web-fetch-surface.ts +1 -3
- package/src/tools/worker-evidence.ts +3 -1
- package/src/worker/spec-contract.ts +4 -0
- package/dist/builtins-UJLMOVOV.js +0 -17
- package/dist/chunk-5QIAJV2D.js +0 -48
- package/dist/chunk-JZWT5J3Y.js +0 -814
- package/dist/chunk-K7VKOLQQ.js +0 -15
- package/dist/chunk-PMZCIOCJ.js +0 -25
- package/dist/chunk-SUW5DORT.js +0 -819
- package/dist/chunk-UOV2BYIW.js +0 -107
- package/dist/chunk-WR6U3OVP.js +0 -45
- package/dist/chunk-Y45G3AXC.js +0 -1558
- package/dist/reset-EOLM7GVE.js +0 -230
- package/dist/uninstall-N34PCTGJ.js +0 -331
- package/dist/upgrade-H7TOM7YL.js +0 -323
- package/docs/artifact-versions.md +0 -67
- package/docs/development-pipeline.md +0 -121
- package/docs/documentation-coverage.md +0 -46
- package/docs/documentation-guide.md +0 -167
- package/docs/time-conventions.md +0 -101
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: cut-it
|
|
3
|
-
description:
|
|
3
|
+
description: Slices an existing plan, PRD, or milestone into an executable sprint of dependency-ordered vertical slices sized for one agent run each, with done-when verification per slice; never fabricates a plan. Not for deciding the approach; use architecture.
|
|
4
4
|
triggers:
|
|
5
5
|
- cut it
|
|
6
6
|
- slice this plan
|
|
7
7
|
- make this plan executable
|
|
8
8
|
- turn this milestone into a sprint
|
|
9
9
|
- write dependency-ordered vertical slices
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.4.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
allowed-tools:
|
|
13
13
|
- read
|
|
@@ -18,7 +18,6 @@ allowed-tools:
|
|
|
18
18
|
- context
|
|
19
19
|
- code_nav
|
|
20
20
|
- write
|
|
21
|
-
- artifact
|
|
22
21
|
- ask_user
|
|
23
22
|
clio-coder:
|
|
24
23
|
registry-id: iowarp/clio-coder
|
|
@@ -39,6 +38,43 @@ can run one at a time, leaving the build green after every slice. The output
|
|
|
39
38
|
is a `SPRINT.md` another agent can execute cold — no conversation context
|
|
40
39
|
required.
|
|
41
40
|
|
|
41
|
+
## Arguments
|
|
42
|
+
|
|
43
|
+
```text
|
|
44
|
+
cut it [<path to plan>]
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
There is no flag syntax; the trigger is conversational — "cut it", "slice
|
|
48
|
+
this plan", "turn this milestone into a sprint". A path the user names in
|
|
49
|
+
the same request (a specific `PLAN.md`, `PRD.md`, or `milestones/*/prompt.md`)
|
|
50
|
+
is the plan to slice; a path with no plan words near it, or a bare
|
|
51
|
+
destination like "write it to docs/SPRINT.md", is Step 3's output location,
|
|
52
|
+
not the input. When neither is named, Step 1 finds the plan and Step 3
|
|
53
|
+
writes to the repo-root default.
|
|
54
|
+
|
|
55
|
+
This run has no back-and-forth. `ask_user` still executes — it is registered
|
|
56
|
+
and the call succeeds — but nothing answers it in a headless run: every round
|
|
57
|
+
returns `{cancelled: true}` immediately, as an ordinary result, not an error.
|
|
58
|
+
The happy path here rarely needs a question at all — Step 1's "stop and say
|
|
59
|
+
so" for a missing or vague plan is already headless-safe, and Step 3's output
|
|
60
|
+
path defaults without asking. If several plans or milestones are plausible
|
|
61
|
+
candidates and the choice matters, do not stop on an open question: pick the
|
|
62
|
+
most recently modified or most specifically named one, state that choice and
|
|
63
|
+
the alternative you set aside, mark it `assumed — confirm`, and keep going in
|
|
64
|
+
the same turn. Do not end a turn on an unanswered question in `ask_user` or
|
|
65
|
+
in plain text.
|
|
66
|
+
|
|
67
|
+
The steps below are the plan; do not open a task list for them. This skill's
|
|
68
|
+
tool surface is exactly `read`, `grep`, `ls`, `find`, `git`, `context`,
|
|
69
|
+
`code_nav`, `write`, and `ask_user` (`context` and `ask_user` are always
|
|
70
|
+
available regardless). `tasks` and `bash` both sit outside it and any call to
|
|
71
|
+
either is refused — track progress by walking the steps below, not a task
|
|
72
|
+
board; check for a build/test/lint setup (a `package.json`, a Makefile, a
|
|
73
|
+
node/toolchain version) with `find`, `ls`, and `read`, not `bash node
|
|
74
|
+
--version` or `bash find`. The read-only `git` tool (`status`, `log`) is
|
|
75
|
+
useful in Step 1 when locating the plan benefits from recent history or
|
|
76
|
+
uncommitted changes.
|
|
77
|
+
|
|
42
78
|
## Step 1 — Locate the plan
|
|
43
79
|
|
|
44
80
|
In priority order: a file the user names, a plan in the conversation,
|
|
@@ -60,10 +96,14 @@ slicing of a vague plan hides gaps; flagging them is the deliverable.
|
|
|
60
96
|
- **Self-contained.** Real file paths, real commands, concrete steps. A reader
|
|
61
97
|
with zero conversation context can execute it.
|
|
62
98
|
|
|
63
|
-
## Step 3 — Write
|
|
99
|
+
## Step 3 — Write SPRINT.md
|
|
64
100
|
|
|
65
|
-
|
|
66
|
-
|
|
101
|
+
Use the `write` tool. `SPRINT.md` is a plain file in the working tree, not a
|
|
102
|
+
generated report — do not call a tool literally named `artifact` for this;
|
|
103
|
+
that tool is not on this skill's surface and, in this harness, is a
|
|
104
|
+
terminal call that ends the run the instant it is invoked, before you can
|
|
105
|
+
report back. Default output path is `SPRINT.md` at the repo root; honor a
|
|
106
|
+
caller-supplied path from Arguments instead. Format:
|
|
67
107
|
|
|
68
108
|
```markdown
|
|
69
109
|
# Sprint: <name>
|
|
@@ -86,9 +126,29 @@ Format:
|
|
|
86
126
|
"Done when" is the contract, not decoration. If you cannot write a testable
|
|
87
127
|
done-when for a slice, the slice is not ready to cut — go back to the plan.
|
|
88
128
|
|
|
129
|
+
## Step 4 — Report back
|
|
130
|
+
|
|
131
|
+
End the turn with a short final reply, not silence after the write: the path
|
|
132
|
+
you wrote (`SPRINT.md` or the caller-supplied path), the number of slices,
|
|
133
|
+
and a one-line summary of the battle order. This is the only confirmation
|
|
134
|
+
the caller gets that the write actually happened.
|
|
135
|
+
|
|
89
136
|
## Red flags (you are doing it wrong)
|
|
90
137
|
|
|
91
138
|
- A slice whose steps say "and related changes" or "etc."
|
|
92
139
|
- Done-when criteria that restate the goal instead of naming a check.
|
|
93
140
|
- A slice that only compiles when a later slice lands.
|
|
94
141
|
- Slicing a plan you had to invent on the spot.
|
|
142
|
+
- Calling a tool literally named `artifact` because Step 3 talks about "the
|
|
143
|
+
artifact" — that word here means "the deliverable document," not the
|
|
144
|
+
`artifact` tool. That tool is off this skill's surface and, in this
|
|
145
|
+
harness, terminates the run on the spot, writes to
|
|
146
|
+
`.clio-coder/artifacts/` instead of `SPRINT.md`, and skips Step 4 entirely.
|
|
147
|
+
Use `write`.
|
|
148
|
+
- Ending the run right after the write with no final reply — Step 4 is not
|
|
149
|
+
optional.
|
|
150
|
+
- Opening a `tasks` list for the steps above; `tasks` is refused.
|
|
151
|
+
- Reaching for `bash` (`node --version`, `find`, `ls -la`, ...) to check the
|
|
152
|
+
repo's toolchain or structure; `bash` is refused, use `find`/`ls`/`read`.
|
|
153
|
+
- Stopping to wait for an `ask_user` reply that headless runs never send;
|
|
154
|
+
see Arguments.
|
|
@@ -40,3 +40,104 @@ Expected:
|
|
|
40
40
|
|
|
41
41
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
42
42
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Sliced the seeded PLAN.md; judge 4/4.
|
|
43
|
+
|
|
44
|
+
## Battletest record (2026-09-03)
|
|
45
|
+
|
|
46
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_cutit.py`. S1 reuses this
|
|
47
|
+
evals.md's own fixture text verbatim (`src/todos.js` stub + a concrete
|
|
48
|
+
three-feature `PLAN.md`); S2 is an empty repo with only a `README.md`; S3 is
|
|
49
|
+
a `PLAN.md` that says only "improve performance and clean up the code."
|
|
50
|
+
S1 is the primary grading fixture, scored on 12 checks against the raw
|
|
51
|
+
JSONL's tool-call/safety-block stream, the actual `SPRINT.md` written to
|
|
52
|
+
disk, and the reconstructed final assistant text: zero safety blocks, zero
|
|
53
|
+
real `artifact` tool calls, zero `tasks` calls, `SPRINT.md` exists with a
|
|
54
|
+
`## Battle order` and 2+ numbered slices, every slice carries all six
|
|
55
|
+
required fields (Goal/Depends on/Files/Steps/Done when/Out of scope), no
|
|
56
|
+
horizontal-layering red-flag language, slices trace to the plan's concrete
|
|
57
|
+
features, done-when blocks are command-shaped and testable, and the final
|
|
58
|
+
reply names the path and slice count. S2/S3 are graded on 5 checks each
|
|
59
|
+
(zero safety blocks, zero `artifact` calls, no `SPRINT.md` fabricated,
|
|
60
|
+
correctly flags the gap, recommends a next step). Primary model
|
|
61
|
+
`qwen3.8-27b` on `dynamo` (LM Studio); cross-model confirm on
|
|
62
|
+
`ornith1.5-35b-moe` on `mini` (llama.cpp), run against S1 and S2 both.
|
|
63
|
+
|
|
64
|
+
**The bug found reading the source, confirmed empirically first**: the
|
|
65
|
+
frozen skill (0.3.0) had `artifact` in `allowed-tools` and titled Step 3
|
|
66
|
+
"Write the artifact." `src/tools/artifact.ts` sets `terminate: true` on
|
|
67
|
+
every successful call — the run ends the instant the tool executes, with
|
|
68
|
+
no further LLM turn to confirm what happened. v1's run called `artifact`
|
|
69
|
+
with `kind: "plan"` and, by luck, an explicit `path: "SPRINT.md"` (the
|
|
70
|
+
model inferred this from "honor a caller-supplied path" even though no
|
|
71
|
+
caller supplied one) — so the file landed in the right place this time, but
|
|
72
|
+
the run still ended mid-sentence ("Writing the sprint:") with no
|
|
73
|
+
confirmation reply, and a separate `tasks` call (also off-surface) drew a
|
|
74
|
+
real safety block. Score 8/12: missing `reply_mentions_sprint_path`, one
|
|
75
|
+
real safety block. Had the model not guessed an explicit path, the default
|
|
76
|
+
would have been `.clio-coder/artifacts/PLAN.md` (kind defaults to `plan`,
|
|
77
|
+
see `core/artifact-paths.ts`) — wrong file, wrong location, same silent
|
|
78
|
+
termination. This is exactly the failure mode the planning category's
|
|
79
|
+
tech-spec baseline hit ("called `artifact` for an early exit... instead of
|
|
80
|
+
a spec").
|
|
81
|
+
|
|
82
|
+
| run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
|
|
83
|
+
|---|---|---|---|---|---|---|---|
|
|
84
|
+
| baseline (no skill) | qwen3.8-27b | 76s | 7 | 73.8k / 7.1k | 5 (repeated `ls ".cl"` truncated-path retries, benign) | 5/12 | never invoked `/skill cut-it`; read the plan and module correctly, reasoned to genuinely good vertical slices with real done-when checks in its head, then hit a tool-call loop guard and delivered the entire sprint as **prose in the reply, never wrote `SPRINT.md`** — the exact gap this skill exists to close |
|
|
85
|
+
| v1 (frozen 0.3.0) | qwen3.8-27b | 115s | 6 | 68.6k / 11.3k | 1 real (`tasks` refused) | 8/12 | called the real `artifact` tool for Step 3 as titled; terminated the turn immediately after writing, mid-sentence, with no confirmation reply — the artifact-tool bug, confirmed |
|
|
86
|
+
| v2 (first hardened cut, 0.4.0) | qwen3.8-27b | 70s | 6 | 72.4k / 6.5k | 0 | 11/12 (12/12 after a grading-regex fix, see below) | used `write` correctly, used the `git` tool in Step 1, reported the path and slice count in the final reply; the one score miss was a test-harness regex that didn't handle a `**Done when** (fresh state...):` label followed by bulleted checks on the next lines — the actual done-when content was already command-shaped and testable, fixed in the harness, not the skill |
|
|
87
|
+
| v3 (bash-reflex found) | qwen3.8-27b | — | — | — | 2 (1 benign ENOENT, 1 real: `bash` refused) | 11/12 | reached for `bash` (`node --version`, `find`, `ls -la` chained with `&&`) to survey the toolchain even though `bash` was never in `allowed-tools` and nothing in the body named it explicitly — added the same explicit `bash`-refusal line the `tasks` refusal already had, plus a Red flags entry |
|
|
88
|
+
| v4 (final, stable) | qwen3.8-27b | 60s | 6 | 71.4k / 5.6k | 0 | **12/12** | clean run: `context` → `read`/`ls` → `git status` → `write`, self-contained final reply naming path, slice count, and battle order |
|
|
89
|
+
| final-s2 (no plan) | qwen3.8-27b | 26s | 4 | 43.2k / 2.0k | 0 | **5/5** | checked all three plan locations plus `git log`/`status`, correctly stopped with no `SPRINT.md` written, cited the skill's own red-flag language, recommended `grill-me` |
|
|
90
|
+
| final-s3 (vague plan) | qwen3.8-27b | 28s | 5 | 56.2k / 2.5k | 0 | **5/5** | read the one-line `PLAN.md`, explicitly invoked "the skill's own test" (a testable done-when), listed three concrete missing pieces (object/baseline/target for "performance", a definition of "clean", the absent codebase), stopped without writing `SPRINT.md` |
|
|
91
|
+
| final-mini (cross-model, S1) | ornith1.5-35b-moe (mini) | 42s | 6 | 10.8k / 3.2k | 0 | **12/12** | same clean shape on the second model family: `context` → `ls`/`read` → `write`, self-contained final reply |
|
|
92
|
+
| final-mini-s2 (cross-model, S2) | ornith1.5-35b-moe (mini) | 19s | 6 | 4.0k / 1.1k | 0 | **5/5** | correctly found nothing to slice, recommended `grill-me`, offered to slice immediately if pointed at a plan |
|
|
93
|
+
|
|
94
|
+
**Changes** (0.3.0 -> 0.4.0):
|
|
95
|
+
|
|
96
|
+
1. **`artifact` removed from `allowed-tools`.** cut-it never needs the real
|
|
97
|
+
`artifact` tool — `SPRINT.md` is a plain file, always written with
|
|
98
|
+
`write`. This is the fix for the bug above.
|
|
99
|
+
2. **Step 3 retitled** "Write SPRINT.md" (was "Write the artifact") and its
|
|
100
|
+
body now says explicitly "use the `write` tool" and names the
|
|
101
|
+
`artifact`-tool confusion directly, including what it actually does
|
|
102
|
+
wrong (terminal call, wrong default path under `.clio-coder/artifacts/`,
|
|
103
|
+
skips the report-back step).
|
|
104
|
+
3. **New Step 4 — Report back**, an explicit final-reply requirement (path
|
|
105
|
+
written, slice count, one-line battle-order summary). Nothing in the
|
|
106
|
+
frozen skill told the model to confirm after writing; every hardened run
|
|
107
|
+
now does.
|
|
108
|
+
4. **`## Arguments` contract**, the section the frozen skill never had:
|
|
109
|
+
conversational trigger syntax, how a named path splits between "the plan
|
|
110
|
+
to slice" and "where to write `SPRINT.md`", the no-operator/`ask_user`-
|
|
111
|
+
auto-cancels rule (adapted from `grill-me` 0.5.0's "this run has no
|
|
112
|
+
back-and-forth" framing — stated as a fact about the run, not gated on a
|
|
113
|
+
cancellation response), and — since cut-it's happy path rarely needs a
|
|
114
|
+
question at all — explicit guidance for the one place it plausibly might
|
|
115
|
+
(choosing among several candidate plans/milestones): pick the best one,
|
|
116
|
+
state the alternative, mark `assumed — confirm`, keep going.
|
|
117
|
+
5. **`tasks` and `bash` explicitly named as refused**, in Arguments and Red
|
|
118
|
+
Flags, both found empirically: v1's `tasks` call (tracking its own
|
|
119
|
+
steps) and v3's `bash` call (`node --version`, `find`, `ls -la` chained
|
|
120
|
+
with `&&`, to survey the toolchain) were both real safety blocks despite
|
|
121
|
+
neither tool ever having been in `allowed-tools`.
|
|
122
|
+
6. Three new Red Flags entries for the concrete failures observed: the
|
|
123
|
+
`artifact`-tool confusion, ending the run with no final reply, and the
|
|
124
|
+
`bash` reflex (`tasks` already had informal coverage, now explicit too).
|
|
125
|
+
|
|
126
|
+
**Still weak**: `code_nav` (in `allowed-tools`) was never exercised — this
|
|
127
|
+
fixture's grounding fit entirely in one small stub file, so `read`/`grep`
|
|
128
|
+
sufficed; a plan referencing a larger call graph might exercise it, none
|
|
129
|
+
was built here. The `ask_user`-unavailable path and the "several candidate
|
|
130
|
+
plans" disambiguation guidance in Arguments are reasoned prose, not
|
|
131
|
+
empirically run — no fixture here seeds multiple plausible plan files or a
|
|
132
|
+
scenario where the model actually reaches for `ask_user`; every hardened
|
|
133
|
+
run reasoned straight to the assumed-confirm default or never needed a
|
|
134
|
+
question at all. S3's "flag as vague" grading is phrase-matching against a
|
|
135
|
+
fixed term list (`vague`, `underspecified`, `insufficient`, ...) — a run
|
|
136
|
+
that flags the same gap in different words would under-score on a
|
|
137
|
+
technicality, though neither observed run did. The v2 grading-regex miss
|
|
138
|
+
(`done_when_testable`, fixed in the harness before v4) is a reminder that
|
|
139
|
+
this fixture's automated score can undercount a genuinely correct skill
|
|
140
|
+
output; the raw `SPRINT.md` files are worth spot-reading, not just the
|
|
141
|
+
score column. No timing was captured for v3 (an incremental re-run to
|
|
142
|
+
confirm the `bash` finding, superseded immediately by v4) — not a gap in
|
|
143
|
+
the skill's coverage, just an artifact of the iteration order.
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: design-council
|
|
3
|
-
description:
|
|
3
|
+
description: Convenes several expert perspectives that challenge each other on a design decision with real trade-offs before code is written; quick mode runs a single round. Not for a one-question-at-a-time interrogation of a plan; use grill-me. Not for splitting implementation across workers; use dispatch directly.
|
|
4
4
|
triggers:
|
|
5
5
|
- convene a design council
|
|
6
6
|
- debate this design
|
|
7
7
|
- get multiple expert perspectives
|
|
8
8
|
- weigh the architecture options
|
|
9
9
|
- what would experts say
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.5.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
allowed-tools:
|
|
13
13
|
- dispatch
|
|
@@ -17,12 +17,13 @@ allowed-tools:
|
|
|
17
17
|
- ls
|
|
18
18
|
- context
|
|
19
19
|
- code_nav
|
|
20
|
+
- ask_user
|
|
20
21
|
clio-coder:
|
|
21
22
|
registry-id: iowarp/clio-coder
|
|
22
23
|
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/workflow/design-council
|
|
23
24
|
audit: pass
|
|
24
25
|
provenance: designed
|
|
25
|
-
eval-status:
|
|
26
|
+
eval-status: smoke-checked
|
|
26
27
|
model-size: large
|
|
27
28
|
agents:
|
|
28
29
|
- scout
|
|
@@ -36,22 +37,95 @@ Run a bounded multi-perspective debate on a real design decision. The council
|
|
|
36
37
|
surfaces the crux of a disagreement before code commits to one side. It is not
|
|
37
38
|
a ritual: if experts would agree, do not convene it.
|
|
38
39
|
|
|
40
|
+
## Arguments
|
|
41
|
+
|
|
42
|
+
```text
|
|
43
|
+
convene a design council on <decision>
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
There is no flag syntax; the trigger is conversational — "convene a design
|
|
47
|
+
council on X", "debate this design", "get multiple expert perspectives".
|
|
48
|
+
Whatever the user names is the decision. A referenced file, doc, or repo path
|
|
49
|
+
in the same request is Step 0/1's grounding to read first, not a separate
|
|
50
|
+
argument.
|
|
51
|
+
|
|
52
|
+
**Headless is the enforced default, not a suggestion.** A council is several
|
|
53
|
+
worker runs; nobody is waiting between rounds in a headless run, and every
|
|
54
|
+
round that actually runs costs real wall-clock time on top of the
|
|
55
|
+
orchestrator's own turns. `ask_user` still executes here even though it is
|
|
56
|
+
not in the list below — it is always available regardless — but nothing
|
|
57
|
+
answers it: the single question Step 1 asks comes back `{cancelled: true}`
|
|
58
|
+
immediately, as an ordinary result, not an error. Treat that cancellation (or
|
|
59
|
+
skip the call and reason from this paragraph directly — one is not more valid
|
|
60
|
+
than the other) as the fixed answer **quick mode: exactly three perspectives,
|
|
61
|
+
exactly one round (Positions) plus synthesis.** Do not compose four or five
|
|
62
|
+
perspectives headlessly and do not run a Responses or Convergence round
|
|
63
|
+
headlessly, no matter how contested the topic looks — four/five perspectives
|
|
64
|
+
and multi-round debate are for a live session with an operator who actually
|
|
65
|
+
asked for the deeper pass. This is the one rule the skill's own prior smoke
|
|
66
|
+
history says a model will not reliably self-infer from prose alone, so treat
|
|
67
|
+
the number 3 and the number 1 as hard, not as defaults to raise if the topic
|
|
68
|
+
seems to deserve more.
|
|
69
|
+
|
|
70
|
+
**Dispatch call shape.** Compose the round's perspectives, then make exactly
|
|
71
|
+
one `dispatch` call with all of them in `tasks` and `mode="parallel"`. If
|
|
72
|
+
that call comes back admission-denied for endpoint or target capacity (a
|
|
73
|
+
single local model instance commonly allows only one concurrent worker, so a
|
|
74
|
+
3-task parallel wave can be denied outright rather than queued), retry the
|
|
75
|
+
identical `tasks` batch in one dispatch call with `mode="sequential"` instead
|
|
76
|
+
— the tool runs them one after another itself. Never split a round into
|
|
77
|
+
several separate one-task `dispatch` calls made one at a time waiting on each
|
|
78
|
+
result before deciding the next; that is the serial-perspectives pattern that
|
|
79
|
+
produced the round-trip cost the skill's own timeout history is about, and it
|
|
80
|
+
does not fix the capacity problem the parallel call already reported. Do not
|
|
81
|
+
call `dispatch(list:true)` to probe capacity first — it answers nothing about
|
|
82
|
+
concurrency and only spends a call.
|
|
83
|
+
|
|
84
|
+
Declare `intent: {read_roots: [...], relevant_paths: [...]}` with paths
|
|
85
|
+
relative to the repo root on every dispatch call instead of pasting an
|
|
86
|
+
absolute path into `task`/`briefing` prose (a config value, a mount point).
|
|
87
|
+
An absolute path token in briefing/task text with no declared `intent` is
|
|
88
|
+
rejected as `legacy_scope_path_absolute`. Keep a persona's argument in prose,
|
|
89
|
+
never literal shell syntax — a phrase like "you can `rm -rf` the directory"
|
|
90
|
+
inside a dispatch call's text can trip the same damage-control pattern that
|
|
91
|
+
blocks a real destructive shell command, even though nothing executes; say
|
|
92
|
+
"delete the directory" instead.
|
|
93
|
+
|
|
94
|
+
A dispatch call's own synchronous result already carries every worker's
|
|
95
|
+
output — do not follow it with a `bash`/`read` pass over the receipt file on
|
|
96
|
+
disk to re-read what you already have. The steps below are the plan; do not
|
|
97
|
+
open a `tasks` list for them. This skill's tool surface is exactly
|
|
98
|
+
`dispatch`, `read`, `grep`, `find`, `ls`, `context`, `code_nav`, and
|
|
99
|
+
`ask_user` (`context` and `ask_user` are always available regardless).
|
|
100
|
+
`tasks` and `bash` both sit outside it and any call to either is refused —
|
|
101
|
+
locate files with `find`/`ls`, not `bash find`/`bash ls`; inspect a receipt
|
|
102
|
+
with `read`, not `bash cat`. This skill never writes: `write` and `artifact`
|
|
103
|
+
are not on its surface, so the synthesis in Step 4 is chat output, never a
|
|
104
|
+
file.
|
|
105
|
+
|
|
39
106
|
## Step 0 — Check the question is contested
|
|
40
107
|
|
|
41
108
|
Before composing anyone, ask: would credible experts actually disagree on the
|
|
42
109
|
answer? If every perspective you can imagine picks the same option and differs
|
|
43
|
-
only in caveats, stop here. Say the council is not
|
|
44
|
-
answer with the caveats attached, and end.
|
|
110
|
+
only in caveats, stop here — do not dispatch anything. Say the council is not
|
|
111
|
+
needed, give the consensus answer with the caveats attached, and end. Do not
|
|
112
|
+
dispatch a round "just to confirm" a consensus call you already reached —
|
|
113
|
+
that is the ritual this step exists to skip, and it still costs the same
|
|
114
|
+
wall-clock time and worker slots as a real debate. Trust this self-check the
|
|
115
|
+
same way you trust the rest of your own reasoning; a council you convened to
|
|
116
|
+
double-check yourself is not more rigorous than the judgment behind it.
|
|
45
117
|
|
|
46
118
|
## Step 1 — Compose perspectives
|
|
47
119
|
|
|
48
120
|
Derive perspectives from the topic itself, never from a generic role menu.
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
121
|
+
On a genuinely contested topic, call `ask_user` once with `mode:
|
|
122
|
+
"single_question"` asking whether this should be a quick pass (three
|
|
123
|
+
perspectives, one round) or the full debate (up to five perspectives, up to
|
|
124
|
+
three rounds). See Arguments for what a headless run does with that call.
|
|
125
|
+
In a live session where the user answers, honor the requested depth. Compose
|
|
126
|
+
three perspectives by default; go to four or five only in that live full-
|
|
127
|
+
debate case, and only when the decision genuinely has that many independent
|
|
128
|
+
stances.
|
|
55
129
|
|
|
56
130
|
Each perspective gets:
|
|
57
131
|
|
|
@@ -74,11 +148,13 @@ perspective from the read-only recipes in the live catalog:
|
|
|
74
148
|
- `researcher`: stance leaning on external docs, standards, or papers.
|
|
75
149
|
- `provenance`: stance arguing from runtime evidence and receipts.
|
|
76
150
|
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
151
|
+
See Arguments for the exact call shape (one batched `tasks` call, the
|
|
152
|
+
`mode="sequential"` capacity fallback, `intent` for scope, no literal shell
|
|
153
|
+
syntax). Rounds are sequential; a round's own perspectives are the one
|
|
154
|
+
dispatch call. Each task prompt carries the persona block, the decision
|
|
155
|
+
context, and the full transcript so far. Workers never edit files; the
|
|
156
|
+
debate is analysis only. Dispatch receipts link every statement to a worker
|
|
157
|
+
run.
|
|
82
158
|
|
|
83
159
|
## Step 3 — Run the rounds
|
|
84
160
|
|
|
@@ -90,14 +166,15 @@ Dispatch receipts link every statement to a worker run.
|
|
|
90
166
|
still disagrees and why that crux is the crux, and its final
|
|
91
167
|
recommendation.
|
|
92
168
|
|
|
93
|
-
**Quick mode** (user
|
|
94
|
-
responses round.
|
|
169
|
+
**Quick mode** (headless default, or a user asking for a light pass): round 1
|
|
170
|
+
plus synthesis. No responses round, no convergence round.
|
|
95
171
|
|
|
96
|
-
**Early termination
|
|
97
|
-
question itself, not on side conditions. If
|
|
98
|
-
option and differs only in caveats, toggles,
|
|
99
|
-
that is consensus: skip rounds 2 and 3, report
|
|
100
|
-
needed, and return the consensus with caveats.
|
|
172
|
+
**Early termination** (full-debate mode only). After round 1, judge
|
|
173
|
+
disagreement on the decision question itself, not on side conditions. If
|
|
174
|
+
every position picks the same option and differs only in caveats, toggles,
|
|
175
|
+
or requests to measure later, that is consensus: skip rounds 2 and 3, report
|
|
176
|
+
that the council was not needed, and return the consensus with caveats.
|
|
177
|
+
Never manufacture friction.
|
|
101
178
|
|
|
102
179
|
## Step 4 — Synthesize
|
|
103
180
|
|
|
@@ -124,10 +201,11 @@ synthesis line has a citation and the recommendation names its crux.
|
|
|
124
201
|
|
|
125
202
|
## Degraded mode
|
|
126
203
|
|
|
127
|
-
If dispatch is unavailable or admission-denied
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
204
|
+
If dispatch is unavailable or admission-denied for a reason other than the
|
|
205
|
+
capacity retry in Arguments (the tool itself is missing from the surface, or
|
|
206
|
+
every retry is denied), run the same rounds inline: write each perspective's
|
|
207
|
+
contribution yourself, sequentially, same round structure and synthesis
|
|
208
|
+
format. Label the output as degraded (single-model debate, no receipts).
|
|
131
209
|
|
|
132
210
|
## Boundaries
|
|
133
211
|
|
|
@@ -141,7 +219,19 @@ this skill when the debate needs the full round structure and receipts.
|
|
|
141
219
|
## Red Flags
|
|
142
220
|
|
|
143
221
|
- Perspectives named "optimist" and "pessimist" (role menu, not topic).
|
|
144
|
-
-
|
|
222
|
+
- Composing four or five perspectives, or running a Responses/Convergence
|
|
223
|
+
round, in a headless run — the enforced default is exactly three and
|
|
224
|
+
exactly one, not a ceiling to raise because the topic looks deep.
|
|
225
|
+
- Splitting a round into several single-task `dispatch` calls issued one at a
|
|
226
|
+
time instead of one batched `tasks` call (retried as `mode="sequential"`
|
|
227
|
+
on a capacity denial).
|
|
228
|
+
- Calling `dispatch(list:true)` before dispatching the round.
|
|
229
|
+
- Pasting an absolute path into a dispatch `task`/`briefing` instead of a
|
|
230
|
+
declared `intent`, or quoting literal shell syntax inside a persona's
|
|
231
|
+
argument.
|
|
232
|
+
- Reaching for `bash`/`read` on a receipt file the dispatch call's own result
|
|
233
|
+
already contains.
|
|
234
|
+
- Opening a `tasks` list for the round structure; `tasks` is refused.
|
|
145
235
|
- A synthesis that averages positions instead of naming the crux.
|
|
146
236
|
- Manufactured disagreement on a settled question.
|
|
147
237
|
- A worker asked to edit files as part of the debate.
|
|
@@ -95,3 +95,164 @@ default and tells a headless run to use three and one round, which is the
|
|
|
95
95
|
blocker the timeout exposed. Not re-run, so `eval-status` stays
|
|
96
96
|
`scenarios-recorded`; the next campaign has to confirm the shortened council
|
|
97
97
|
fits the ceiling.
|
|
98
|
+
|
|
99
|
+
## Battletest record (2026-09-03) — 0.4.0 -> 0.5.0
|
|
100
|
+
|
|
101
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_designcouncil.py`, a
|
|
102
|
+
self-contained repo (`src/checkpoint.py` writing one raw `.npy` per rank per
|
|
103
|
+
step to a shared parallel filesystem, `src/config.py`/`config.yaml` using
|
|
104
|
+
`pyyaml`) adapted from this file's own worked examples. S1 = HDF5-vs-Zarr
|
|
105
|
+
(genuinely contested), S2 = hand-roll-YAML-vs-keep-pyyaml (consensus), S4 =
|
|
106
|
+
"poke holes in my plan" anti-trigger. `runner.py`'s `timeout=` was raised to
|
|
107
|
+
1800s for this skill specifically — the 900s default is a `dispatch` fan-out
|
|
108
|
+
ceiling, not a model-quality one, and the mission called for confirming the
|
|
109
|
+
v0.3.0 "three perspectives, one round" headless fix on its own terms, not
|
|
110
|
+
routing around it. Primary: `qwen3.8-27b`/`dynamo`. Cross-model confirm:
|
|
111
|
+
`ornith1.5-35b-moe`/`mini`. Both runs shared the `dynamo` LM Studio endpoint
|
|
112
|
+
with concurrently active sibling battletest sessions (`grill-me`,
|
|
113
|
+
`workflow-distiller`) launched via `herdr` during this pass — real,
|
|
114
|
+
externally-caused contention, not a fixture artifact; see below.
|
|
115
|
+
|
|
116
|
+
| run | scenario | model | wall | turns | dispatch calls | safety blocks | score | outcome |
|
|
117
|
+
|---|---|---|---|---|---|---|---|---|
|
|
118
|
+
| baseline (no skill) | S1 | qwen3.8-27b | 121s | 5 | 0 | 0 | 3/11 | no skill invoked (ran under `--no-skills`); the model noticed the installed skill anyway and narrated "four positions, real cross-examination" as one inline monologue — a solid single-model analysis but no dispatch, no receipts, no named-recipe workers |
|
|
119
|
+
| v1 (frozen 0.4.0) | S1 | qwen3.8-27b | 1740s, **killed at the 1800s ceiling** (exit -9) | 22 | 12 | 7 | 4/11 | composed **four** perspectives and ran into **round 2**, both violating the stated headless "three perspectives, one round" default; opened a `tasks` plan (refused); two `dispatch` calls rejected for an absolute path token in `briefing` (`legacy_scope_path_absolute`); one denied for endpoint capacity; one **hard-blocked as `rm-recursive-or-force`** because a perspective's own round-1 argument prose said "...you can `ls`, checksum, `rm -rf`, and publish..." — the admission layer's damage-control scan matched the quoted shell syntax inside the debate text itself, not an executed command; reached for `bash` to inspect a receipt file (refused). This is the same failure class the 2026-08-13 smoke record flagged, confirmed still present, worse: the v0.3.0 fix was never actually followed |
|
|
120
|
+
| v2 (first hardened cut) | S1 | qwen3.8-27b | 416s | 11 | 3 | 4 | 8/11 | one batched `tasks`-array `dispatch` call, all 3 capacity/timeout-denied under real endpoint contention; correctly fell back to Degraded mode, cited the exact denial reasons, labeled the output degraded, produced a complete synthesis with agreements/crux/dissent — no `tasks` misuse, no shell-block trigger, no one-at-a-time call splitting |
|
|
121
|
+
| v2 (post Step 0 strengthen) | S1 | qwen3.8-27b | 304s | 7 | 3 | 4 | 8/11 | same shape: parallel wave denied, sequential retry denied/timed out, correct Degraded fallback, well-formed synthesis citing the transcript |
|
|
122
|
+
| v2 | S2 | qwen3.8-27b | 52.5s / 60.6s (before/after Step 0 edit) | 5 / 6 | **0** | 0 | **5/5** both | Step 0 stopped before any dispatch, stated the council was not needed, gave the consensus (keep `pyyaml`, `safe_load` only) with caveats, and named the repo's actually-contested question (the checkpoint format) as a pointer — no manufactured friction either run |
|
|
123
|
+
| v2 | S4 | qwen3.8-27b | 69.4s | — | 0 | 0 | 3/3 | did not convene a council; ran a grill-me-shaped one-question-at-a-time interrogation instead and named `grill-me` |
|
|
124
|
+
| v2 (cross-model) | S1 | ornith1.5-35b-moe/mini | 384s | 10 | 4 | 5 | 7/11 | 3 of 4 `dispatch` calls denied/timed out on the **same `dynamo` endpoint** (workers defaulted there even though the orchestrator itself ran on `mini`); one stray `monitor` call refused (outside the skill's surface); correctly diagnosed the capacity pattern in its own words ("times out at ~57s... despite reporting 0/4 slots in use... exhausted the capacity fallback") and ran Degraded mode with a clearly labeled warning banner |
|
|
125
|
+
| v2 (cross-model) | S2 | ornith1.5-35b-moe/mini | 239.5s / 188.1s (before/after Step 0 edit) | 6 | 3 / 2 | 3 / 2 | 2/5 both | did **not** skip Step 0's dispatch the way qwen did — ran one capacity-retried round anyway, but stopped there (no round 2/3), reached the same correct consensus ("keep pyyaml", supply-chain caveat preserved as a contingent dissent, not manufactured), and stated the council-not-needed conclusion explicitly; the Step 0 wording strengthen did not change this model's behavior |
|
|
126
|
+
|
|
127
|
+
**S3 (dispatch unavailable) — no clean forced trigger found, confirmed by
|
|
128
|
+
reading the CLI, not assumed**: `clio-coder run --help`'s `--tool-profile`
|
|
129
|
+
narrows a *dispatched sub-agent's own* tool set and requires `--agent`
|
|
130
|
+
(`clio-coder run: fleet dispatch flags require --agent <recipe-id>`); there
|
|
131
|
+
is no flag that strips `dispatch` from a top-level `--skill` orchestrator
|
|
132
|
+
run's own surface in this harness. Editing the skill's own `allowed-tools`
|
|
133
|
+
to omit `dispatch` would test a different skill, not this one. This is a
|
|
134
|
+
real, documented harness gap, not faked around. In its place, S3's expected
|
|
135
|
+
behavior was exercised **organically, repeatedly, for real**: every S1 run
|
|
136
|
+
on both models hit genuine `dispatch` admission denial or timeout from
|
|
137
|
+
endpoint capacity, and Degraded mode fired correctly every single time —
|
|
138
|
+
labeled degraded, same round structure, synthesis format intact, no
|
|
139
|
+
receipts claimed that did not exist.
|
|
140
|
+
|
|
141
|
+
**The dispatch-viability and recipe-existence questions, resolved
|
|
142
|
+
empirically:**
|
|
143
|
+
|
|
144
|
+
- `scout`, `researcher`, and `provenance` all exist as builtin shadow-agent
|
|
145
|
+
recipes (`src/domains/agents/builtins/*.md`, `capabilityClass: read-only`)
|
|
146
|
+
and are dispatchable — every hardened run that got a `dispatch` call
|
|
147
|
+
admitted used one of them by name and got real worker output back.
|
|
148
|
+
- Each recipe's own system prompt declares a rigid JSON-only result
|
|
149
|
+
contract (`{"findings":[...]}`, `{"source":...}`, `{"confirmedFacts":...}`)
|
|
150
|
+
that has nothing to do with a debate position. In practice this was
|
|
151
|
+
**not** a blocker: V1's successfully-admitted round 1 came back as full
|
|
152
|
+
argumentative prose (named claims, attacks, citations, mind-change
|
|
153
|
+
conditions) from `researcher`/`scout` workers, not the declared narrow
|
|
154
|
+
JSON shape — the persona in the task prompt won out over the recipe's
|
|
155
|
+
own stated output contract for the caller-facing transcript. Worth
|
|
156
|
+
knowing, not worth re-architecting around.
|
|
157
|
+
- The actual, dominant blocker is **`dispatch` admission capacity on a
|
|
158
|
+
single-instance local LM Studio target**. `~/.local/state/clio-coder/
|
|
159
|
+
endpoint-slots.json` records the `dynamo` endpoint (`100.104.197.69:1234`)
|
|
160
|
+
at `"slots": 1`. A 3-task parallel wave is denied outright
|
|
161
|
+
(`capacity exceeded (3/1 slots)`), and the `mode="sequential"` retry this
|
|
162
|
+
pass added to the skill is followed correctly by both models but still
|
|
163
|
+
gets denied or times out (`admission timed out after ~57s... 0/4 worker
|
|
164
|
+
slots in use` — a scheduling state that never resolves within the wait
|
|
165
|
+
window), evidently because the orchestrator's own foreground session
|
|
166
|
+
already holds the endpoint's only slot. This reproduced independently on
|
|
167
|
+
`mini` too: workers dispatched from an orchestrator running on `mini`
|
|
168
|
+
still routed to `dynamo`'s endpoint by default, so the same 1-slot
|
|
169
|
+
contention applied there as well — confirming the brief's suspicion that
|
|
170
|
+
worker fan-out may silently default to an unexpected node/target. Some
|
|
171
|
+
of this pass's contention was real concurrent load, not just self-
|
|
172
|
+
contention: `ps` showed sibling `grill-me`/`workflow-distiller`
|
|
173
|
+
battletest sessions actively running via `herdr` against the same
|
|
174
|
+
endpoint during these runs.
|
|
175
|
+
- The full-auto/`authorityBasis` auto-grant fact supplied going in
|
|
176
|
+
(`deps.getAutonomy?.() === "full-auto" ? "full-auto-policy" :
|
|
177
|
+
"operator-plan-approval"`) turned out to be **moot for this skill**:
|
|
178
|
+
that gate only applies to `agent:"auto"` dispatch requests
|
|
179
|
+
(`src/tools/dispatch-arguments.ts`, `agentSelection` is only populated
|
|
180
|
+
when `requestedAgent === "auto"`). Design-council always pins an explicit
|
|
181
|
+
recipe id (`scout`/`researcher`/`provenance`), so `agentSelection` stays
|
|
182
|
+
`undefined` and the operator-approval/full-auto-policy distinction never
|
|
183
|
+
engages — admission for this skill's calls is governed purely by the
|
|
184
|
+
capacity/reservation machinery above, independent of `--autonomy`.
|
|
185
|
+
- `legacy_scope_path_absolute`: an absolute path token (e.g. a value read
|
|
186
|
+
from `config.yaml`) pasted into `task`/`briefing` prose without a
|
|
187
|
+
declared `intent` is rejected. Fixed by telling the skill to declare
|
|
188
|
+
`intent.read_roots`/`relevant_paths` (relative) on every call instead.
|
|
189
|
+
|
|
190
|
+
**Changes (0.4.0 -> 0.5.0):**
|
|
191
|
+
|
|
192
|
+
1. **`## Arguments` contract**, ported from `grill-me`/`cut-it`'s shape.
|
|
193
|
+
States headless is the *enforced* default (exactly three perspectives,
|
|
194
|
+
exactly one round), not a self-assessed suggestion — the v0.3.0 fix's
|
|
195
|
+
prose alone did not hold on either model (V1 composed four perspectives
|
|
196
|
+
and ran round 2 headlessly).
|
|
197
|
+
2. **Dispatch call shape spelled out**: one batched `tasks`-array call per
|
|
198
|
+
round; on capacity denial, retry the *same* batch with
|
|
199
|
+
`mode="sequential"` in one call rather than splitting into several
|
|
200
|
+
single-task calls issued one at a time (V1's actual failure — 12
|
|
201
|
+
separate `dispatch` calls, a `list:true` probe, and manual receipt
|
|
202
|
+
reads via `bash`, all of which V2 stopped doing).
|
|
203
|
+
3. **`intent.read_roots`/`relevant_paths` guidance** to avoid
|
|
204
|
+
`legacy_scope_path_absolute` rejections from paths quoted in prose.
|
|
205
|
+
4. **No literal shell syntax in a persona's argument text** — added after
|
|
206
|
+
V1's `rm-recursive-or-force` hard block fired on debate prose, not a
|
|
207
|
+
real command.
|
|
208
|
+
5. **`ask_user` added to `allowed-tools`** (already always-exempt, now
|
|
209
|
+
documented) with one call at Step 1 on a contested topic, asking
|
|
210
|
+
quick-vs-full-debate depth; headlessly it cancels, which *is* the
|
|
211
|
+
three-perspective/one-round confirmation, mirroring the auto-cancel
|
|
212
|
+
pattern the planning category established rather than relying on
|
|
213
|
+
unaided self-assessment.
|
|
214
|
+
6. **Explicit `tasks`/`bash`/receipt-re-read refusal lines**, matching the
|
|
215
|
+
sibling skills' pattern (`tasks` opened a plan in V1; `bash` was reached
|
|
216
|
+
for twice across the runs above to re-inspect a receipt the dispatch
|
|
217
|
+
call's own result already contained).
|
|
218
|
+
7. **Step 0 strengthened** against dispatching "just to confirm" a
|
|
219
|
+
consensus call already reached — tested on `ornith1.5-35b-moe`/`mini`
|
|
220
|
+
(still dispatches once) and `qwen3.8-27b`/`dynamo` (unaffected, already
|
|
221
|
+
correct); see Still weak.
|
|
222
|
+
8. Five new Red flags entries naming the concrete failures observed above.
|
|
223
|
+
|
|
224
|
+
**Still weak:**
|
|
225
|
+
|
|
226
|
+
- **Step 0's short-circuit is model-family-dependent.** `qwen3.8-27b`
|
|
227
|
+
trusts its own contested-or-not judgment and skips `dispatch` entirely
|
|
228
|
+
for S2 (5/5, 0 dispatch calls, both before and after the Step 0 edit).
|
|
229
|
+
`ornith1.5-35b-moe` does not: it dispatches one (capacity-retried) round
|
|
230
|
+
"to confirm" even on the same consensus topic, both before and after the
|
|
231
|
+
strengthened wording — 2/5 unchanged. It still stops after that one
|
|
232
|
+
round, still reaches the correct consensus, still preserves the dissent
|
|
233
|
+
as a contingent caveat rather than manufacturing friction, so the
|
|
234
|
+
user-facing outcome is fine; the wasted dispatch cost is the actual gap,
|
|
235
|
+
and more prose did not move it, consistent with the planning category's
|
|
236
|
+
own finding that some model-family behaviors don't fully generalize no
|
|
237
|
+
matter how much repetition is added.
|
|
238
|
+
- **S1 never completed as a genuine 3-perspective/1-round council with
|
|
239
|
+
real receipts** on this pass — every attempt on both models hit real
|
|
240
|
+
endpoint capacity contention and degraded. The Degraded path is now
|
|
241
|
+
proven solid, but the "happy path" (dispatch succeeds, synthesis cites
|
|
242
|
+
real worker receipts) is unverified under this specific pass's
|
|
243
|
+
conditions; V1's transcript shows it *can* succeed (four real worker
|
|
244
|
+
runs completed with citable output before the round-2/shell-block
|
|
245
|
+
failure), so this reads as an availability problem this pass's timing
|
|
246
|
+
ran into, not a structural block — but it means the exact scoring
|
|
247
|
+
bullets that depend on `dispatch_receipts_ok` (perspective count,
|
|
248
|
+
receipts-ok) were not exercised clean this round.
|
|
249
|
+
`perspectives_dispatched`/`used_named_recipes` in the grading script
|
|
250
|
+
count attempted, not completed, dispatch tasks for this reason.
|
|
251
|
+
- **No purpose-built S3 fixture** — see above; a future pass with a
|
|
252
|
+
dedicated low-capacity or offline dispatch target would let this be
|
|
253
|
+
tested directly instead of relying on organic contention.
|
|
254
|
+
- `code_nav` and `grep` were barely exercised (fixture is small enough
|
|
255
|
+
that `read`/`ls`/`find` covered grounding in most runs).
|
|
256
|
+
- Only one fixture domain (HDF5-vs-Zarr / YAML-parser) ran this pass; the
|
|
257
|
+
richer four-perspective MPI-checkpoint example from "Observed live-smoke
|
|
258
|
+
results" above was not re-run against 0.5.0.
|