@iowarp/clio-coder 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +127 -0
- package/CONTRIBUTING.md +142 -52
- package/README.md +434 -473
- package/SECURITY.md +2 -1
- package/dist/{acp-ZILU3AUO.js → acp-H2NGRPWO.js} +12 -12
- package/dist/{agents-HYWGBGQR.js → agents-TL5LLUQP.js} +56 -55
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-N3QT7CBO.js → auth-E5SW4HMS.js} +23 -21
- package/dist/builtins-IA7V7FUC.js +22 -0
- package/dist/{chunk-7RY5VZPH.js → chunk-2APPQIER.js} +8 -8
- package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
- package/dist/{chunk-JA5QWE4Z.js → chunk-2UG5F4C5.js} +1973 -1664
- package/dist/{chunk-5YHDIDBP.js → chunk-2UH2KFUP.js} +2 -2
- package/dist/{chunk-CTJ4RNAA.js → chunk-2VIKGWFZ.js} +2 -2
- package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
- package/dist/{chunk-GIZNH63R.js → chunk-35MSIRKH.js} +9 -4
- package/dist/chunk-3EBYEESD.js +314 -0
- package/dist/{chunk-J5LZHVIT.js → chunk-3M6DQK6S.js} +113 -35
- package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
- package/dist/{chunk-6FN3E6KX.js → chunk-4O6MANBS.js} +2 -2
- package/dist/chunk-4UVU7BJ5.js +39 -0
- package/dist/{chunk-VKRH2TCS.js → chunk-4WR7VSYB.js} +2 -2
- package/dist/{chunk-BBTJOK6Y.js → chunk-54CBCGIR.js} +5 -5
- package/dist/{chunk-AP73CFDC.js → chunk-5ICU3EUH.js} +2 -2
- package/dist/chunk-5MEZN6CB.js +1334 -0
- package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
- package/dist/{chunk-ABLSQ6JX.js → chunk-64I3JVYM.js} +8 -2
- package/dist/{chunk-AFKWHWXF.js → chunk-6PTFB5VS.js} +39 -22
- package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
- package/dist/chunk-7DRAWPTZ.js +360 -0
- package/dist/chunk-7E7I3WLS.js +3762 -0
- package/dist/{chunk-BJGUKIG4.js → chunk-7ZYNNDKC.js} +7 -7
- package/dist/{chunk-XKA2ICR3.js → chunk-AF4YM7Z4.js} +652 -252
- package/dist/{chunk-GVQJ5CCZ.js → chunk-AX2THNSA.js} +12 -12
- package/dist/{chunk-IG7BCQBA.js → chunk-B4OAX3SI.js} +65 -3
- package/dist/{chunk-TD3PGPQA.js → chunk-B4VEBZKF.js} +3 -3
- package/dist/{chunk-74YWRRU5.js → chunk-BEPZRGGU.js} +10 -10
- package/dist/{chunk-FEFIFZTL.js → chunk-CE5AX47J.js} +2 -2
- package/dist/{chunk-UAPGZHYC.js → chunk-DWUOQKRU.js} +25 -11
- package/dist/{chunk-THYWACCR.js → chunk-E3TPLWFX.js} +3 -3
- package/dist/{chunk-7EPLI7VL.js → chunk-EKCHAPYA.js} +2 -2
- package/dist/{chunk-HLW2MRKE.js → chunk-F4EKGO4N.js} +3 -1
- package/dist/{chunk-PJJ6MY27.js → chunk-F5JHEYZM.js} +7 -7
- package/dist/{chunk-6CCS4G3W.js → chunk-FTMGRKEF.js} +3 -3
- package/dist/{chunk-SINK3QR6.js → chunk-G76U63X4.js} +17 -17
- package/dist/{chunk-EIMVLWB3.js → chunk-GHS5EBTQ.js} +64 -9
- package/dist/{chunk-QMXC4JB7.js → chunk-GI7YYQ3F.js} +187 -1419
- package/dist/{chunk-TZSKNMZG.js → chunk-GTUD2WMY.js} +2 -1
- package/dist/{chunk-6HMJX2VU.js → chunk-GWZNEVM2.js} +44 -12
- package/dist/chunk-GYV6VZOC.js +26 -0
- package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
- package/dist/{chunk-UXN6JT4W.js → chunk-HEQY7ZFI.js} +3 -3
- package/dist/{chunk-7PWAODYW.js → chunk-I7XBWTYH.js} +2 -2
- package/dist/{chunk-GCSMB2KY.js → chunk-I7ZPNEJM.js} +145 -102
- package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
- package/dist/{chunk-QTFGO774.js → chunk-IGLP3ODT.js} +29 -16
- package/dist/chunk-IJNZMHLA.js +101 -0
- package/dist/{chunk-BDPT6GTK.js → chunk-INY6HTFL.js} +7 -7
- package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
- package/dist/{chunk-6NJQITNH.js → chunk-IWT4SF4R.js} +6 -3
- package/dist/{chunk-R23Z6K6I.js → chunk-JDAY6FIL.js} +19 -19
- package/dist/chunk-JEQ3XTHC.js +42 -0
- package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
- package/dist/{chunk-TVH4ONAM.js → chunk-JKKCYP3C.js} +10 -10
- package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
- package/dist/{chunk-C537JADH.js → chunk-KK4JZPBQ.js} +19 -141
- package/dist/{chunk-K6BF4U2H.js → chunk-KKOJXO6R.js} +62 -14
- package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
- package/dist/{chunk-6DWBAZ5U.js → chunk-L47TF46W.js} +5 -7
- package/dist/{chunk-HUAS7ITX.js → chunk-LDJG7DW3.js} +91 -42
- package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
- package/dist/{chunk-YPI3QQCF.js → chunk-MCEPRMZW.js} +2 -4
- package/dist/{chunk-Y4CAGMM6.js → chunk-MNJGS2IN.js} +5 -6
- package/dist/{chunk-VKFQTNDV.js → chunk-MUW2BDDH.js} +4 -4
- package/dist/{chunk-E67WX76H.js → chunk-MWUZBSAQ.js} +104 -152
- package/dist/{chunk-OJTRZGR3.js → chunk-N2Z7HLVY.js} +21 -21
- package/dist/{chunk-TVHHYFHE.js → chunk-NEDJ26B5.js} +2 -2
- package/dist/{chunk-FYUN5KZ3.js → chunk-NIQJ66N4.js} +21 -21
- package/dist/{chunk-U2WB7TZS.js → chunk-NMJXSHBJ.js} +97 -85
- package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
- package/dist/{chunk-VEGN6WIQ.js → chunk-O5CVSAG5.js} +3 -3
- package/dist/{chunk-MOPSG2X7.js → chunk-OML5D5V5.js} +8 -8
- package/dist/{chunk-2VG7KLYV.js → chunk-PAJQJ7BS.js} +5816 -3255
- package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
- package/dist/{chunk-BTGG6BG2.js → chunk-QWGDJJYJ.js} +158 -19
- package/dist/chunk-R6Q67RJH.js +134 -0
- package/dist/{chunk-ZJLUDYFY.js → chunk-RRNP2ANY.js} +6 -6
- package/dist/{chunk-PVAMAVBB.js → chunk-RSJ25QSL.js} +102 -2
- package/dist/{chunk-NLFAQR7Z.js → chunk-S66XZJOF.js} +3 -23
- package/dist/chunk-SKHCAU7K.js +385 -0
- package/dist/chunk-SZAA6XDG.js +30 -0
- package/dist/{chunk-J4HBWF6Y.js → chunk-TM6LQDI3.js} +131 -28
- package/dist/chunk-UOIZ7DA4.js +41 -0
- package/dist/{chunk-MA3H6DM5.js → chunk-UPZU6GE4.js} +25 -3
- package/dist/{chunk-BWW4HLO4.js → chunk-UXCU4E3T.js} +8 -6
- package/dist/{chunk-N5UK64DP.js → chunk-V2ANDPVT.js} +4 -4
- package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
- package/dist/{chunk-6VC4OV3Z.js → chunk-VIA6RFQZ.js} +3 -11
- package/dist/{chunk-ZAZB4JMW.js → chunk-VKPAQYEB.js} +27 -8
- package/dist/{chunk-QKIFBZKT.js → chunk-VW6DOEDG.js} +497 -81
- package/dist/{chunk-SCYB3HA4.js → chunk-W6RRQCPQ.js} +63 -19
- package/dist/{chunk-2NM363SV.js → chunk-WBKFA554.js} +10 -10
- package/dist/{chunk-R32CLGZ6.js → chunk-WCXUNS7U.js} +82 -21
- package/dist/{chunk-GPPB3JBE.js → chunk-WRBAGUNF.js} +3 -3
- package/dist/{chunk-IXJT6DCX.js → chunk-XIVNBFZS.js} +85 -30
- package/dist/{chunk-UEDMSP56.js → chunk-XPWWI35G.js} +417 -201
- package/dist/chunk-XRZT5WY5.js +47 -0
- package/dist/{chunk-3QSOM6PA.js → chunk-Y3CBHOR6.js} +2 -2
- package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
- package/dist/{chunk-AKB4GYDL.js → chunk-YQWYVTMC.js} +5 -5
- package/dist/{chunk-6I5ILFOF.js → chunk-ZA4VCIGV.js} +3 -3
- package/dist/{chunk-7OBGU7UB.js → chunk-ZDN3Y73Y.js} +12 -18
- package/dist/{chunk-3I5NY75V.js → chunk-ZWPRK62N.js} +8 -5
- package/dist/cli/index.js +41 -39
- package/dist/{clio-IT3G3VQH.js → clio-CMMK4KRR.js} +9 -9
- package/dist/{code-nav-RK6S7F6E.js → code-nav-MDZNQS33.js} +89 -21
- package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
- package/dist/{config-3QZRWZJF.js → config-SVM5P5YI.js} +131 -84
- package/dist/{configure-FL7Y3KJF.js → configure-LE3IK2TJ.js} +28 -26
- package/dist/{context-5HE7ODYK.js → context-2OHRKS42.js} +69 -64
- package/dist/{context-KYQFRVDC.js → context-E3VC7RX5.js} +15 -11
- package/dist/{context-XNHL75JV.js → context-VNCR7KAG.js} +93 -65
- package/dist/{context-clear-N545L53A.js → context-clear-BW4O37TG.js} +64 -60
- package/dist/context-map-COB37XXN.js +505 -0
- package/dist/{context-working-set-QHKXSV2F.js → context-working-set-VDS25HXZ.js} +19 -18
- package/dist/{dispatch-runner-RGIE5PCT.js → dispatch-runner-5AHT53RF.js} +93 -82
- package/dist/{docs-5NAF6AU7.js → docs-PD3EXDKU.js} +21 -20
- package/dist/{doctor-ZGPEGHIP.js → doctor-WNNVO6FY.js} +48 -47
- package/dist/{eval-GXLL44RD.js → eval-7G7SGAYO.js} +287 -115
- package/dist/{eval-inventory-HBWSWQOK.js → eval-inventory-Y6QRFOH5.js} +4 -4
- package/dist/{evidence-HWLBRH3Q.js → evidence-VD6736FQ.js} +67 -64
- package/dist/{evolve-FTZBMNVW.js → evolve-AL3NGVRL.js} +65 -62
- package/dist/{extensions-VHRBEID7.js → extensions-MOVJ32NM.js} +9 -7
- package/dist/{fleet-CKZHJWZJ.js → fleet-QZHUMAGI.js} +114 -111
- package/dist/{fleet-commands-EXDXBMV6.js → fleet-commands-BAYT5FJZ.js} +10 -10
- package/dist/{fleet-decisions-OTHB6KRL.js → fleet-decisions-IREVMRU4.js} +7 -6
- package/dist/{fleet-graph-YTEZUCUT.js → fleet-graph-YCTT3HTI.js} +22 -19
- package/dist/{fleet-inspect-SS6YMDCK.js → fleet-inspect-QVJTDAVB.js} +58 -55
- package/dist/{fleet-preflight-PBY4VYOM.js → fleet-preflight-25QAFPK4.js} +4 -4
- package/dist/{fleet-validate-KMEM5L3S.js → fleet-validate-5O57AAJ7.js} +26 -23
- package/dist/{fleet-verify-QD5M7E7Q.js → fleet-verify-CPH2W2T6.js} +59 -56
- package/dist/{fleet-view-WAMJYNDT.js → fleet-view-SWBR3VGQ.js} +58 -55
- package/dist/{init-5XQRBOFV.js → init-J477LKZH.js} +82 -79
- package/dist/{interop-34TVO25M.js → interop-3FCM6XLG.js} +11 -11
- package/dist/{library-3QY6KF57.js → library-QUQEIUG6.js} +30 -27
- package/dist/{memory-L4UTIIIW.js → memory-SGGSEP65.js} +67 -64
- package/dist/{models-ZVX3QOWE.js → models-HEKUAXXK.js} +53 -46
- package/dist/{monitor-CEKVSYTS.js → monitor-HKU57TYQ.js} +63 -60
- package/dist/{orchestrator-77BAP6BC.js → orchestrator-VDFAEFAI.js} +1831 -1057
- package/dist/{panes-7STHOAUJ.js → panes-DN2SSFOH.js} +5 -5
- package/dist/{panes-SHAUIRXY.js → panes-TALGNPZT.js} +29 -14
- package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
- package/dist/reset-EAJFFJVB.js +344 -0
- package/dist/{resources-74GKTLSF.js → resources-OVKSEFVE.js} +29 -20
- package/dist/{run-HBAUJNNZ.js → run-7DP7ZF2J.js} +120 -115
- package/dist/{share-G3APVLVP.js → share-WML67FT3.js} +32 -27
- package/dist/{skills-35HHUKCR.js → skills-SG662R2K.js} +41 -31
- package/dist/{skills-eval-QN4HSHDC.js → skills-eval-VVZEUU46.js} +78 -77
- package/dist/{skills-inventory-J357J34F.js → skills-inventory-I2E23GET.js} +23 -20
- package/dist/{slash-commands-JZZCQA32.js → slash-commands-S7MBJDQK.js} +40 -36
- package/dist/{steer-XAVHJM22.js → steer-2LQOMCPB.js} +3 -3
- package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
- package/dist/{targets-DSM6CY3M.js → targets-4QC3HIEW.js} +54 -54
- package/dist/{terminal-lease-JOPFUVEM.js → terminal-lease-TUHIJ6Y2.js} +5 -5
- package/dist/{tools-MKNWVPBH.js → tools-TFGJICCU.js} +10 -10
- package/dist/{trace-ECQ7TIYZ.js → trace-FXMXUZUF.js} +55 -7
- package/dist/uninstall-5PEVOE5B.js +408 -0
- package/dist/upgrade-M4WXY6KN.js +303 -0
- package/dist/{usage-X52N3IDJ.js → usage-N7ZNVLEM.js} +151 -104
- package/dist/{verifiers-EJTVVSMA.js → verifiers-DJTP4XX6.js} +15 -15
- package/dist/{verify-YJL6XET2.js → verify-RWE4PPEK.js} +9 -9
- package/dist/{web-fetch-MPIFL3LL.js → web-fetch-MPARV2K7.js} +2 -2
- package/dist/{wiki-generate-4NDZTQ4B.js → wiki-generate-C7IQOXSP.js} +89 -86
- package/dist/{with-panes-OBOBFIIR.js → with-panes-4GCGSL7J.js} +53 -257
- package/dist/worker/entry.js +90 -74
- package/docs/README.md +176 -81
- package/docs/{acp.md → architecture/acp.md} +36 -20
- package/docs/{alcf-provider.md → architecture/alcf-provider.md} +8 -5
- package/docs/{architecture.md → architecture/architecture.md} +43 -22
- package/docs/{artifact-placement.md → architecture/artifact-placement.md} +27 -23
- package/docs/architecture/artifact-versions.md +90 -0
- package/docs/{capacity-and-scheduling.md → architecture/capacity-and-scheduling.md} +26 -13
- package/docs/{context-engine.md → architecture/context-engine.md} +29 -25
- package/docs/{context-working-set.md → architecture/context-working-set.md} +13 -10
- package/docs/{dispatch-architecture-rationale.md → architecture/dispatch-architecture-rationale.md} +12 -9
- package/docs/{dispatch-typed-intent.md → architecture/dispatch-typed-intent.md} +68 -46
- package/docs/{evidence-and-memory.md → architecture/evidence-and-memory.md} +23 -16
- package/docs/{middleware-and-components.md → architecture/middleware-and-components.md} +11 -5
- package/docs/{model-catalog.md → architecture/model-catalog.md} +61 -27
- package/docs/{observability.md → architecture/observability.md} +38 -14
- package/docs/{pi-boundary.md → architecture/pi-boundary.md} +24 -11
- package/docs/{prompt-envelope-and-tools.md → architecture/prompt-envelope-and-tools.md} +57 -20
- package/docs/{provider-adapter-cookbook.md → architecture/provider-adapter-cookbook.md} +99 -25
- package/docs/{safety-model.md → architecture/safety-model.md} +35 -20
- package/docs/{session-lifecycle.md → architecture/session-lifecycle.md} +8 -5
- package/docs/architecture/time-conventions.md +125 -0
- package/docs/{trace-store.md → architecture/trace-store.md} +13 -5
- package/docs/{tui-design.md → architecture/tui-design.md} +13 -13
- package/docs/{worker-dispatch-mechanics.md → architecture/worker-dispatch-mechanics.md} +27 -30
- package/docs/{built-in-agents.md → guide/built-in-agents.md} +65 -35
- package/docs/{commands-and-modes.md → guide/commands-and-modes.md} +66 -61
- package/docs/{configuration-and-targets.md → guide/configuration-and-targets.md} +323 -297
- package/docs/guide/configuration-reference.md +1163 -0
- package/docs/{environment-variables.md → guide/environment-variables.md} +33 -28
- package/docs/{exit-codes-and-output.md → guide/exit-codes-and-output.md} +6 -3
- package/docs/{extensions-and-sharing.md → guide/extensions-and-sharing.md} +41 -14
- package/docs/{fleet-dispatch.md → guide/fleet-dispatch.md} +39 -43
- package/docs/{glossary.md → guide/glossary.md} +14 -11
- package/docs/{installation-and-lifecycle.md → guide/installation-and-lifecycle.md} +81 -17
- package/docs/guide/panes-and-files.md +290 -0
- package/docs/{proactive-memory.md → guide/proactive-memory.md} +131 -107
- package/docs/{resource-library.md → guide/resource-library.md} +13 -4
- package/docs/{skills-marketplace.md → guide/skills-marketplace.md} +25 -3
- package/docs/{tool-usage.md → guide/tool-usage.md} +87 -23
- package/docs/{troubleshooting.md → guide/troubleshooting.md} +9 -4
- package/docs/{config-knobs-audit.md → history/config-knobs-audit.md} +11 -11
- package/docs/{release-cut-checklist.md → history/release-cut-checklist.md} +29 -2
- package/docs/process/development-pipeline.md +152 -0
- package/docs/process/documentation-coverage.md +100 -0
- package/docs/process/documentation-guide.md +187 -0
- package/docs/{eval-runner.md → process/eval-runner.md} +108 -53
- package/docs/{evals-internal.md → process/evals-internal.md} +10 -10
- package/docs/{evolution.md → process/evolution.md} +2 -2
- package/docs/{fleet-demo-runbook.md → process/fleet-demo-runbook.md} +11 -7
- package/docs/{git-commit-provenance.md → process/git-commit-provenance.md} +11 -4
- package/docs/{performance-methodology.md → process/performance-methodology.md} +87 -69
- package/docs/{scientific-validation.md → process/scientific-validation.md} +4 -4
- package/evals/README.md +2 -2
- package/evals/behavioral-model.yaml +3 -2
- package/package.json +10 -8
- package/skills/README.md +52 -41
- package/skills/coding/ast-grep/SKILL.md +102 -31
- package/skills/coding/ast-grep/evals.md +26 -0
- package/skills/coding/coding-standards/SKILL.md +41 -6
- package/skills/coding/coding-standards/evals.md +23 -0
- package/skills/coding/prototype/SKILL.md +88 -29
- package/skills/coding/prototype/evals.md +19 -0
- package/skills/coding/tdd/SKILL.md +81 -54
- package/skills/coding/tdd/evals.md +20 -0
- package/skills/context/context-handoff/SKILL.md +44 -3
- package/skills/context/context-handoff/evals.md +44 -0
- package/skills/context/context-prime/SKILL.md +46 -16
- package/skills/context/context-prime/evals.md +45 -0
- package/skills/git/branch-closeout/SKILL.md +132 -0
- package/skills/git/branch-closeout/evals.md +133 -0
- package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
- package/skills/git/file-ticket/SKILL.md +78 -64
- package/skills/git/file-ticket/assets/issue-template.md +22 -0
- package/skills/git/file-ticket/evals.md +31 -26
- package/skills/git/file-ticket/references/issue-discovery.md +49 -0
- package/skills/git/fix-issue/SKILL.md +88 -65
- package/skills/git/fix-issue/evals.md +35 -31
- package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +101 -52
- package/skills/git/resolve-merge-conflicts/evals.md +52 -25
- package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
- package/skills/git/ship/SKILL.md +103 -67
- package/skills/git/ship/assets/pr-template.md +21 -0
- package/skills/git/ship/evals.md +44 -28
- package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
- package/skills/git/worktree-create/SKILL.md +80 -50
- package/skills/git/worktree-create/evals.md +40 -33
- package/skills/git/worktree-create/references/worktree-setup.md +62 -66
- package/skills/git/worktree-merge/SKILL.md +112 -65
- package/skills/git/worktree-merge/evals.md +42 -34
- package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
- package/skills/meta/clio-coder-dev/SKILL.md +9 -5
- package/skills/meta/clio-coder-dev/evals.md +3 -2
- package/skills/meta/clio-coder-test/SKILL.md +102 -95
- package/skills/meta/clio-coder-test/evals.md +9 -4
- package/skills/meta/clio-coder-test/references/harness.md +100 -124
- package/skills/meta/clio-coder-test/references/test-map.md +77 -50
- package/skills/meta/credentials/SKILL.md +2 -2
- package/skills/meta/find-skills/SKILL.md +2 -2
- package/skills/meta/herdr/SKILL.md +2 -2
- package/skills/meta/skill-craft/SKILL.md +22 -16
- package/skills/planning/archify/SKILL.md +196 -0
- package/skills/planning/archify/evals.md +65 -0
- package/skills/planning/architecture/SKILL.md +62 -13
- package/skills/planning/architecture/evals.md +65 -0
- package/skills/planning/backlog/SKILL.md +131 -15
- package/skills/planning/backlog/evals.md +142 -0
- package/skills/planning/prd/SKILL.md +47 -7
- package/skills/planning/prd/evals.md +54 -0
- package/skills/planning/product-intent/SKILL.md +58 -3
- package/skills/planning/product-intent/evals.md +70 -0
- package/skills/planning/tech-spec/SKILL.md +54 -3
- package/skills/planning/tech-spec/evals.md +73 -0
- package/skills/registry.yaml +70 -62
- package/skills/remote.yaml +13 -0
- package/skills/research/arxiv-literature/SKILL.md +77 -19
- package/skills/research/arxiv-literature/evals.md +50 -0
- package/skills/research/experiment-protocol/SKILL.md +21 -2
- package/skills/research/experiment-protocol/evals.md +23 -0
- package/skills/research/scientific-debugging/SKILL.md +24 -2
- package/skills/research/scientific-debugging/evals.md +18 -0
- package/skills/research/scientific-modernization/SKILL.md +27 -2
- package/skills/research/scientific-modernization/evals.md +27 -0
- package/skills/skill-marketplace.json +97 -62
- package/skills/workflow/cut-it/SKILL.md +66 -6
- package/skills/workflow/cut-it/evals.md +101 -0
- package/skills/workflow/design-council/SKILL.md +118 -28
- package/skills/workflow/design-council/evals.md +161 -0
- package/skills/workflow/grill-me/SKILL.md +87 -11
- package/skills/workflow/grill-me/evals.md +153 -0
- package/skills/workflow/workflow-distiller/SKILL.md +77 -18
- package/skills/workflow/workflow-distiller/evals.md +118 -0
- package/src/cli/args.ts +2 -2
- package/src/cli/bootstrap-generate.ts +1 -1
- package/src/cli/config-inspect.ts +65 -12
- package/src/cli/configure-interop.ts +105 -13
- package/src/cli/configure-oauth.ts +57 -0
- package/src/cli/configure-onboarding.ts +980 -0
- package/src/cli/configure-target.ts +594 -0
- package/src/cli/configure.ts +1082 -532
- package/src/cli/context-map.ts +114 -0
- package/src/cli/context.ts +4 -0
- package/src/cli/docs.ts +22 -14
- package/src/cli/doctor-naming.ts +5 -5
- package/src/cli/doctor-toolchain.ts +3 -3
- package/src/cli/eval.ts +1 -2
- package/src/cli/extensions.ts +2 -1
- package/src/cli/fleet.ts +1 -1
- package/src/cli/index.ts +3 -1
- package/src/cli/internal-dispatch.ts +3 -4
- package/src/cli/lifecycle-presenter.ts +436 -0
- package/src/cli/models.ts +10 -2
- package/src/cli/modes/print.ts +5 -1
- package/src/cli/panes.ts +19 -5
- package/src/cli/reset.ts +228 -106
- package/src/cli/run.ts +9 -4
- package/src/cli/select.ts +664 -0
- package/src/cli/share.ts +5 -1
- package/src/cli/skills-eval.ts +3 -3
- package/src/cli/skills.ts +9 -2
- package/src/cli/targets.ts +5 -6
- package/src/cli/trace.ts +55 -4
- package/src/cli/uninstall.ts +233 -165
- package/src/cli/upgrade.ts +204 -149
- package/src/cli/usage.ts +86 -27
- package/src/cli/validate-model.ts +3 -3
- package/src/cli/wiki-generate.ts +1 -1
- package/src/core/artifact-paths.ts +1 -1
- package/src/core/bash-exec.ts +131 -86
- package/src/core/bus-events.ts +51 -6
- package/src/core/config.ts +61 -1
- package/src/core/defaults.ts +7 -4
- package/src/core/dispatch-outcome.ts +16 -0
- package/src/core/external-diagnostic.ts +44 -0
- package/src/core/gateway-routing.ts +157 -0
- package/src/core/guardrails.ts +10 -49
- package/src/core/prompt-hint.ts +9 -0
- package/src/core/safe-exec.ts +17 -2
- package/src/core/skill-activation.ts +89 -2
- package/src/domains/agents/builtins/architect.md +2 -3
- package/src/domains/agents/builtins/coder.md +3 -2
- package/src/domains/agents/builtins/debugger.md +2 -2
- package/src/domains/agents/builtins/documenter.md +2 -2
- package/src/domains/agents/builtins/git-master.md +1 -1
- package/src/domains/agents/builtins/oracle.md +1 -1
- package/src/domains/agents/builtins/provenance.md +1 -1
- package/src/domains/agents/builtins/researcher.md +1 -1
- package/src/domains/agents/builtins/scout.md +1 -1
- package/src/domains/agents/builtins/tester.md +2 -2
- package/src/domains/agents/builtins/verifier.md +2 -2
- package/src/domains/agents/builtins/wiki-writer.md +1 -1
- package/src/domains/agents/builtins/world-knowledge.md +31 -0
- package/src/domains/agents/catalog.ts +13 -15
- package/src/domains/agents/contract.ts +2 -0
- package/src/domains/agents/extension.ts +23 -1
- package/src/domains/agents/result-contract.ts +70 -0
- package/src/domains/config/keybindings.ts +8 -0
- package/src/domains/context/extension.ts +0 -3
- package/src/domains/context/wiki/map-seed.ts +589 -0
- package/src/domains/context/wiki/plan.ts +2 -2
- package/src/domains/context/working-set/path-index.ts +1 -0
- package/src/domains/dispatch/admission.ts +29 -0
- package/src/domains/dispatch/agent-candidates.ts +10 -0
- package/src/domains/dispatch/budget-envelope.ts +86 -1
- package/src/domains/dispatch/capability-match.ts +11 -0
- package/src/domains/dispatch/capacity-lease.ts +17 -0
- package/src/domains/dispatch/contract.ts +11 -1
- package/src/domains/dispatch/extension.ts +237 -49
- package/src/domains/dispatch/host-verification.ts +435 -39
- package/src/domains/dispatch/intent-requirements.ts +10 -0
- package/src/domains/dispatch/intent.ts +18 -1
- package/src/domains/dispatch/path-scope.ts +235 -24
- package/src/domains/dispatch/run-event-journal.ts +4 -15
- package/src/domains/dispatch/state.ts +2 -3
- package/src/domains/dispatch/transport.ts +45 -21
- package/src/domains/dispatch/types.ts +58 -3
- package/src/domains/dispatch/worker-model-metadata.ts +38 -0
- package/src/domains/eval/artifacts/store.ts +5 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
- package/src/domains/eval/metrics/token-stream.ts +201 -31
- package/src/domains/eval/metrics/tracked.ts +40 -4
- package/src/domains/eval/runners/clio-run.ts +5 -2
- package/src/domains/eval/schema/suite.ts +28 -0
- package/src/domains/eval/schema/verdict.ts +2 -2
- package/src/domains/eval/store.ts +8 -1
- package/src/domains/eval/suites/resolve.ts +13 -1
- package/src/domains/eval/suites/run.ts +24 -3
- package/src/domains/evidence/trust-status.ts +10 -1
- package/src/domains/extensions/contract.ts +15 -1
- package/src/domains/extensions/discovery.ts +238 -41
- package/src/domains/extensions/extension.ts +105 -6
- package/src/domains/extensions/index.ts +24 -0
- package/src/domains/extensions/integrity.ts +189 -0
- package/src/domains/extensions/manager.ts +17 -1
- package/src/domains/extensions/resource-path.ts +27 -0
- package/src/domains/extensions/resources.ts +18 -38
- package/src/domains/extensions/snapshot-store.ts +39 -0
- package/src/domains/extensions/snapshot.ts +180 -0
- package/src/domains/extensions/state.ts +385 -57
- package/src/domains/extensions/types.ts +118 -1
- package/src/domains/interop/registry.ts +6 -2
- package/src/domains/interop/types.ts +4 -0
- package/src/domains/lifecycle/migrations/2026-09-01-extension-install-digests.ts +27 -0
- package/src/domains/lifecycle/migrations/index.ts +6 -0
- package/src/domains/lifecycle/naming-resources.ts +19 -4
- package/src/domains/lifecycle/naming-yazi.ts +10 -5
- package/src/domains/memory/task-memory-policy.ts +70 -26
- package/src/domains/memory/task-memory-telemetry.ts +1 -0
- package/src/domains/middleware/contract.ts +26 -0
- package/src/domains/middleware/extension.ts +24 -24
- package/src/domains/middleware/hook-receipts.ts +27 -4
- package/src/domains/middleware/hooks-io.ts +65 -32
- package/src/domains/middleware/hooks.ts +64 -0
- package/src/domains/middleware/index.ts +28 -5
- package/src/domains/middleware/marketplace-offer.ts +3 -35
- package/src/domains/middleware/memory-intervention.ts +127 -32
- package/src/domains/middleware/memory-step-endpoint.ts +3 -2
- package/src/domains/middleware/registrations.ts +326 -0
- package/src/domains/middleware/runtime.ts +28 -0
- package/src/domains/middleware/skills-reminder.ts +31 -2
- package/src/domains/middleware/snapshot.ts +20 -7
- package/src/domains/mux/contract.ts +38 -0
- package/src/domains/mux/detect.ts +6 -13
- package/src/domains/mux/index.ts +1 -1
- package/src/domains/mux/operations.ts +44 -5
- package/src/domains/mux/yazi/assets/yazi.toml +2 -2
- package/src/domains/mux/yazi/session.ts +53 -4
- package/src/domains/mux/yazi/theme.ts +117 -17
- package/src/domains/observability/compaction-usage.ts +118 -0
- package/src/domains/observability/contract.ts +10 -11
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/extension.ts +17 -4
- package/src/domains/observability/out-of-turn-usage.ts +52 -21
- package/src/domains/observability/projection.ts +14 -90
- package/src/domains/observability/trace-store.ts +43 -7
- package/src/domains/prompts/compiler.ts +73 -53
- package/src/domains/prompts/contract.ts +15 -3
- package/src/domains/prompts/extension.ts +97 -9
- package/src/domains/prompts/fragments/identity/clio-worker.md +1 -3
- package/src/domains/prompts/fragments/identity/clio.md +6 -12
- package/src/domains/prompts/fragments/identity/docs-routing.md +1 -2
- package/src/domains/prompts/fragments/identity/self-awareness.md +3 -11
- package/src/domains/prompts/fragments/operating/contract.md +7 -15
- package/src/domains/prompts/fragments/operating/delegation.md +32 -34
- package/src/domains/prompts/fragments/operating/skills.md +10 -24
- package/src/domains/prompts/fragments/operating/worker.md +1 -8
- package/src/domains/providers/contract.ts +4 -1
- package/src/domains/providers/extension.ts +40 -9
- package/src/domains/providers/index.ts +1 -1
- package/src/domains/providers/model-capabilities.ts +9 -0
- package/src/domains/providers/model-discovery.ts +2 -0
- package/src/domains/providers/model-runtime-capabilities.ts +99 -25
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +699 -114
- package/src/domains/providers/runtime-resolution.ts +31 -0
- package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
- package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +7 -2
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +9 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
- package/src/domains/providers/support.ts +11 -5
- package/src/domains/providers/target-model-cache.ts +25 -2
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/cost-provenance.ts +19 -0
- package/src/domains/providers/types/local-model-quirks.ts +85 -37
- package/src/domains/providers/types/runtime-descriptor.ts +20 -1
- package/src/domains/providers/types/target-descriptor.ts +19 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/skills/install.ts +72 -7
- package/src/domains/resources/skills/loader.ts +23 -19
- package/src/domains/resources/skills/marketplace.ts +63 -11
- package/src/domains/safety/autonomy.ts +15 -0
- package/src/domains/safety/call-target.ts +1 -1
- package/src/domains/safety/index.ts +1 -0
- package/src/domains/safety/loop-detector.ts +7 -4
- package/src/domains/safety/path-policy.ts +1 -1
- package/src/domains/safety/policy-engine.ts +34 -11
- package/src/domains/safety/protected-artifacts.ts +191 -88
- package/src/domains/safety/run-effects.ts +2 -22
- package/src/domains/safety/skill-authority.ts +55 -0
- package/src/domains/session/compaction/compact.ts +72 -22
- package/src/domains/session/entries.ts +6 -0
- package/src/domains/session/task-board.ts +10 -9
- package/src/domains/session/usage.ts +3 -3
- package/src/domains/share/archive.ts +164 -7
- package/src/engine/acp/server.ts +62 -9
- package/src/engine/agent.ts +13 -3
- package/src/engine/ai.ts +26 -8
- package/src/engine/antigravity/subprocess-runtime.ts +386 -120
- package/src/engine/api-registry.ts +3 -0
- package/src/engine/apis/llamacpp-residency.ts +3 -4
- package/src/engine/apis/lmstudio.ts +3 -3
- package/src/engine/apis/ollama-native.ts +6 -6
- package/src/engine/apis/openai-completions.ts +145 -39
- package/src/engine/apis/output-budget.ts +8 -18
- package/src/engine/apis/residency.ts +8 -27
- package/src/engine/external-subprocess.ts +114 -6
- package/src/engine/gemma-channel-filter.ts +19 -0
- package/src/engine/loop-guard.ts +92 -12
- package/src/engine/worker-runtime.ts +40 -11
- package/src/engine/worker-tools.ts +3 -1
- package/src/entry/background-model-metadata.ts +18 -0
- package/src/entry/compaction-prompt.ts +57 -0
- package/src/entry/extension-hook-sources.ts +28 -0
- package/src/entry/extension-reload.ts +309 -0
- package/src/entry/orchestrator.ts +464 -251
- package/src/entry/task-memory-lifecycle.ts +35 -0
- package/src/interactive/application-controller.ts +2 -1
- package/src/interactive/bus-notices.ts +8 -1
- package/src/interactive/chat-loop-messages.ts +16 -17
- package/src/interactive/chat-loop.ts +75 -3
- package/src/interactive/chat-panel.ts +36 -13
- package/src/interactive/chat-renderer.ts +72 -7
- package/src/interactive/cost-overlay.ts +26 -2
- package/src/interactive/dispatch-board.ts +6 -11
- package/src/interactive/footer/widgets.ts +13 -0
- package/src/interactive/interactive-application.ts +39 -4
- package/src/interactive/interactive-input-runtime.ts +4 -0
- package/src/interactive/interactive-presentation.ts +2 -2
- package/src/interactive/interactive-slash-runtime.ts +4 -1
- package/src/interactive/overlays/extensions.ts +9 -1
- package/src/interactive/overlays/help-reference.ts +13 -0
- package/src/interactive/overlays/settings.ts +27 -16
- package/src/interactive/panes-runtime.ts +111 -35
- package/src/interactive/prompt-cache-identity.ts +88 -0
- package/src/interactive/renderers/worker-entry.ts +32 -0
- package/src/interactive/slash-commands.ts +153 -20
- package/src/interactive/stream-pacing-policy.ts +0 -23
- package/src/interactive/theme/labels.ts +19 -13
- package/src/interactive/turn-context.ts +39 -20
- package/src/interactive/turn-recovery.ts +8 -0
- package/src/interactive/turn-runtime.ts +27 -11
- package/src/interactive/turn-state.ts +7 -0
- package/src/interactive/worker-receipts.ts +1 -0
- package/src/interactive/worker-stream.ts +6 -1
- package/src/interactive/yazi-bridge.ts +60 -6
- package/src/tools/agent-tools.ts +30 -1
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +3 -3
- package/src/tools/bash.ts +1 -1
- package/src/tools/bootstrap.ts +4 -0
- package/src/tools/builtin-tool-catalog.ts +52 -22
- package/src/tools/codewiki/code-nav-surface.ts +6 -0
- package/src/tools/codewiki/code-nav.ts +99 -13
- package/src/tools/context/docs-engine.ts +20 -7
- package/src/tools/context/index.ts +59 -21
- package/src/tools/core-bootstrap.ts +28 -6
- package/src/tools/credential-present.ts +1 -2
- package/src/tools/dispatch-arguments.ts +6 -1
- package/src/tools/dispatch-event-text.ts +10 -0
- package/src/tools/dispatch-plan.ts +49 -4
- package/src/tools/dispatch-run-events.ts +1 -1
- package/src/tools/dispatch-runner.ts +12 -0
- package/src/tools/dispatch-schema.ts +338 -0
- package/src/tools/dispatch-types.ts +3 -0
- package/src/tools/dispatch.ts +9 -254
- package/src/tools/ledger.ts +3 -5
- package/src/tools/monitor-surface.ts +5 -13
- package/src/tools/observation.ts +4 -5
- package/src/tools/panes-surface.ts +4 -11
- package/src/tools/panes.ts +4 -2
- package/src/tools/policy.ts +15 -2
- package/src/tools/read.ts +5 -6
- package/src/tools/registry.ts +41 -12
- package/src/tools/result-shaping.ts +18 -14
- package/src/tools/steer-surface.ts +1 -1
- package/src/tools/tasks.ts +1 -1
- package/src/tools/truncate.ts +6 -5
- package/src/tools/verify/surface.ts +6 -12
- package/src/tools/web-fetch-surface.ts +1 -3
- package/src/tools/worker-evidence.ts +3 -1
- package/src/worker/spec-contract.ts +4 -0
- package/dist/builtins-UJLMOVOV.js +0 -17
- package/dist/chunk-5QIAJV2D.js +0 -48
- package/dist/chunk-JZWT5J3Y.js +0 -814
- package/dist/chunk-K7VKOLQQ.js +0 -15
- package/dist/chunk-PMZCIOCJ.js +0 -25
- package/dist/chunk-SUW5DORT.js +0 -819
- package/dist/chunk-UOV2BYIW.js +0 -107
- package/dist/chunk-WR6U3OVP.js +0 -45
- package/dist/chunk-Y45G3AXC.js +0 -1558
- package/dist/reset-EOLM7GVE.js +0 -230
- package/dist/uninstall-N34PCTGJ.js +0 -331
- package/dist/upgrade-H7TOM7YL.js +0 -323
- package/docs/artifact-versions.md +0 -67
- package/docs/development-pipeline.md +0 -121
- package/docs/documentation-coverage.md +0 -46
- package/docs/documentation-guide.md +0 -167
- package/docs/time-conventions.md +0 -101
|
@@ -47,3 +47,57 @@ Expected:
|
|
|
47
47
|
|
|
48
48
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
49
49
|
(30B local, llamacpp on mini), full-auto sandbox. WEAK PASS. Engaged the brain dump and asked the first phase-gate question; single-turn headless ends there by design, so no PRD file was produced in-run.
|
|
50
|
+
|
|
51
|
+
## Battletest record (2026-09-03)
|
|
52
|
+
|
|
53
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_prd.py`, continuing the
|
|
54
|
+
planning category's shared HPC log-triage domain from `product-intent`.
|
|
55
|
+
Seeds the actual `docs/hpc-log-triage.prd.md` product-intent output, its two
|
|
56
|
+
evidence docs, and a partial codebase (`src/scanner.py`: a working
|
|
57
|
+
`FailureEvent` + `scan_oom`, OOM only — ECC/Xid not yet implemented) inside a
|
|
58
|
+
git repo. The brain-dump prompt names ten scope-creep features (dashboard,
|
|
59
|
+
Slack, always-on pipeline, auto-remediation, learned ranking, federation,
|
|
60
|
+
audit export, RBAC, mobile app) and one explicit one-way-door tension
|
|
61
|
+
(on-demand reads vs. an always-on ingestion pipeline), combining S1
|
|
62
|
+
(existing foundation, stack detection), S2 (scope honesty), and S3 (existing
|
|
63
|
+
foundation reuse) into one gradable run. Graded 14 checks against real
|
|
64
|
+
post-run disk state (`PRD.md` at the exact promised path, all eight required
|
|
65
|
+
sections, the out-of-scope section itself — not just anywhere in the
|
|
66
|
+
document — actually containing the pushed-out features, `FailureEvent`
|
|
67
|
+
reused rather than re-specced, ≥2 self-contained milestone prompts with no
|
|
68
|
+
"see PRD" phrase) plus the reconstructed final assistant text and the raw
|
|
69
|
+
JSONL's tool-call/safety-block stream.
|
|
70
|
+
|
|
71
|
+
| run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
|
|
72
|
+
|---|---|---|---|---|---|---|---|
|
|
73
|
+
| baseline (no skill) | qwen3.8-27b | 311s | 11 | 283.0k / 25.4k | 0 | 1/14 | never invoked `/skill prd`; misread the brain dump as an architecture request (it saw the installed `architecture` skill via `context(scope="skills")`) and wrote `docs/architecture-log-triage-v1.md` instead — no `PRD.md`, no milestones |
|
|
74
|
+
| v1 (frozen 0.3.0) | qwen3.8-27b | 492s | 12 | 408.3k / 43.4k | 2 | 12/14 | correct `PRD.md` + 5 self-contained milestone prompts, honest out-of-scope, reused `FailureEvent`; on its own initiative noticed no `ask_user` tool was present and ran the full nine-phase loop as a monologue, recording each lock — but opened a `tasks` plan and one `bash` call, both refused by the narrowed surface (self-recovered, but the two-safety-block outcome is exactly what an explicit refusal line prevents) |
|
|
75
|
+
| v2 (live 0.4.0) | qwen3.8-27b | 299s | 9 | 197.4k / 22.9k | 0 | 14/14 | same correctness as v1, zero safety blocks, no `tasks`/`bash` calls at all; final reply names the monologue explicitly ("no `ask_user` tool exists in my surface, so every gate was run as the skill's assumed-confirm monologue") |
|
|
76
|
+
| v2 confirm | ornith-1.5-35b-a3b | 79s | 11 | 169.9k / 11.5k | 0 | 14/14 | fastest of the four runs by a wide margin, same shape and grounding, zero safety blocks |
|
|
77
|
+
|
|
78
|
+
**Changes**: (1) `## Arguments` contract with an explicit headless/no-operator
|
|
79
|
+
rule — every one of the nine phases runs as an assumed-confirm monologue
|
|
80
|
+
when no one answers a gate, not just the first one, matching the pattern
|
|
81
|
+
ported from `architecture`/`product-intent`; (2) an explicit `tasks` and
|
|
82
|
+
`bash` refusal line — these were v1's only two failures, both self-recovered
|
|
83
|
+
by this model but a real safety-block pair on a weaker or more literal one;
|
|
84
|
+
(3) "Read the repo before asking" now says explicitly that an existing
|
|
85
|
+
entity gets reused and marked, not re-specced, closing S3; (4) the
|
|
86
|
+
`PRD.md` line now names the wrong shapes to avoid (`docs/PRD.md`, a slugged
|
|
87
|
+
filename, a report-style name), mirroring `architecture`'s
|
|
88
|
+
`final_report.md` fix; (5) three new Red flags for the failures actually
|
|
89
|
+
observed: an unconfirmed phase left that way instead of run as the
|
|
90
|
+
monologue, an existing module re-specced as new, and the `bash`/`tasks`
|
|
91
|
+
refusals named explicitly.
|
|
92
|
+
|
|
93
|
+
**Still weak**: the baseline's failure mode (skipping the skill entirely and
|
|
94
|
+
misreading the task as an architecture request) is a skill-selection gap
|
|
95
|
+
this SKILL.md cannot fix from inside its own body — it only activates once
|
|
96
|
+
invoked. Only the combined S1+S2+S3 fixture ran; a plain "just write it,
|
|
97
|
+
no interview" decline path and a genuinely blank invocation weren't tested
|
|
98
|
+
standalone. `ask_user` was never actually called on either model tested —
|
|
99
|
+
both recognized the headless gap and went straight to the monologue without
|
|
100
|
+
attempting the tool first, so the explicit degradation prose is a defensive
|
|
101
|
+
addition, not a proven repro-then-fix (the same caveat the context category
|
|
102
|
+
noted for its own headless guidance). `code_nav` (in allowed-tools) was
|
|
103
|
+
never exercised. Only 27–35B class models tried, no small-model run.
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: product-intent
|
|
3
|
-
description:
|
|
3
|
+
description: "Writes a problem-first product document for a greenfield effort: interviews for the problem, evidence, and a falsifiable hypothesis, with zero engineering decisions. Not for engineering decisions; use architecture. Not for turning a locked idea into milestone build prompts; use prd."
|
|
4
4
|
triggers:
|
|
5
5
|
- why are we building this
|
|
6
6
|
- write the product thesis
|
|
7
7
|
- problem-first PRD
|
|
8
8
|
- define a falsifiable product hypothesis
|
|
9
9
|
- greenfield product intent
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.4.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
allowed-tools:
|
|
13
13
|
- read
|
|
@@ -37,6 +37,35 @@ team can challenge before building and judge after shipping. Engineering
|
|
|
37
37
|
decisions (library, data model, boundaries) never enter it; they belong to
|
|
38
38
|
`architecture`.
|
|
39
39
|
|
|
40
|
+
## Arguments
|
|
41
|
+
|
|
42
|
+
```text
|
|
43
|
+
/skill product-intent <idea or problem, in a few sentences>
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
- The text is the raw idea, problem statement, or "just write it" request
|
|
47
|
+
that starts Step 0. Reference docs (interviews, tickets, analytics,
|
|
48
|
+
competitor notes) named or pathed in the request are evidence to read
|
|
49
|
+
first, not more arguments.
|
|
50
|
+
- Nothing is required beyond some text; a blank invocation gets Step 0's own
|
|
51
|
+
"What do you want to build? A few sentences." question.
|
|
52
|
+
|
|
53
|
+
There is no operator in a headless run: `ask_user` still executes, but with
|
|
54
|
+
nothing to answer it every call returns immediately with no answers, every
|
|
55
|
+
time — calling it again will not produce a different result. Treat the
|
|
56
|
+
first empty response exactly like the user saying "just write it" (see "If
|
|
57
|
+
the user declines the interview" below), and apply that treatment from
|
|
58
|
+
wherever it happened onward — Step 0's evidence check included, not just
|
|
59
|
+
the five clusters: state the question, your best evidence-grounded answer
|
|
60
|
+
(or, absent evidence, the most defensible product default) and the
|
|
61
|
+
reasoning, mark it `assumed — confirm`, and move to the next step. Never
|
|
62
|
+
invent evidence to back an assumption; one with nothing behind it stays an
|
|
63
|
+
open question, not a fact.
|
|
64
|
+
|
|
65
|
+
The interview clusters below are the plan; do not open a task list for
|
|
66
|
+
them. `tasks` sits outside this skill's tool surface and any call to it is
|
|
67
|
+
refused.
|
|
68
|
+
|
|
40
69
|
Two hard guards, checked before writing anything:
|
|
41
70
|
|
|
42
71
|
1. **Intent-framed.** If only one solution could fit your problem statement,
|
|
@@ -59,7 +88,10 @@ the same turn. Thin answers get reflected back and dug into.
|
|
|
59
88
|
|
|
60
89
|
If the user declines the interview ("just write it"): honor it, name what
|
|
61
90
|
you will have to leave TBD, ask only the two or three highest-leverage
|
|
62
|
-
questions, and mark everything else "TBD — needs validation".
|
|
91
|
+
questions, and mark everything else "TBD — needs validation". This is also
|
|
92
|
+
the headless default: see Arguments above for what an empty `ask_user`
|
|
93
|
+
response means and how to apply this same treatment cluster by cluster
|
|
94
|
+
instead of stopping after the first one.
|
|
63
95
|
|
|
64
96
|
1. **Initiate.** Input given → restate and confirm. Blank → "What do you
|
|
65
97
|
want to build? A few sentences." GATE.
|
|
@@ -116,3 +148,26 @@ offered: `architecture` for the engineering decisions this PRD
|
|
|
116
148
|
deliberately left open. Failing any of the five tests below means not done:
|
|
117
149
|
evidence-grounded problem · hypothesis with separate RIGHT and WRONG ·
|
|
118
150
|
outcome-shaped metrics · explicit non-goals · zero engineering decisions.
|
|
151
|
+
|
|
152
|
+
## Red flags
|
|
153
|
+
|
|
154
|
+
- A stack, library, database, or framework name anywhere in the document —
|
|
155
|
+
"React + Postgres" appearing at all is an instant fail; that decision
|
|
156
|
+
belongs to `architecture`, not here.
|
|
157
|
+
- A hypothesis with a RIGHT condition and no WRONG condition, or a WRONG
|
|
158
|
+
condition that is just the RIGHT one negated instead of a real
|
|
159
|
+
counter-signal.
|
|
160
|
+
- The literal filename `PRD.md`, or anything outside `docs/`, instead of
|
|
161
|
+
`docs/<kebab-slug>.prd.md`.
|
|
162
|
+
- Calling `ask_user` again after an empty response, instead of switching to
|
|
163
|
+
the decline treatment for every step from there on.
|
|
164
|
+
- Opening a task list for the interview clusters; `tasks` is refused.
|
|
165
|
+
- Reaching Generate without ever attempting Step 0 or the first cluster —
|
|
166
|
+
the decline/headless treatment is a fallback for a gate that ran and came
|
|
167
|
+
back empty, not a license to skip the loop from the start.
|
|
168
|
+
- Claims in the document that trace to neither the seeded evidence nor a
|
|
169
|
+
marked assumption — an invented fact reads as confident and is the
|
|
170
|
+
hardest failure to catch after the fact.
|
|
171
|
+
- Reaching for `bash` to grep or count-check the written PRD: `bash` is not
|
|
172
|
+
in this skill's tool surface and the call is refused. Verify with `grep`
|
|
173
|
+
and `read` instead.
|
|
@@ -34,3 +34,73 @@ Expected:
|
|
|
34
34
|
|
|
35
35
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
36
36
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Interview degraded gracefully headless; PRD written to docs/, judge 5/5.
|
|
37
|
+
|
|
38
|
+
## Battletest record (2026-09-03)
|
|
39
|
+
|
|
40
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_productintent.py`. S1's own
|
|
41
|
+
domain ("a log-triage tool for HPC operators") made concrete: a repo with
|
|
42
|
+
`docs/evidence/support-tickets.md` (3 tickets, a 45-min OOM triage, a
|
|
43
|
+
silent-ECC lost queue, a tmux/grep cope with a ~12-node ceiling) and
|
|
44
|
+
`docs/evidence/interview-notes.md` (3 operator quotes, including an explicit
|
|
45
|
+
switch signal). Task: "write the PRD," grounding docs named but not pasted,
|
|
46
|
+
so Step 0's read-first behavior is load-bearing. Graded 12 checks against
|
|
47
|
+
real post-run disk state (file at `docs/<slug>.prd.md`, all 9 sections, a
|
|
48
|
+
hypothesis with distinct RIGHT/WRONG, >=3 seeded facts grounded, zero
|
|
49
|
+
stack-term leaks, non-goals, checkbox open questions) plus the reconstructed
|
|
50
|
+
final assistant text (names the path, offers `architecture` next) and
|
|
51
|
+
process (zero safety blocks, no `tasks` call). `qwen3.8-27b` on `dynamo`
|
|
52
|
+
throughout; one confirm run on `ornith-1.5-35b-a3b`.
|
|
53
|
+
|
|
54
|
+
| run | model | wall | turns | in / out tok | safety blocks | score | outcome |
|
|
55
|
+
|---|---|---|---|---|---|---|---|
|
|
56
|
+
| baseline (no skill) | qwen3.8-27b | 145s | 11 | 189.1k / 13.1k | 0 | 2/12 | wrote `PRD.md` at repo root (wrong name/path); no interview at all, no hypothesis RIGHT/WRONG block, no non-goals/open-questions sections; opened a task list (harmless here, no skill narrowing the surface) |
|
|
57
|
+
| v1 (frozen 0.3.0) | qwen3.8-27b | 137s | 7 | 102.5k / 12.8k | 1 | 10/12 | correct path, sections, hypothesis, grounding, non-goals; opened a `tasks` call refused by the narrowed surface (self-recovered); degraded past the interview on its own reasoning ("`ask_user` isn't in this session's tool surface" — false, it is listed, the model just never tried it) rather than on any instruction in the skill |
|
|
58
|
+
| v2 (first hardened cut) | qwen3.8-27b | 224s | 8 | 170.9k / 20.2k | 1 | 11/12 | no `tasks` call; ran the assumed-confirm monologue explicitly through all 5 clusters citing evidence; one `bash` call (a `$(...)` count-check on the written PRD) refused — `bash` was never in this skill's surface, model reached for it anyway to self-verify, then recovered with `grep` |
|
|
59
|
+
| v3 (final 0.4.0) | qwen3.8-27b | 227s | 7 | 132.5k / 20.5k | 0 | 12/12 | same correctness as v2, self-verified with `grep`/`read` instead of `bash` after the added Red flags line; zero safety blocks, zero stack leaks, explicit "Process notes" section narrating the headless degradation cluster by cluster |
|
|
60
|
+
| confirm (0.4.0) | ornith-1.5-35b-a3b | 69s | 11 | 158.6k / 10.8k | 1 | 11/12 | same content correctness; independently reached for a `bash` echo ("attempting ask_user via context") once, blocked, self-recovered with `grep` — the Red flags line reduced but did not eliminate the `bash` reflex on a second model family |
|
|
61
|
+
|
|
62
|
+
**Changes** (0.3.0 -> 0.4.0): (1) an `## Arguments` contract stating there is
|
|
63
|
+
no operator in a headless run, that `ask_user` returns immediately with no
|
|
64
|
+
answers every time regardless of how many times it's called, and that the
|
|
65
|
+
fix is to apply the existing "user declines" treatment cluster by cluster
|
|
66
|
+
from wherever the first empty response lands — including Step 0's evidence
|
|
67
|
+
check, which the old text left ungated but unaddressed for headless; (2) the
|
|
68
|
+
decline paragraph in "The interview" now cross-references that headless
|
|
69
|
+
default explicitly instead of leaving the model to infer it (v1 inferred a
|
|
70
|
+
*wrong* reason — a nonexistent tool-surface gap — and got lucky); (3) an
|
|
71
|
+
explicit "the clusters below are the plan; `tasks` is refused" line, which
|
|
72
|
+
closed v1's one real safety block; (4) a new `## Red flags` section (the
|
|
73
|
+
skill had none) naming the concrete failures seen across runs: stack-term
|
|
74
|
+
leaks, a WRONG condition that's just RIGHT negated, the literal `PRD.md`
|
|
75
|
+
name, re-calling `ask_user` after an empty response, skipping the loop
|
|
76
|
+
outright instead of degrading into it, ungrounded claims, and reaching for
|
|
77
|
+
`bash` (not in this skill's surface) to self-verify instead of `grep`/`read`.
|
|
78
|
+
|
|
79
|
+
**Design note on the biggest named risk**: the mission brief flagged gating
|
|
80
|
+
hard on Step 0/cluster 1 and never reaching Generate as the single biggest
|
|
81
|
+
risk for this skill. It did not reproduce on either model tested, on any
|
|
82
|
+
version including the unhardened v1 baseline snapshot — `ask_user`'s
|
|
83
|
+
headless behavior in this harness (confirmed by reading
|
|
84
|
+
`src/tools/ask-user.ts`: with no operator handler wired by `clio-coder run`,
|
|
85
|
+
every `ask_user` call resolves immediately to `{cancelled: true}`, framed as
|
|
86
|
+
an ok result with "proceed with defaults" guidance, never an error or a
|
|
87
|
+
hang) means a stalled interview was never actually the failure mode to
|
|
88
|
+
defend against here. What *was* real and reproduced on both models: an
|
|
89
|
+
unprompted reach for `bash` to self-verify a written document, refused
|
|
90
|
+
because `bash` is correctly outside this skill's surface. The hardening
|
|
91
|
+
therefore targets the reproduced failure (`tasks` in v1, `bash` in v2/
|
|
92
|
+
confirm), not the hypothesized one — matching context/context-handoff's
|
|
93
|
+
own finding that the ask_user-stall defense is precautionary, not
|
|
94
|
+
repro-driven, here too.
|
|
95
|
+
|
|
96
|
+
**Still weak**: the `bash`-reach-to-verify reflex was reduced (v2 -> v3 on
|
|
97
|
+
qwen3.8-27b: fixed) but not eliminated on ornith-1.5-35b-a3b, which hit the
|
|
98
|
+
identical refused-tool pattern even after the Red flags line existed — a
|
|
99
|
+
prose warning did not fully generalize across model families, only across
|
|
100
|
+
runs of the same one. S2 (explicit "skip the questions, just write it") and
|
|
101
|
+
S3 (solution-shaped request, "PRD for adding a reply button") from the
|
|
102
|
+
scenario list above were not run standalone against 0.4.0 — only the S1-style
|
|
103
|
+
combined evidence fixture ran, five times. The `git` tool (in allowed-tools)
|
|
104
|
+
was never exercised in any run; a fixture with prior commits/branches to
|
|
105
|
+
reference might exercise it. Only 27-35B class models tried, no small-model
|
|
106
|
+
run.
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: tech-spec
|
|
3
|
-
description:
|
|
3
|
+
description: "Writes a typed call-stack architecture handoff: code-shaped contracts plus execution flows, implementation-ready for another engineer. User-invoked only. Not for weighing approaches or deciding the design; use architecture first."
|
|
4
4
|
triggers:
|
|
5
5
|
- write a tech spec
|
|
6
6
|
- typed call-stack handoff
|
|
7
7
|
- code-shaped contracts
|
|
8
8
|
- implementation-ready technical specification
|
|
9
9
|
- specify execution flows
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.3.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
disable-model-invocation: true
|
|
13
13
|
allowed-tools:
|
|
@@ -42,6 +42,46 @@ TypeScript pseudocode plus end-to-end execution flows. Prose explains why;
|
|
|
42
42
|
types and call stacks define what changes. Design only — never implement,
|
|
43
43
|
and save a file only when the user asks; otherwise return the spec inline.
|
|
44
44
|
|
|
45
|
+
## Arguments
|
|
46
|
+
|
|
47
|
+
```text
|
|
48
|
+
/skill tech-spec <the change, in a few sentences, or a path to read first>
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
- The text is the design problem: what's changing and why. A doc or file
|
|
52
|
+
path named in the request (a PRD, an architecture decision, a module) is
|
|
53
|
+
context to read, not more arguments — see "Load local context" below.
|
|
54
|
+
- Nothing is required beyond some text; a blank invocation falls straight
|
|
55
|
+
to Path B's first question rather than inventing a change to spec.
|
|
56
|
+
- **Output defaults to inline.** Write a file only when the request says
|
|
57
|
+
so explicitly — "save it", "write it to `<path>`", "put it in `docs/`".
|
|
58
|
+
Absent that, the finished spec is the reply itself: no file, in this run
|
|
59
|
+
or a prior one in the same session, gets created for it. This holds
|
|
60
|
+
regardless of which path below runs or how long the spec is — length is
|
|
61
|
+
never itself a reason to write a file.
|
|
62
|
+
- Disabled for model self-invocation and requires the `tdd` skill be
|
|
63
|
+
installed to reference in the TDD Test Plan section; both are frontmatter
|
|
64
|
+
facts, not something to explain to the user unless asked.
|
|
65
|
+
|
|
66
|
+
There is no operator in a headless run: `ask_user` either isn't registered
|
|
67
|
+
or nothing answers it, and a call that goes unanswered will not resolve
|
|
68
|
+
differently on a second try. In Path B (below), that means: state the
|
|
69
|
+
question, your recommendation grounded in the codebase and any docs read
|
|
70
|
+
(or the most defensible engineering default when nothing grounds it), and
|
|
71
|
+
the reasoning; adopt the recommendation; mark it `assumed — confirm`; move
|
|
72
|
+
to the next question. Run every question this way, end to end, not just
|
|
73
|
+
the first — the interview is the plan to execute, not an outline to
|
|
74
|
+
abbreviate because no one answered the opening question. Never invent a
|
|
75
|
+
fact or a codebase detail to back an assumption; anything genuinely
|
|
76
|
+
unknown becomes an Open Question in the spec, not a plausible guess. This
|
|
77
|
+
degrades the interview only — it never licenses writing a file that
|
|
78
|
+
wasn't asked for.
|
|
79
|
+
|
|
80
|
+
The steps below are the plan; do not open a task list for them. `tasks`
|
|
81
|
+
sits outside this skill's tool surface and any call to it is refused.
|
|
82
|
+
`bash` is also outside this skill's tool surface — verify what you wrote
|
|
83
|
+
with `grep`, `read`, and `find`, never `bash`.
|
|
84
|
+
|
|
45
85
|
## Choose the path
|
|
46
86
|
|
|
47
87
|
- **Path A — convert context to spec**: the conversation, docs, or codebase
|
|
@@ -51,6 +91,8 @@ and save a file only when the user asks; otherwise return the spec inline.
|
|
|
51
91
|
with a recommended answer per question (the grill-me posture); anything
|
|
52
92
|
answerable by exploring the codebase is explored, not asked. When context
|
|
53
93
|
suffices, run Path A. Never invent requirements to skip the interview.
|
|
94
|
+
See Arguments above for how a headless run carries every question
|
|
95
|
+
through instead of stalling on the first one.
|
|
54
96
|
|
|
55
97
|
## Path A
|
|
56
98
|
|
|
@@ -110,7 +152,8 @@ contracts, seams, call stacks, or the test plan for being hard):
|
|
|
110
152
|
|
|
111
153
|
The spec follows the outline, every boundary has a typed contract or a
|
|
112
154
|
stated reason it needs none, every behavior has a call stack, unknowns are
|
|
113
|
-
open questions rather than invented design,
|
|
155
|
+
open questions rather than invented design, nothing was implemented, and
|
|
156
|
+
no file was written unless the request asked for one.
|
|
114
157
|
|
|
115
158
|
## Red flags
|
|
116
159
|
|
|
@@ -119,3 +162,11 @@ open questions rather than invented design, and nothing was implemented.
|
|
|
119
162
|
- Speculative seams no invariant, boundary, or test earns.
|
|
120
163
|
- The same rule restated in three sections.
|
|
121
164
|
- "While I'm here" implementation.
|
|
165
|
+
- Writing the spec to a file when nothing in the request asked for one —
|
|
166
|
+
the default output is always the inline reply.
|
|
167
|
+
- A Path B question left unanswered instead of run as the assumed-confirm
|
|
168
|
+
monologue, or an interview skipped straight into Path A without ever
|
|
169
|
+
asking the first question.
|
|
170
|
+
- Opening a task list for the steps above; `tasks` is refused. Reaching for
|
|
171
|
+
`bash` to grep or verify the spec; `bash` is not in this skill's tool
|
|
172
|
+
surface and the call is refused — use `grep`/`read`/`find`.
|
|
@@ -45,3 +45,76 @@ Expected:
|
|
|
45
45
|
|
|
46
46
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
47
47
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Spec written and its claims exercised with node -e; judge 4/4.
|
|
48
|
+
|
|
49
|
+
## Battletest record (2026-09-03)
|
|
50
|
+
|
|
51
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_techspec.py`, continuing the
|
|
52
|
+
planning category's shared HPC log-triage domain (`product-intent` -> `prd`
|
|
53
|
+
-> `tech-spec`). Seeds a plausible root `PRD.md` (purpose, features,
|
|
54
|
+
out-of-scope, stack, integrations, data model, milestones, and the
|
|
55
|
+
always-on-vs-on-demand ingestion tension explicitly marked as *not this
|
|
56
|
+
document's decision*) plus the existing partial codebase (`src/scanner.py`:
|
|
57
|
+
a working `FailureEvent` + `scan_oom`, OOM only) and two sample dmesg logs
|
|
58
|
+
carrying real OOM/ECC/Xid line formats, inside a git repo. Three task
|
|
59
|
+
variants, one fixture:
|
|
60
|
+
|
|
61
|
+
- **base** (S1, Path A): "spec ECC + Xid detection and cross-signature
|
|
62
|
+
top-3 ranking" — sufficient context, no save request. Graded on 9 checks
|
|
63
|
+
against the *reconstructed final assistant text* (this skill's default
|
|
64
|
+
output is inline, not a file): all 11 outline-derived sections present,
|
|
65
|
+
>=5 domain grounding terms, >=2 materially different alternatives,
|
|
66
|
+
`FailureEvent` reused not respecced, nothing implemented (`scanner.py`
|
|
67
|
+
byte-identical to seed), the ingestion trade-off left unresolved, zero
|
|
68
|
+
safety blocks, no `tasks` call, and — the check this run exists to catch —
|
|
69
|
+
**no file written when nothing asked for one**.
|
|
70
|
+
- **save** (S1 variant, confirmation only, run once on the final version):
|
|
71
|
+
same task plus an explicit "save it to docs/tech-spec-log-triage.md" —
|
|
72
|
+
10 checks, same 8 plus the file existing at exactly that path and no
|
|
73
|
+
other new file appearing.
|
|
74
|
+
- **thin** (S2, Path B, confirmation only, run once on the final version):
|
|
75
|
+
a genuinely vague "improve our failure detection, you'll need to ask me
|
|
76
|
+
stuff" request with 5 lighter checks — zero safety blocks, no silent
|
|
77
|
+
stall, no `tasks` call, and either a real `ask_user` exchange or the
|
|
78
|
+
assumed-confirm monologue (`assumed` + `confirm` both present), with a
|
|
79
|
+
real spec still produced.
|
|
80
|
+
|
|
81
|
+
| run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
|
|
82
|
+
|---|---|---|---|---|---|---|---|
|
|
83
|
+
| baseline (no skill) | ornith-1.5-35b-a3b | 80s | 5 | 54.9k / 12.1k | 0 | 4/9 | never invoked `/skill tech-spec`; discovered the installed skill itself via `context(scope="skills")`, read its SKILL.md directly, then called `artifact` and terminated early with a `.clio-coder/artifacts/PLAN.md` instead of a spec — no alternatives, no sections, wrong output shape |
|
|
84
|
+
| v1 (frozen 0.2.0) | ornith-1.5-35b-a3b | 95s | 8 | 115.0k / 15.1k | 1 | 6/9 | ran Path A correctly and produced a genuinely strong spec (11/11 sections, 3 material alternatives, `FailureEvent` reused, nothing implemented) but opened a `tasks` plan (refused, safety block) and **wrote the spec to `docs/tech-spec-scanner-ecc-xid.md` without being asked to** — the exact Path-A/B default-output risk flagged going in |
|
|
85
|
+
| v2 (live 0.3.0) | ornith-1.5-35b-a3b | 68s | 5 | 57.2k / 10.4k | 0 | 9/9 | same spec quality, zero safety blocks, no `tasks` call, correctly returned inline with no file written; final text states explicitly "I did **not** write a file, since nothing in the request asked to save it" |
|
|
86
|
+
| v2 confirm — save | ornith-1.5-35b-a3b | 110s | 11 | 184.8k / 16.5k | 0 | 10/10 | explicit "save it to docs/tech-spec-log-triage.md" correctly produces exactly that file at that path, nothing else |
|
|
87
|
+
| v2 confirm — thin (Path B) | ornith-1.5-35b-a3b | 69s | 9 | 111.8k / 11.7k | 0 | 5/5 | correctly identified insufficient context, ran Path B, and carried all five scope decisions (S1-S5) through as an explicit assumed-confirm monologue headlessly instead of stalling or silently skipping to Path A |
|
|
88
|
+
|
|
89
|
+
**Changes**: (1) `## Arguments` contract with the slash-invocation syntax,
|
|
90
|
+
what's required vs. inferred, and — the section that mattered most here —
|
|
91
|
+
an explicit "output defaults to inline" rule stated as its own bullet
|
|
92
|
+
before the headless-monologue prose, so the fix for Path B's ask_user gap
|
|
93
|
+
can't be misread as license to always write a file; (2) the headless
|
|
94
|
+
no-operator paragraph, ported from `product-intent`/`prd`, applied to Path
|
|
95
|
+
B's grill-me interview: every question runs as state-question /
|
|
96
|
+
grounded-recommendation / reasoning / adopt / mark `assumed — confirm`,
|
|
97
|
+
end to end, not just the first one; (3) explicit `tasks` and `bash`
|
|
98
|
+
refusal lines — `tasks` was v1's only safety block; (4) `Done when` and
|
|
99
|
+
`Red flags` both gained a line naming the unrequested-file failure and the
|
|
100
|
+
unanswered-Path-B-question failure by name, plus the existing `tasks`/`bash`
|
|
101
|
+
refusal repeated as a red flag (matching `prd`'s and `product-intent`'s
|
|
102
|
+
pattern of naming the exact observed failure, not a generic reminder).
|
|
103
|
+
Version 0.2.0 -> 0.3.0.
|
|
104
|
+
|
|
105
|
+
**Still weak**: per this pass's coordinator note, no secondary-model
|
|
106
|
+
confirmation was run (qwen3.8-27b was skipped in favor of running one full
|
|
107
|
+
cycle on ornith-1.5-35b-a3b at speed, concurrently with a sibling agent
|
|
108
|
+
hardening `architecture` on `mini`); the fix is validated on one model
|
|
109
|
+
class only. The baseline's failure mode (discovering and improvising from
|
|
110
|
+
the installed skill file directly, without ever invoking it, then calling
|
|
111
|
+
`artifact` for an unrelated early exit) is a skill-selection/tool-scoping
|
|
112
|
+
gap this SKILL.md cannot fix from inside its own body. `code_nav` (in
|
|
113
|
+
`allowed-tools`) was never exercised — the fixture's one-file codebase
|
|
114
|
+
never needed it. `requires: [skill:tdd]` is a diagnostic-only reference in
|
|
115
|
+
this harness (unmet requires warn, never block `--skill`-path invocation);
|
|
116
|
+
the TDD Test Plan section reads fine without the `tdd` skill installed, but
|
|
117
|
+
that was not tested with `tdd` actually present to see if the reference
|
|
118
|
+
changes. Genuine unknowns (S3 from the original evals) were exercised only
|
|
119
|
+
incidentally via the Xid-severity and ECC-correctable open questions, not
|
|
120
|
+
as an isolated scenario.
|
package/skills/registry.yaml
CHANGED
|
@@ -6,131 +6,139 @@ skills:
|
|
|
6
6
|
# ── coding ──
|
|
7
7
|
- name: ast-grep
|
|
8
8
|
path: coding/ast-grep
|
|
9
|
-
version: 0.
|
|
10
|
-
sha256:
|
|
9
|
+
version: 0.3.0
|
|
10
|
+
sha256: 0a72f4c906303f550be78f80a927da7e24e5c2076f058ccafdec80dfbed275af
|
|
11
11
|
- name: coding-standards
|
|
12
12
|
path: coding/coding-standards
|
|
13
|
-
version: 0.
|
|
14
|
-
sha256:
|
|
13
|
+
version: 0.3.0
|
|
14
|
+
sha256: 3ee3481430591a8f8d041bd1c5fa078becd301d88eb49610e41d056494d3413d
|
|
15
15
|
- name: prototype
|
|
16
16
|
path: coding/prototype
|
|
17
|
-
version: 0.
|
|
18
|
-
sha256:
|
|
17
|
+
version: 0.4.0
|
|
18
|
+
sha256: 0f82df386c2ee565b0e968216ece746b1a11b4acd79676812074c1a406b698e1
|
|
19
19
|
- name: tdd
|
|
20
20
|
path: coding/tdd
|
|
21
|
-
version: 0.
|
|
22
|
-
sha256:
|
|
21
|
+
version: 0.4.0
|
|
22
|
+
sha256: 63b88f29424091a92a8d8c0cc94474ae7e78af76491aea17e6f54af66f32db2d
|
|
23
23
|
# ── context ──
|
|
24
24
|
- name: context-handoff
|
|
25
25
|
path: context/context-handoff
|
|
26
|
-
version: 0.
|
|
27
|
-
sha256:
|
|
26
|
+
version: 0.5.0
|
|
27
|
+
sha256: e69adb1533a6a850cb83babe7e580dd260e034781d7c35e65921453e3fa191a5
|
|
28
28
|
- name: context-prime
|
|
29
29
|
path: context/context-prime
|
|
30
|
-
version: 0.
|
|
31
|
-
sha256:
|
|
30
|
+
version: 0.4.0
|
|
31
|
+
sha256: 21587263297a7ee9e81a2d66f6fc800a1ee6db15f76de1e650563b71fdf19d8e
|
|
32
32
|
# ── git ──
|
|
33
|
+
- name: branch-closeout
|
|
34
|
+
path: git/branch-closeout
|
|
35
|
+
version: 0.1.0
|
|
36
|
+
sha256: 3227f31b0693bb6428abb20a2d4f449aeba1d8e45a79cf0c2c7c333c30c10664
|
|
33
37
|
- name: file-ticket
|
|
34
38
|
path: git/file-ticket
|
|
35
|
-
version: 0.
|
|
36
|
-
sha256:
|
|
39
|
+
version: 0.3.0
|
|
40
|
+
sha256: 65029c2d9d545728750c8a713f92035bd6bdbfa9874de442df72bd059846cb7a
|
|
37
41
|
- name: fix-issue
|
|
38
42
|
path: git/fix-issue
|
|
39
|
-
version: 0.
|
|
40
|
-
sha256:
|
|
43
|
+
version: 0.3.0
|
|
44
|
+
sha256: 626e3aa8ca95bc603f2e0cdd2e7502aa96be85af6bfa92b635c39ef99d3845a5
|
|
41
45
|
- name: resolve-merge-conflicts
|
|
42
46
|
path: git/resolve-merge-conflicts
|
|
43
|
-
version: 0.
|
|
44
|
-
sha256:
|
|
47
|
+
version: 0.4.0
|
|
48
|
+
sha256: 901f482aa2231552d64a721e0d1c58dc8f211347543b5a91fd2ab3c7e413c11c
|
|
45
49
|
- name: ship
|
|
46
50
|
path: git/ship
|
|
47
|
-
version: 0.
|
|
48
|
-
sha256:
|
|
51
|
+
version: 0.5.0
|
|
52
|
+
sha256: 4d626718fdb8b67cf3f22d09ac5a228e5c8103befa1295a672587a6605f038b1
|
|
49
53
|
- name: worktree-create
|
|
50
54
|
path: git/worktree-create
|
|
51
|
-
version: 0.
|
|
52
|
-
sha256:
|
|
55
|
+
version: 0.6.0
|
|
56
|
+
sha256: a4ae790d916a170977814b4334e14aef4f96fb304e3289f7beb5e31e8fb83881
|
|
53
57
|
- name: worktree-merge
|
|
54
58
|
path: git/worktree-merge
|
|
55
|
-
version: 0.
|
|
56
|
-
sha256:
|
|
59
|
+
version: 0.6.0
|
|
60
|
+
sha256: 0fbe2297622ed955b2e4fb14306d75a51bd36db7eebf59e8e92fac1da5b6e64e
|
|
57
61
|
# ── meta ──
|
|
58
62
|
- name: clio-coder-dev
|
|
59
63
|
path: meta/clio-coder-dev
|
|
60
|
-
version: 0.
|
|
61
|
-
sha256:
|
|
64
|
+
version: 0.4.0
|
|
65
|
+
sha256: 8980f3bed0f06c24685bc0a02f3ffbbcdac9442616dc3ce54122265b666ae0d7
|
|
62
66
|
- name: clio-coder-test
|
|
63
67
|
path: meta/clio-coder-test
|
|
64
|
-
version: 0.
|
|
65
|
-
sha256:
|
|
68
|
+
version: 0.3.0
|
|
69
|
+
sha256: 87f6689d083080e4b80294304f2b2fd36f70ce941ab7265cb573e84e4c5e6a12
|
|
66
70
|
- name: credentials
|
|
67
71
|
path: meta/credentials
|
|
68
|
-
version: 0.
|
|
69
|
-
sha256:
|
|
72
|
+
version: 0.2.0
|
|
73
|
+
sha256: d9382790ab30c9a13e5ca451f6f6b6b81eb8bcc6d458ad4931cc0495d39ea78d
|
|
70
74
|
- name: find-skills
|
|
71
75
|
path: meta/find-skills
|
|
72
|
-
version: 0.
|
|
73
|
-
sha256:
|
|
76
|
+
version: 0.2.0
|
|
77
|
+
sha256: a2f8469b3a15059f8547e7566dcd27894c6e2a2126b3ed83acb470df051cd11d
|
|
74
78
|
- name: herdr
|
|
75
79
|
path: meta/herdr
|
|
76
|
-
version: 0.
|
|
77
|
-
sha256:
|
|
80
|
+
version: 0.2.0
|
|
81
|
+
sha256: eeda051fe8736344b480fbd0d482a0e0ab82d4fc44e7ced5ff7058378beae3b6
|
|
78
82
|
- name: skill-craft
|
|
79
83
|
path: meta/skill-craft
|
|
80
|
-
version: 0.
|
|
81
|
-
sha256:
|
|
84
|
+
version: 0.3.0
|
|
85
|
+
sha256: ba81e09412f26647a9384b07359201efb08949865c50934778a4ab3173d43096
|
|
82
86
|
# ── planning ──
|
|
87
|
+
- name: archify
|
|
88
|
+
path: planning/archify
|
|
89
|
+
version: 0.1.0
|
|
90
|
+
sha256: d9e891eb5f3c27678de14165a0eb35d1eb2289be480e55009f9a57bcfe55be2d
|
|
83
91
|
- name: architecture
|
|
84
92
|
path: planning/architecture
|
|
85
|
-
version: 0.
|
|
86
|
-
sha256:
|
|
93
|
+
version: 0.4.0
|
|
94
|
+
sha256: e9017412b6261492b987fe27f7573b72013a340c0d3eb0e7e568fd588b22a214
|
|
87
95
|
- name: backlog
|
|
88
96
|
path: planning/backlog
|
|
89
|
-
version: 0.
|
|
90
|
-
sha256:
|
|
97
|
+
version: 0.4.0
|
|
98
|
+
sha256: e634e2fb5035731e7daa2ef246ae4ab3ef5d792b0b604f4216dfe10335ee10c0
|
|
91
99
|
- name: prd
|
|
92
100
|
path: planning/prd
|
|
93
|
-
version: 0.
|
|
94
|
-
sha256:
|
|
101
|
+
version: 0.4.0
|
|
102
|
+
sha256: 39fe417f495be4153541dd70c886cf773265636dc9fa1844cfc7012b71cbbec1
|
|
95
103
|
- name: product-intent
|
|
96
104
|
path: planning/product-intent
|
|
97
|
-
version: 0.
|
|
98
|
-
sha256:
|
|
105
|
+
version: 0.4.0
|
|
106
|
+
sha256: d41689db9a8c9b2dad0cd630412bebaed5792f9d22d4edb18770e4b1b3308699
|
|
99
107
|
- name: tech-spec
|
|
100
108
|
path: planning/tech-spec
|
|
101
|
-
version: 0.
|
|
102
|
-
sha256:
|
|
109
|
+
version: 0.3.0
|
|
110
|
+
sha256: 5f9add0d01e43feb6808b2eaeae9ac68075cf369cbfc793e020181a642d4d2c4
|
|
103
111
|
# ── research ──
|
|
104
112
|
- name: arxiv-literature
|
|
105
113
|
path: research/arxiv-literature
|
|
106
|
-
version: 0.
|
|
107
|
-
sha256:
|
|
114
|
+
version: 0.5.0
|
|
115
|
+
sha256: 0bc6b39d12934568804fc11b23219de4c6c2172e28c4bfedcee3104e2217aae6
|
|
108
116
|
- name: experiment-protocol
|
|
109
117
|
path: research/experiment-protocol
|
|
110
|
-
version: 0.
|
|
111
|
-
sha256:
|
|
118
|
+
version: 0.3.0
|
|
119
|
+
sha256: 235256cbdf44ebdd25d87cce60940e00ad1fca42a1b2b4e534bfc245f0d1135c
|
|
112
120
|
- name: scientific-debugging
|
|
113
121
|
path: research/scientific-debugging
|
|
114
|
-
version: 0.
|
|
115
|
-
sha256:
|
|
122
|
+
version: 0.3.0
|
|
123
|
+
sha256: 9c29d70b4b82a440640b41665c23731b4112125915bde0e91016fc60aec67207
|
|
116
124
|
- name: scientific-modernization
|
|
117
125
|
path: research/scientific-modernization
|
|
118
|
-
version: 0.
|
|
119
|
-
sha256:
|
|
126
|
+
version: 0.4.0
|
|
127
|
+
sha256: 491522d547fa0bddef19d2ec92bc35236b169787bded6ecc3a29dfec19892886
|
|
120
128
|
# ── workflow ──
|
|
121
129
|
- name: cut-it
|
|
122
130
|
path: workflow/cut-it
|
|
123
|
-
version: 0.
|
|
124
|
-
sha256:
|
|
131
|
+
version: 0.4.0
|
|
132
|
+
sha256: d3742221f0ace1f1b62c07f5902030e478496059a58ef194e7cb3303ccdff05e
|
|
125
133
|
- name: design-council
|
|
126
134
|
path: workflow/design-council
|
|
127
|
-
version: 0.
|
|
128
|
-
sha256:
|
|
135
|
+
version: 0.5.0
|
|
136
|
+
sha256: c279a94a4cd66d46980f9a5024aedac3e673144fadc9d36b53ad87fc2ee63501
|
|
129
137
|
- name: grill-me
|
|
130
138
|
path: workflow/grill-me
|
|
131
|
-
version: 0.
|
|
132
|
-
sha256:
|
|
139
|
+
version: 0.5.0
|
|
140
|
+
sha256: f48687566a8419aa52f30f8234f8dcfd050eaefdfdcfa281723d3603f3e08fcc
|
|
133
141
|
- name: workflow-distiller
|
|
134
142
|
path: workflow/workflow-distiller
|
|
135
|
-
version: 0.
|
|
136
|
-
sha256:
|
|
143
|
+
version: 0.4.0
|
|
144
|
+
sha256: dd7c312d6904f861fa105334c7c2f18c9d6e26a70a76874c85858cee90bede58
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Skills whose content lives in another repository at a pinned ref.
|
|
2
|
+
# Clio never vendors these. `npm run skills:pin` publishes each entry into
|
|
3
|
+
# skill-marketplace.json with the upstream tree as its sourceUrl, the catalog
|
|
4
|
+
# overlay whose files land on top of that tree at install, and the upstream
|
|
5
|
+
# top-level members the install drops. The overlay SKILL.md is pinned in
|
|
6
|
+
# registry.yaml like every other catalog skill.
|
|
7
|
+
version: 1
|
|
8
|
+
skills:
|
|
9
|
+
- name: archify
|
|
10
|
+
category: planning
|
|
11
|
+
sourceUrl: https://github.com/tt-a1i/archify/tree/v2.16.0/archify
|
|
12
|
+
overlay: skills/planning/archify
|
|
13
|
+
exclude: [test, package-lock.json]
|