@iowarp/clio-coder 0.4.2 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +35 -0
- package/CONTRIBUTING.md +86 -19
- package/README.md +35 -6
- package/dist/{acp-TMDQZDIG.js → acp-H2NGRPWO.js} +11 -11
- package/dist/{agents-5N5NG3XG.js → agents-TL5LLUQP.js} +54 -53
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-Z5CCBXKQ.js → auth-E5SW4HMS.js} +19 -16
- package/dist/{builtins-K6TNDT24.js → builtins-IA7V7FUC.js} +9 -4
- package/dist/{chunk-ZW4HH5JJ.js → chunk-2APPQIER.js} +6 -6
- package/dist/{chunk-72GZI5EV.js → chunk-2JH2WHGE.js} +2 -2
- package/dist/{chunk-UBRFI4HS.js → chunk-2UG5F4C5.js} +127 -47
- package/dist/{chunk-UH632ZYL.js → chunk-2UH2KFUP.js} +2 -2
- package/dist/{chunk-3F7VUY77.js → chunk-2VIKGWFZ.js} +2 -2
- package/dist/{chunk-I66EAJFY.js → chunk-2WZ546HR.js} +267 -232
- package/dist/{chunk-RKSR6VSF.js → chunk-4IUZQIJ3.js} +29 -1
- package/dist/{chunk-2X4RYJTJ.js → chunk-4UVU7BJ5.js} +2 -2
- package/dist/{chunk-M2DAX4F6.js → chunk-4WR7VSYB.js} +2 -2
- package/dist/{chunk-5KW52TEP.js → chunk-54CBCGIR.js} +5 -5
- package/dist/{chunk-3KIPBMUA.js → chunk-5ICU3EUH.js} +2 -2
- package/dist/{chunk-77QIVUZB.js → chunk-5MEZN6CB.js} +4 -4
- package/dist/{chunk-O42A54GG.js → chunk-5OIVVPHF.js} +2 -2
- package/dist/{chunk-LJID3DYZ.js → chunk-64I3JVYM.js} +2 -2
- package/dist/{chunk-4JDLP6ZS.js → chunk-6PTFB5VS.js} +7 -7
- package/dist/{chunk-VN3SHNBN.js → chunk-7DICMOS6.js} +2 -2
- package/dist/{chunk-HLAFFSEK.js → chunk-7DRAWPTZ.js} +2 -2
- package/dist/chunk-7E7I3WLS.js +3762 -0
- package/dist/{chunk-YJISEZKC.js → chunk-7ZYNNDKC.js} +6 -6
- package/dist/{chunk-I66ZTYNP.js → chunk-AF4YM7Z4.js} +236 -101
- package/dist/{chunk-2HFQNRV3.js → chunk-AX2THNSA.js} +12 -12
- package/dist/{chunk-PGF63K6I.js → chunk-B4OAX3SI.js} +65 -3
- package/dist/{chunk-W6NIE6OW.js → chunk-B4VEBZKF.js} +3 -3
- package/dist/{chunk-JBCS7CRR.js → chunk-BEPZRGGU.js} +10 -10
- package/dist/{chunk-XGDPUNND.js → chunk-CE5AX47J.js} +2 -2
- package/dist/{chunk-I64IFBLB.js → chunk-DWUOQKRU.js} +17 -10
- package/dist/{chunk-DZAW46HP.js → chunk-E3TPLWFX.js} +3 -3
- package/dist/{chunk-HIICAHCJ.js → chunk-EKCHAPYA.js} +2 -2
- package/dist/{chunk-XE3PCIXH.js → chunk-F5JHEYZM.js} +7 -7
- package/dist/{chunk-5PFYMY2V.js → chunk-FTMGRKEF.js} +2 -2
- package/dist/{chunk-ZNT2M6TG.js → chunk-G76U63X4.js} +17 -17
- package/dist/{chunk-34BHNEE3.js → chunk-GHS5EBTQ.js} +58 -7
- package/dist/{chunk-2NHR3NAY.js → chunk-GI7YYQ3F.js} +40 -34
- package/dist/{chunk-DYHAXKHD.js → chunk-GWZNEVM2.js} +12 -8
- package/dist/chunk-GYV6VZOC.js +26 -0
- package/dist/{chunk-MQXIVJ35.js → chunk-HAXOFFRH.js} +5 -5
- package/dist/{chunk-W4YEMFBX.js → chunk-HEQY7ZFI.js} +2 -2
- package/dist/{chunk-IKOZFYBN.js → chunk-I7ZPNEJM.js} +145 -102
- package/dist/{chunk-WNP7O5WZ.js → chunk-ID64D7PE.js} +4 -4
- package/dist/{chunk-XQRY4DTA.js → chunk-IGLP3ODT.js} +10 -10
- package/dist/chunk-IJNZMHLA.js +101 -0
- package/dist/{chunk-JWJGP5DQ.js → chunk-INY6HTFL.js} +7 -7
- package/dist/{chunk-PBP4B7XR.js → chunk-IUE3Y34X.js} +2 -2
- package/dist/{chunk-B74PXLU7.js → chunk-IWT4SF4R.js} +3 -3
- package/dist/{chunk-B7HM5Z7T.js → chunk-JDAY6FIL.js} +5 -5
- package/dist/{chunk-PJX3WQUQ.js → chunk-JEQ3XTHC.js} +2 -2
- package/dist/{chunk-FSP7CMNU.js → chunk-JGRC33J2.js} +50 -4
- package/dist/{chunk-X7IARSHT.js → chunk-JKKCYP3C.js} +9 -9
- package/dist/{chunk-HJWWJ6IL.js → chunk-JSC3U7TI.js} +16 -4
- package/dist/{chunk-SSEYRH53.js → chunk-KK4JZPBQ.js} +19 -140
- package/dist/{chunk-IHXBNWMM.js → chunk-KXDSS5WJ.js} +7 -3
- package/dist/{chunk-Q4XWMHX6.js → chunk-L47TF46W.js} +2 -2
- package/dist/{chunk-O3YUNJZ2.js → chunk-LDJG7DW3.js} +81 -24
- package/dist/{chunk-CDNVLKUX.js → chunk-LLDJM5XK.js} +13 -7
- package/dist/{chunk-F2I26BDK.js → chunk-MUW2BDDH.js} +4 -4
- package/dist/{chunk-HKMD33FO.js → chunk-MWUZBSAQ.js} +79 -76
- package/dist/{chunk-QQLGQY2A.js → chunk-N2Z7HLVY.js} +20 -20
- package/dist/{chunk-DZEK6CJN.js → chunk-NIQJ66N4.js} +19 -19
- package/dist/{chunk-CWVRRIEI.js → chunk-NZMNUPZZ.js} +2 -2
- package/dist/{chunk-462T4EGZ.js → chunk-O5CVSAG5.js} +2 -2
- package/dist/{chunk-TPEQIQIE.js → chunk-OML5D5V5.js} +8 -8
- package/dist/{chunk-IKSLQ4XV.js → chunk-PAJQJ7BS.js} +558 -216
- package/dist/{chunk-ZW55JB7N.js → chunk-PUVDKJ2Y.js} +2 -2
- package/dist/{chunk-UH347SHR.js → chunk-QWGDJJYJ.js} +11 -11
- package/dist/chunk-R6Q67RJH.js +134 -0
- package/dist/{chunk-CRFOIAX3.js → chunk-RRNP2ANY.js} +6 -6
- package/dist/{chunk-IDNA72AH.js → chunk-RSJ25QSL.js} +2 -2
- package/dist/chunk-SKHCAU7K.js +385 -0
- package/dist/{chunk-RLYRBIYQ.js → chunk-TM6LQDI3.js} +20 -12
- package/dist/chunk-UOIZ7DA4.js +41 -0
- package/dist/{chunk-P75RZCJW.js → chunk-UPZU6GE4.js} +3 -3
- package/dist/{chunk-MCMZMDAC.js → chunk-V2ANDPVT.js} +4 -4
- package/dist/{chunk-AK5XEFVZ.js → chunk-VA5FNYMT.js} +26 -13
- package/dist/{chunk-IMXMHHMQ.js → chunk-VW6DOEDG.js} +332 -57
- package/dist/{chunk-XOXV5GKE.js → chunk-W6RRQCPQ.js} +16 -7
- package/dist/{chunk-CYZW7JHJ.js → chunk-WBKFA554.js} +8 -8
- package/dist/{chunk-BO7Y52RY.js → chunk-WCXUNS7U.js} +7 -7
- package/dist/{chunk-ZGNYYXQ6.js → chunk-WRBAGUNF.js} +3 -3
- package/dist/{chunk-FVDGR2ZL.js → chunk-XIVNBFZS.js} +85 -30
- package/dist/{chunk-BYMNWQ7O.js → chunk-XPWWI35G.js} +299 -58
- package/dist/{chunk-KPXDY6QF.js → chunk-XRZT5WY5.js} +2 -2
- package/dist/{chunk-AZ4WMN4W.js → chunk-Y3CBHOR6.js} +2 -2
- package/dist/{chunk-VXMFAE2W.js → chunk-YPC6ZR5L.js} +19 -6
- package/dist/{chunk-54ODD65L.js → chunk-YQWYVTMC.js} +4 -4
- package/dist/{chunk-M2WXEHER.js → chunk-ZA4VCIGV.js} +2 -2
- package/dist/{chunk-7BHIY2MW.js → chunk-ZDN3Y73Y.js} +6 -6
- package/dist/{chunk-E7GT7O5N.js → chunk-ZWPRK62N.js} +7 -4
- package/dist/cli/index.js +38 -37
- package/dist/{clio-7VB377CC.js → clio-CMMK4KRR.js} +7 -7
- package/dist/{code-nav-YVLCYA7V.js → code-nav-MDZNQS33.js} +7 -7
- package/dist/{components-UBWCQSRW.js → components-UCUQ4QXW.js} +4 -4
- package/dist/{config-4HVOS65E.js → config-SVM5P5YI.js} +76 -74
- package/dist/{configure-PIWO7B24.js → configure-LE3IK2TJ.js} +26 -24
- package/dist/{context-IYEHL3WQ.js → context-2OHRKS42.js} +66 -63
- package/dist/{context-N6ZE3LGJ.js → context-E3VC7RX5.js} +15 -11
- package/dist/{context-KQYIWPWT.js → context-VNCR7KAG.js} +60 -45
- package/dist/{context-clear-G4OGZJDS.js → context-clear-BW4O37TG.js} +61 -59
- package/dist/context-map-COB37XXN.js +505 -0
- package/dist/{context-working-set-BWLF6LJP.js → context-working-set-VDS25HXZ.js} +17 -16
- package/dist/{dispatch-runner-2QQAITS3.js → dispatch-runner-5AHT53RF.js} +85 -74
- package/dist/{doctor-LHBD36VU.js → doctor-WNNVO6FY.js} +37 -37
- package/dist/{eval-C45FYRJ6.js → eval-7G7SGAYO.js} +285 -114
- package/dist/{eval-inventory-6DEJPLBF.js → eval-inventory-Y6QRFOH5.js} +4 -4
- package/dist/{evidence-6SHONYAF.js → evidence-VD6736FQ.js} +63 -62
- package/dist/{evolve-KRKMV72X.js → evolve-AL3NGVRL.js} +62 -61
- package/dist/{extensions-KPZ2UHBB.js → extensions-MOVJ32NM.js} +7 -7
- package/dist/{fleet-IVTCKDHT.js → fleet-QZHUMAGI.js} +110 -108
- package/dist/{fleet-commands-EDWL3IT7.js → fleet-commands-BAYT5FJZ.js} +10 -10
- package/dist/{fleet-decisions-YP3YEFGK.js → fleet-decisions-IREVMRU4.js} +7 -6
- package/dist/{fleet-graph-ZFWKHY2M.js → fleet-graph-YCTT3HTI.js} +19 -18
- package/dist/{fleet-inspect-FVUNCBML.js → fleet-inspect-QVJTDAVB.js} +55 -54
- package/dist/{fleet-preflight-UN5XED4R.js → fleet-preflight-25QAFPK4.js} +4 -4
- package/dist/{fleet-validate-XOWC4HSX.js → fleet-validate-5O57AAJ7.js} +23 -22
- package/dist/{fleet-verify-UN3SODEL.js → fleet-verify-CPH2W2T6.js} +56 -55
- package/dist/{fleet-view-TWHJKCN6.js → fleet-view-SWBR3VGQ.js} +55 -54
- package/dist/{init-T2QORQ3Y.js → init-J477LKZH.js} +78 -76
- package/dist/{interop-IN5I2A66.js → interop-3FCM6XLG.js} +11 -11
- package/dist/{library-LSCATDLZ.js → library-QUQEIUG6.js} +28 -27
- package/dist/{memory-HYOKAGGJ.js → memory-SGGSEP65.js} +64 -63
- package/dist/{models-2GPMFYCM.js → models-HEKUAXXK.js} +49 -43
- package/dist/{monitor-E4ASVUJH.js → monitor-HKU57TYQ.js} +61 -60
- package/dist/{orchestrator-DDMPR3PY.js → orchestrator-VDFAEFAI.js} +919 -546
- package/dist/{panes-E3RUXOW5.js → panes-DN2SSFOH.js} +3 -3
- package/dist/{panes-IXKLOKA2.js → panes-TALGNPZT.js} +8 -8
- package/dist/{paths-L7LGY6RN.js → paths-NBMFAIEZ.js} +5 -5
- package/dist/reset-EAJFFJVB.js +344 -0
- package/dist/{resources-OTRSN34L.js → resources-OVKSEFVE.js} +27 -20
- package/dist/{run-5DEYH5QK.js → run-7DP7ZF2J.js} +113 -109
- package/dist/{share-IHWTLO3M.js → share-WML67FT3.js} +26 -25
- package/dist/{skills-IYMXMKW4.js → skills-SG662R2K.js} +39 -31
- package/dist/{skills-eval-DROHSJAR.js → skills-eval-VVZEUU46.js} +74 -73
- package/dist/{skills-inventory-D7X4L4ZX.js → skills-inventory-I2E23GET.js} +21 -20
- package/dist/{slash-commands-QBM7UZ3B.js → slash-commands-S7MBJDQK.js} +35 -34
- package/dist/{steer-Z5DO23FJ.js → steer-2LQOMCPB.js} +3 -3
- package/dist/{support-U7QOWY26.js → support-CC2UJBJ6.js} +6 -6
- package/dist/{targets-P2FUC4IL.js → targets-4QC3HIEW.js} +48 -45
- package/dist/{terminal-lease-YREJ3JX2.js → terminal-lease-TUHIJ6Y2.js} +2 -2
- package/dist/{tools-5B7RO6MV.js → tools-TFGJICCU.js} +8 -8
- package/dist/{trace-YMGMUM6A.js → trace-FXMXUZUF.js} +7 -7
- package/dist/uninstall-5PEVOE5B.js +408 -0
- package/dist/upgrade-M4WXY6KN.js +303 -0
- package/dist/{usage-ME5MPXGX.js → usage-N7ZNVLEM.js} +147 -102
- package/dist/{verifiers-BVZ7IWOO.js → verifiers-DJTP4XX6.js} +15 -15
- package/dist/{verify-5K7ZKQFC.js → verify-RWE4PPEK.js} +9 -9
- package/dist/{wiki-generate-F5W5QTYY.js → wiki-generate-C7IQOXSP.js} +84 -82
- package/dist/{with-panes-BYOJCLAM.js → with-panes-4GCGSL7J.js} +9 -9
- package/dist/worker/entry.js +61 -60
- package/docs/architecture/artifact-placement.md +1 -0
- package/docs/architecture/artifact-versions.md +1 -1
- package/docs/architecture/context-engine.md +4 -0
- package/docs/architecture/middleware-and-components.md +1 -1
- package/docs/architecture/model-catalog.md +21 -10
- package/docs/architecture/observability.md +12 -1
- package/docs/architecture/prompt-envelope-and-tools.md +2 -0
- package/docs/architecture/provider-adapter-cookbook.md +63 -0
- package/docs/architecture/safety-model.md +15 -5
- package/docs/guide/built-in-agents.md +17 -3
- package/docs/guide/commands-and-modes.md +1 -1
- package/docs/guide/configuration-and-targets.md +97 -9
- package/docs/guide/configuration-reference.md +7 -2
- package/docs/guide/environment-variables.md +2 -0
- package/docs/guide/installation-and-lifecycle.md +37 -4
- package/docs/guide/proactive-memory.md +66 -55
- package/docs/guide/skills-marketplace.md +18 -0
- package/docs/process/development-pipeline.md +34 -1
- package/docs/process/eval-runner.md +67 -3
- package/evals/behavioral-model.yaml +3 -2
- package/package.json +2 -2
- package/skills/README.md +7 -5
- package/skills/coding/ast-grep/SKILL.md +101 -30
- package/skills/coding/ast-grep/evals.md +26 -0
- package/skills/coding/coding-standards/SKILL.md +40 -5
- package/skills/coding/coding-standards/evals.md +23 -0
- package/skills/coding/prototype/SKILL.md +87 -28
- package/skills/coding/prototype/evals.md +19 -0
- package/skills/coding/tdd/SKILL.md +80 -53
- package/skills/coding/tdd/evals.md +20 -0
- package/skills/context/context-handoff/SKILL.md +43 -2
- package/skills/context/context-handoff/evals.md +44 -0
- package/skills/context/context-prime/SKILL.md +45 -15
- package/skills/context/context-prime/evals.md +45 -0
- package/skills/git/branch-closeout/SKILL.md +132 -0
- package/skills/git/branch-closeout/evals.md +133 -0
- package/skills/git/branch-closeout/references/closeout-checklist.md +81 -0
- package/skills/git/file-ticket/SKILL.md +77 -63
- package/skills/git/file-ticket/assets/issue-template.md +22 -0
- package/skills/git/file-ticket/evals.md +31 -26
- package/skills/git/file-ticket/references/issue-discovery.md +49 -0
- package/skills/git/fix-issue/SKILL.md +87 -64
- package/skills/git/fix-issue/evals.md +35 -31
- package/skills/git/fix-issue/references/diagnosis-and-rca.md +46 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +100 -51
- package/skills/git/resolve-merge-conflicts/evals.md +52 -25
- package/skills/git/resolve-merge-conflicts/references/conflict-matrix.md +126 -0
- package/skills/git/ship/SKILL.md +103 -67
- package/skills/git/ship/assets/pr-template.md +21 -0
- package/skills/git/ship/evals.md +44 -28
- package/skills/git/ship/references/remote-and-branch-policy.md +62 -0
- package/skills/git/worktree-create/SKILL.md +80 -50
- package/skills/git/worktree-create/evals.md +40 -33
- package/skills/git/worktree-create/references/worktree-setup.md +62 -66
- package/skills/git/worktree-merge/SKILL.md +112 -65
- package/skills/git/worktree-merge/evals.md +42 -34
- package/skills/git/worktree-merge/references/merge-strategies.md +52 -0
- package/skills/planning/archify/SKILL.md +196 -0
- package/skills/planning/archify/evals.md +65 -0
- package/skills/planning/architecture/SKILL.md +61 -12
- package/skills/planning/architecture/evals.md +65 -0
- package/skills/planning/backlog/SKILL.md +130 -14
- package/skills/planning/backlog/evals.md +142 -0
- package/skills/planning/prd/SKILL.md +46 -6
- package/skills/planning/prd/evals.md +54 -0
- package/skills/planning/product-intent/SKILL.md +57 -2
- package/skills/planning/product-intent/evals.md +70 -0
- package/skills/planning/tech-spec/SKILL.md +53 -2
- package/skills/planning/tech-spec/evals.md +73 -0
- package/skills/registry.yaml +58 -50
- package/skills/remote.yaml +13 -0
- package/skills/research/arxiv-literature/SKILL.md +76 -18
- package/skills/research/arxiv-literature/evals.md +50 -0
- package/skills/research/experiment-protocol/SKILL.md +20 -1
- package/skills/research/experiment-protocol/evals.md +23 -0
- package/skills/research/scientific-debugging/SKILL.md +23 -1
- package/skills/research/scientific-debugging/evals.md +18 -0
- package/skills/research/scientific-modernization/SKILL.md +26 -1
- package/skills/research/scientific-modernization/evals.md +27 -0
- package/skills/skill-marketplace.json +63 -28
- package/skills/workflow/cut-it/SKILL.md +65 -5
- package/skills/workflow/cut-it/evals.md +101 -0
- package/skills/workflow/design-council/SKILL.md +117 -27
- package/skills/workflow/design-council/evals.md +161 -0
- package/skills/workflow/grill-me/SKILL.md +86 -10
- package/skills/workflow/grill-me/evals.md +153 -0
- package/skills/workflow/workflow-distiller/SKILL.md +76 -17
- package/skills/workflow/workflow-distiller/evals.md +118 -0
- package/src/cli/configure-interop.ts +105 -13
- package/src/cli/configure-oauth.ts +57 -0
- package/src/cli/configure-onboarding.ts +980 -0
- package/src/cli/configure-target.ts +594 -0
- package/src/cli/configure.ts +1082 -528
- package/src/cli/context-map.ts +114 -0
- package/src/cli/context.ts +4 -0
- package/src/cli/index.ts +1 -0
- package/src/cli/lifecycle-presenter.ts +436 -0
- package/src/cli/models.ts +10 -2
- package/src/cli/modes/print.ts +5 -1
- package/src/cli/reset.ts +228 -106
- package/src/cli/run.ts +7 -2
- package/src/cli/select.ts +664 -0
- package/src/cli/skills.ts +9 -2
- package/src/cli/targets.ts +3 -0
- package/src/cli/uninstall.ts +233 -165
- package/src/cli/upgrade.ts +204 -149
- package/src/cli/usage.ts +86 -27
- package/src/cli/validate-model.ts +3 -3
- package/src/core/config.ts +56 -0
- package/src/core/external-diagnostic.ts +44 -0
- package/src/core/gateway-routing.ts +157 -0
- package/src/core/safe-exec.ts +17 -2
- package/src/core/skill-activation.ts +89 -2
- package/src/domains/agents/builtins/world-knowledge.md +31 -0
- package/src/domains/agents/catalog.ts +1 -1
- package/src/domains/agents/result-contract.ts +70 -0
- package/src/domains/context/wiki/map-seed.ts +589 -0
- package/src/domains/context/wiki/plan.ts +2 -2
- package/src/domains/dispatch/admission.ts +29 -0
- package/src/domains/dispatch/agent-candidates.ts +10 -0
- package/src/domains/dispatch/budget-envelope.ts +86 -1
- package/src/domains/dispatch/capability-match.ts +1 -0
- package/src/domains/dispatch/capacity-lease.ts +17 -0
- package/src/domains/dispatch/contract.ts +11 -1
- package/src/domains/dispatch/extension.ts +134 -29
- package/src/domains/dispatch/types.ts +3 -0
- package/src/domains/dispatch/worker-model-metadata.ts +38 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +34 -11
- package/src/domains/eval/metrics/token-stream.ts +201 -31
- package/src/domains/eval/metrics/tracked.ts +40 -4
- package/src/domains/eval/runners/clio-run.ts +5 -2
- package/src/domains/eval/schema/suite.ts +28 -0
- package/src/domains/eval/schema/verdict.ts +2 -2
- package/src/domains/eval/suites/resolve.ts +13 -1
- package/src/domains/eval/suites/run.ts +24 -3
- package/src/domains/interop/registry.ts +6 -2
- package/src/domains/interop/types.ts +4 -0
- package/src/domains/lifecycle/migrations/index.ts +4 -0
- package/src/domains/memory/task-memory-policy.ts +70 -26
- package/src/domains/memory/task-memory-telemetry.ts +1 -0
- package/src/domains/middleware/index.ts +0 -1
- package/src/domains/middleware/marketplace-offer.ts +3 -35
- package/src/domains/middleware/memory-intervention.ts +127 -32
- package/src/domains/middleware/memory-step-endpoint.ts +3 -2
- package/src/domains/middleware/skills-reminder.ts +31 -2
- package/src/domains/observability/compaction-usage.ts +118 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/extension.ts +6 -1
- package/src/domains/observability/out-of-turn-usage.ts +52 -21
- package/src/domains/providers/contract.ts +4 -1
- package/src/domains/providers/extension.ts +40 -9
- package/src/domains/providers/model-capabilities.ts +9 -0
- package/src/domains/providers/model-discovery.ts +2 -0
- package/src/domains/providers/model-runtime-capabilities.ts +15 -5
- package/src/domains/providers/models/local-models/clio-coder-local-coding-targets.yaml +32 -12
- package/src/domains/providers/runtimes/antigravity/antigravity-code.ts +225 -45
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +6 -2
- package/src/domains/providers/runtimes/common/local-synth.ts +2 -0
- package/src/domains/providers/runtimes/protocol/litellm.ts +119 -29
- package/src/domains/providers/support.ts +11 -5
- package/src/domains/providers/target-model-cache.ts +25 -2
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/runtime-descriptor.ts +20 -1
- package/src/domains/providers/types/target-descriptor.ts +19 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/skills/install.ts +72 -7
- package/src/domains/resources/skills/loader.ts +7 -0
- package/src/domains/resources/skills/marketplace.ts +63 -11
- package/src/domains/safety/autonomy.ts +15 -0
- package/src/domains/safety/index.ts +1 -0
- package/src/domains/safety/path-policy.ts +1 -1
- package/src/domains/safety/policy-engine.ts +34 -11
- package/src/domains/safety/protected-artifacts.ts +191 -88
- package/src/domains/safety/run-effects.ts +2 -22
- package/src/domains/safety/skill-authority.ts +55 -0
- package/src/domains/session/compaction/compact.ts +72 -22
- package/src/domains/session/entries.ts +6 -0
- package/src/domains/session/usage.ts +3 -3
- package/src/engine/agent.ts +13 -3
- package/src/engine/ai.ts +26 -8
- package/src/engine/antigravity/subprocess-runtime.ts +386 -120
- package/src/engine/api-registry.ts +3 -0
- package/src/engine/apis/openai-completions.ts +117 -14
- package/src/engine/external-subprocess.ts +114 -6
- package/src/entry/background-model-metadata.ts +18 -0
- package/src/entry/compaction-prompt.ts +57 -0
- package/src/entry/orchestrator.ts +405 -216
- package/src/entry/task-memory-lifecycle.ts +35 -0
- package/src/interactive/chat-loop-messages.ts +13 -4
- package/src/interactive/chat-loop.ts +65 -2
- package/src/interactive/chat-renderer.ts +1 -0
- package/src/interactive/cost-overlay.ts +26 -2
- package/src/interactive/interactive-slash-runtime.ts +2 -1
- package/src/interactive/renderers/worker-entry.ts +32 -0
- package/src/interactive/slash-commands.ts +24 -6
- package/src/interactive/theme/labels.ts +19 -13
- package/src/interactive/turn-context.ts +9 -5
- package/src/interactive/turn-recovery.ts +8 -0
- package/src/interactive/turn-runtime.ts +27 -11
- package/src/interactive/turn-state.ts +7 -0
- package/src/interactive/worker-receipts.ts +1 -0
- package/src/interactive/worker-stream.ts +6 -1
- package/src/tools/context/index.ts +30 -9
- package/src/tools/dispatch-arguments.ts +1 -0
- package/src/tools/dispatch-event-text.ts +10 -0
- package/src/tools/dispatch-plan.ts +1 -0
- package/src/tools/dispatch-runner.ts +12 -0
- package/src/tools/registry.ts +11 -5
- package/src/tools/worker-evidence.ts +3 -1
- package/src/worker/spec-contract.ts +4 -0
- package/dist/chunk-2Z2IKEXI.js +0 -1554
- package/dist/reset-OAQP3W4O.js +0 -230
- package/dist/uninstall-N34PCTGJ.js +0 -331
- package/dist/upgrade-PXK3S2YM.js +0 -325
|
@@ -76,3 +76,156 @@ Expected:
|
|
|
76
76
|
|
|
77
77
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
78
78
|
(30B local, llamacpp on mini), full-auto sandbox. PASS (smoke). Skill loaded and interviewed; judge emitted nothing (truncation).
|
|
79
|
+
|
|
80
|
+
## Battletest record (2026-09-03)
|
|
81
|
+
|
|
82
|
+
Fixture: `/home/akougkas/eval-temp/harness/test_grillme.py`, continuing the
|
|
83
|
+
planning category's shared HPC log-triage domain. Seeds the *decided* v1
|
|
84
|
+
architecture doc (`docs/hpc-log-triage-architecture.md`: on-demand reads
|
|
85
|
+
chosen over an always-on pipeline, alerting explicitly listed "Not
|
|
86
|
+
decided"), the partial `src/scanner.py` (`FailureEvent` + OOM-only
|
|
87
|
+
`scan_oom`), and a `pyproject.toml` declaring `pytest` as the test runner
|
|
88
|
+
— a repo fact the interview must find via `read`, not ask about (evals.md
|
|
89
|
+
S2). The prompt is S1's vague-but-grounded feature request: "grill me on
|
|
90
|
+
adding Slack alerting... nothing about notifications or Slack is decided
|
|
91
|
+
yet." `ask_user` in this harness auto-cancels immediately with no operator
|
|
92
|
+
(confirmed by the planning category and reconfirmed here), which makes
|
|
93
|
+
this skill's entire premise — a one-question-at-a-time live interview — the
|
|
94
|
+
thing under test. Graded 10 checks against the reconstructed final
|
|
95
|
+
assistant text and the raw JSONL's tool-call/safety-block stream: zero
|
|
96
|
+
safety blocks; zero `tasks` calls; no wasted `ask_user` retries (<=3
|
|
97
|
+
calls); repo facts (`scan_oom`, `FailureEvent`, on-demand, `click`,
|
|
98
|
+
`pytest`, `scan_ecc`/`scan_xid`, top-3) grounded in the final text; the
|
|
99
|
+
Step 5 decision-log shape present (`Decision Log`/`Deferred`/`Open
|
|
100
|
+
risks`/`Recommended next step`, not a summary paragraph); decisions
|
|
101
|
+
numbered; the `assumed — confirm` monologue used 3+ times; the Question
|
|
102
|
+
Priority order respected (user/problem framing before naming/polish); and
|
|
103
|
+
**the phase map actually completed** (4+ distinct phase numbers named,
|
|
104
|
+
not a stall after phase 0/1). S3 (user defers) and S5 (stop signal) are
|
|
105
|
+
not separately exercised — a headless run cannot produce a live "whatever
|
|
106
|
+
you think" or "stop" reply to react to; by construction, the
|
|
107
|
+
assumed-confirm monologue *is* the "whatever you think" case running for
|
|
108
|
+
every decision, so S3's expected behavior (record the recommendation,
|
|
109
|
+
say so explicitly) is exercised implicitly on every phase, every run.
|
|
110
|
+
S4 (long phased interview, near round-limit closeout) was not exercised
|
|
111
|
+
standalone. Primary model `qwen3.8-27b` on `dynamo` (LM Studio); cross-model
|
|
112
|
+
confirm on `ornith1.5-35b-moe` on `mini` (llama.cpp), required per this
|
|
113
|
+
session's brief because a prior pass in the same mission found fixes tuned
|
|
114
|
+
on one model family did not always transfer to another.
|
|
115
|
+
|
|
116
|
+
| run | model | wall | turns | in / out tokens | safety blocks | score | outcome |
|
|
117
|
+
|---|---|---|---|---|---|---|---|
|
|
118
|
+
| baseline (no skill) | qwen3.8-27b | 42s | 3 | 32.5k / 3.7k | 0 | 5/10 | never invoked `/skill grill-me`; read both key files and grepped for "alerting", grounded correctly, but produced one live-style question and stopped — no phase map, no decision log (expected: this is what the skill exists to fix) |
|
|
119
|
+
| v1 (frozen 0.4.0) | qwen3.8-27b | 57s | 7 | 91.8k / 5.2k | 2 (1 real: `tasks` refused; 1 benign ENOENT on a fixture-dangling doc path) | 3/10 | opened a `tasks` plan for its own phase map (refused — `tasks` was never called out as off-surface in the frozen body), scanned thoroughly, then asked exactly **one** question in plain text and ended the turn — never called `ask_user` at all, never produced a decision log. This is the exact failure the session was built to catch: the frozen skill's fallback line ("if `ask_user` is unavailable, ask in plain text... keep an internal decision log") reads as permission to have an ordinary single-question live conversation, and the model took it literally |
|
|
120
|
+
| v2 (first hardened cut, 0.5.0) | qwen3.8-27b | 121s | 6 | 78.9k / 11.2k | 1 (benign ENOENT, same dangling doc path) | 9/10 | scanned via `git`/`read`/`ls`, reasoned "ask_user is not in my tool surface" (not quite accurate — it's always exempt — but harmless), and correctly ran the full assumed-confirm monologue through Phases 0-5 to a complete, correctly-shaped decision log, entirely without ever calling `ask_user`. Fixture's dangling PRD reference removed after this run to isolate real vs. benign blocks |
|
|
121
|
+
| v2 re-run (fixture fixed) | qwen3.8-27b | 134s | 7 | 101.0k / 12.1k | 1 real: `bash wc`/`bash head` on the sample log, refused | 9/10 | same strong monologue and decision log; the one new gap was a `bash` reflex the frozen and first-cut bodies never explicitly forbade (grill-me never had `bash` in `allowed-tools`, but nothing told the model not to reach for it) |
|
|
122
|
+
| v3 (bash-refusal added) | qwen3.8-27b | 97s | 5 | n/a | 0 | 6/10 | `bash` reflex gone, but a **regression**: stated "this harness has no `ask_user` tool" (still not quite accurate) and, without ever calling `ask_user`, asked one plain-text question and ended the turn on "Answer 1 / 2 / 3" — the exact frozen-skill failure recurring under the hardened body, proving the earlier fix's trigger ("switch on the first `cancelled` result") had a gap: a model that never calls `ask_user` at all never receives a `cancelled` result to trigger the switch |
|
|
123
|
+
| v4 (no-back-and-forth line added) | qwen3.8-27b | 124s | 6 | 85.1k / 11.0k | 0 | **10/10** | full 9-round monologue (Phases 0-5), correctly-shaped decision log, zero `tasks`/`bash` calls |
|
|
124
|
+
| v5 (stability re-run) | qwen3.8-27b | 95s | 5 | n/a | 0 | **10/10** | repeat of v4's result, confirming v3's regression was not the new steady state |
|
|
125
|
+
| final (cross-model confirm) | ornith1.5-35b-moe (mini) | 64s | 5 | 12.5k / 5.4k | 0 | **10/10** | full monologue and decision log on the second model family too; independently proposed a Slack-message trigger threshold not in the fixture and, in its closing line, distinguished what it would still "press on live" from what it correctly resolved on its own headless — the clearest sign the headless/live distinction actually landed as a real distinction, not just prose the model echoes back |
|
|
126
|
+
|
|
127
|
+
**Changes** (0.4.0 -> 0.5.0):
|
|
128
|
+
|
|
129
|
+
1. **`## Arguments` contract**, the section the skill never had — slash-free
|
|
130
|
+
conversational syntax, and the ported no-operator/`ask_user`-auto-cancels
|
|
131
|
+
rule from the planning category, adapted for a skill whose entire
|
|
132
|
+
contract is "one question at a time" rather than a document write.
|
|
133
|
+
2. **The critical fix, found empirically, not guessed up front**: the
|
|
134
|
+
frozen skill's fallback line ("if `ask_user` is unavailable, ask in
|
|
135
|
+
plain text... one question at a time... internal decision log") reads,
|
|
136
|
+
correctly, as "have a normal one-question conversation" — which is
|
|
137
|
+
exactly wrong when there is no second turn for a reply to land in. The
|
|
138
|
+
fix that actually held (v4/v5, and the cross-model run) is a **"this run
|
|
139
|
+
has no back-and-forth" statement that does not gate on receiving a
|
|
140
|
+
`cancelled` result** from `ask_user` — v3's first attempt gated the
|
|
141
|
+
switch-to-monologue on "the first round comes back cancelled," which
|
|
142
|
+
left a real gap: a model that reasons "`ask_user` isn't available" and
|
|
143
|
+
never calls it at all never receives that trigger, and fell straight
|
|
144
|
+
back into the frozen skill's exact failure. The held version states the
|
|
145
|
+
no-reply-coming rule as a fact about the run itself, independent of
|
|
146
|
+
whether `ask_user` was ever invoked.
|
|
147
|
+
3. **Explicit `tasks` and `bash` refusal**, in Arguments and Red Flags —
|
|
148
|
+
`tasks` was v1's real safety block (opened a plan for its own phase
|
|
149
|
+
map); `bash` was a reflex on the v2 re-run (`bash wc`/`bash head` on a
|
|
150
|
+
file that should have been `read`). Both are outside `allowed-tools`
|
|
151
|
+
already; the gap was that nothing said so in the body.
|
|
152
|
+
4. **Step 3 and Step 4 rewritten** to name the headless-cancellation
|
|
153
|
+
handling explicitly at the exact point in the workflow it applies (not
|
|
154
|
+
only in Arguments): Step 3 states that neither a `cancelled` result nor
|
|
155
|
+
never calling `ask_user` at all is a reason to wait; Step 4 draws the
|
|
156
|
+
line between a live stop signal (still respected) and a headless
|
|
157
|
+
`cancelled` result (not a stop signal, a cue to keep going).
|
|
158
|
+
5. **`git` tool and `code_nav` called out explicitly in Step 1** for repo
|
|
159
|
+
state and symbol lookups respectively — `code_nav` remains unexercised
|
|
160
|
+
by every run this session (see Still weak); `git` was used in every
|
|
161
|
+
hardened run once named.
|
|
162
|
+
6. Five Red Flags entries rewritten or added around the concrete failures
|
|
163
|
+
observed: `tasks`/`bash` reaches, re-calling `ask_user` after a
|
|
164
|
+
cancellation, treating `cancelled` as a user stop signal, and — the
|
|
165
|
+
headline one — ending a turn on "Answer 1/2/3" instead of running the
|
|
166
|
+
monologue to the decision log.
|
|
167
|
+
|
|
168
|
+
**Still weak**: `code_nav` (in `allowed-tools`) was never exercised by any
|
|
169
|
+
run this session — this fixture's grounding lived entirely in prose files
|
|
170
|
+
and one Python module small enough that `read`/`grep` sufficed; a fixture
|
|
171
|
+
with a larger call graph might exercise it, but none was built. S3 and S5
|
|
172
|
+
are reasoned about, not directly run (see the fixture note above) — a
|
|
173
|
+
harness that could inject a specific `ask_user` reply (rather than always
|
|
174
|
+
auto-cancelling) would let those two scenarios run for real instead of by
|
|
175
|
+
inference. S4's near-round-limit closeout behavior is unverified; every
|
|
176
|
+
hardened run here converged well under any plausible `max_rounds` value.
|
|
177
|
+
v3's regression is the one data point worth remembering past this session:
|
|
178
|
+
gating a headless-degradation rule on "the first tool result that comes
|
|
179
|
+
back a certain way" is fragile against a model that skips the tool call
|
|
180
|
+
entirely and reasons its way to the same wrong conclusion by a different
|
|
181
|
+
path — the fix needed to be a fact about the run, not a reaction to one
|
|
182
|
+
tool's return value. Two consecutive clean runs on the primary model and
|
|
183
|
+
one clean cross-model run is reasonable but not exhaustive evidence that
|
|
184
|
+
v3's failure mode is fully closed rather than just less frequent; only a
|
|
185
|
+
larger run count would raise that confidence further.
|
|
186
|
+
|
|
187
|
+
## Live interactive confirmation (2026-09-03, Herdr pane)
|
|
188
|
+
|
|
189
|
+
The headless harness can only prove the assumed-confirm monologue path; it
|
|
190
|
+
cannot produce a genuine "whatever you think" or "stop" reply because
|
|
191
|
+
`ask_user` always auto-cancels with no operator. To exercise S3 and S5 for
|
|
192
|
+
real, this session ran the hardened 0.5.0 skill interactively: a Herdr pane
|
|
193
|
+
running `clio-coder` (target `mini`, model `ornith1.5-35b-moe`, switched
|
|
194
|
+
in-session via `/model`) against the same fixture repo the battletest used
|
|
195
|
+
(`/home/akougkas/eval-temp/grillme-final`, skill installed project-locally
|
|
196
|
+
at `.clio-coder/skills/grill-me/` so the live session resolved this
|
|
197
|
+
repo's edited SKILL.md rather than the separately npm-installed package
|
|
198
|
+
copy — `/skill grill-me` has no path-override flag the way `clio-coder run
|
|
199
|
+
--skill` does), with a human answering each `ask_user` round live via the
|
|
200
|
+
TUI's modal.
|
|
201
|
+
|
|
202
|
+
Result: Step 1 scanned the repo before asking anything (architecture doc,
|
|
203
|
+
scanner.py, sample log). Round 1 asked a single root-decision question
|
|
204
|
+
(trigger model) with the recommended option first and real tradeoffs on
|
|
205
|
+
the alternatives. After the human's real answer came back, the model's own
|
|
206
|
+
reasoning explicitly named the distinction the whole hardening pass turned
|
|
207
|
+
on: *"The modal returned round_answered — there is an operator here...
|
|
208
|
+
this is a live interview, not headless. I'll continue with one question
|
|
209
|
+
per round and wait for your answers."* — proof the headless/live branch is
|
|
210
|
+
a real fork in the model's behavior, not just prose it echoes. Five rounds
|
|
211
|
+
ran in priority order (trigger, success measure, non-goals, delivery/
|
|
212
|
+
secrets, payload shape); round 3 was answered "whatever you think is
|
|
213
|
+
best" and recorded the stated recommendation as the decision (S3,
|
|
214
|
+
confirmed live, not just implied by the monologue path); round 5 was
|
|
215
|
+
answered with an appended stop signal ("...; stop") and the model called
|
|
216
|
+
`ask_user` `action: "complete"` immediately, with no confirmation question
|
|
217
|
+
(S5, confirmed live). The final decision log matched the Step 5 shape
|
|
218
|
+
exactly (numbered decisions, Deferred, Open risks, Recommended next step)
|
|
219
|
+
and, unprompted, flagged a real cross-cutting risk the fixture didn't
|
|
220
|
+
spell out: the alerting feature's success measure depends on Phase 2's
|
|
221
|
+
ranking/CLI, which doesn't exist in `scanner.py` yet — a genuine "hole
|
|
222
|
+
poked," not filler. Zero safety blocks; zero `tasks` calls. One aside, not
|
|
223
|
+
a skill defect: the TUI's own usage nudge fired ("9+ read-only exploration
|
|
224
|
+
calls without a successful Scout dispatch") — Step 1's repo scan currently
|
|
225
|
+
reaches for `read`/`grep`/`ls` directly rather than delegating broad
|
|
226
|
+
reconnaissance to a Scout dispatch, worth a look in a future pass but out
|
|
227
|
+
of scope for this one since the scan itself was correct and grounded.
|
|
228
|
+
|
|
229
|
+
This closes the S3/S5-not-directly-run gap noted above for the specific
|
|
230
|
+
case of a genuine live operator; the headless assumed-confirm path
|
|
231
|
+
remains separately and repeatedly confirmed on its own terms.
|
|
@@ -7,7 +7,7 @@ triggers:
|
|
|
7
7
|
- turn this into a reusable workflow
|
|
8
8
|
- distill this repeated process
|
|
9
9
|
- create a skill from this session
|
|
10
|
-
version: 0.
|
|
10
|
+
version: 0.4.0
|
|
11
11
|
license: Apache-2.0
|
|
12
12
|
allowed-tools:
|
|
13
13
|
- read
|
|
@@ -38,6 +38,37 @@ following skill-craft. skill-craft also governs how every skill this
|
|
|
38
38
|
distiller produces is written: description, body, and pruning rules live
|
|
39
39
|
there, not here.
|
|
40
40
|
|
|
41
|
+
## Arguments
|
|
42
|
+
|
|
43
|
+
```text
|
|
44
|
+
/skill workflow-distiller [<what to distill>]
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
With arguments, the text names the workflow to distill; without, distill
|
|
48
|
+
whichever workflow just ran in this session. Everything else in the request
|
|
49
|
+
(the conversation so far) is the session record Phase 1 reconstructs from,
|
|
50
|
+
not more arguments.
|
|
51
|
+
|
|
52
|
+
A live operator can answer Phase 2's interview and Phase 4's design gate for
|
|
53
|
+
real - treat a real, non-cancelled `ask_user` reply as proof an operator is
|
|
54
|
+
present and continue asking one question per round. There is no operator in
|
|
55
|
+
a headless run: `ask_user` is not registered and every call resolves
|
|
56
|
+
`cancelled` immediately, whether or not it is ever called at all. When that
|
|
57
|
+
is the run's condition, do not wait for a reply that cannot come: answer
|
|
58
|
+
every remaining interview question and the Phase 4 gate yourself as an
|
|
59
|
+
assumed-confirm monologue (state the question, give your best-grounded
|
|
60
|
+
answer or design choice, mark it `assumed - confirm`) and proceed straight
|
|
61
|
+
through to Phase 5, the same way as a live "looks fine, write it." A skill
|
|
62
|
+
file is a plain, reversible artifact under version control, not an
|
|
63
|
+
irreversible external action - write it and say so in the final reply,
|
|
64
|
+
rather than stopping at the gate with nothing produced.
|
|
65
|
+
|
|
66
|
+
`tasks` sits outside this skill's tool surface (`read`, `grep`, `find`,
|
|
67
|
+
`ls`, `context`, `write`, `ask_user`) and any call is refused; the six
|
|
68
|
+
phases above are the plan, not a task list. This skill never has `bash` or
|
|
69
|
+
`git`, so it never commits or pushes the file it writes - say so in the
|
|
70
|
+
final reply and let the user commit it.
|
|
71
|
+
|
|
41
72
|
## Phase 1 - Reconstruct From Evidence
|
|
42
73
|
|
|
43
74
|
Before asking anything, list the concrete steps that visibly executed in this
|
|
@@ -90,27 +121,47 @@ Validation scenario: <prompt, expected observable behavior>
|
|
|
90
121
|
```
|
|
91
122
|
|
|
92
123
|
No skill file is written before the user approves. "Looks fine, but change X"
|
|
93
|
-
means revise and re-present.
|
|
124
|
+
means revise and re-present. In a headless run with no operator (see
|
|
125
|
+
Arguments), treat the design as approved once it is internally consistent
|
|
126
|
+
with Phase 2's decisions, mark it `assumed - confirm` in your final reply,
|
|
127
|
+
and proceed to Phase 5 - do not stop the run with a presented-but-unwritten
|
|
128
|
+
design.
|
|
94
129
|
|
|
95
130
|
## Phase 5 - Create
|
|
96
131
|
|
|
97
|
-
Write `SKILL.md` under `.clio-coder/skills/<approved-name
|
|
98
|
-
|
|
99
|
-
the pruning pass
|
|
100
|
-
check referenced.
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
132
|
+
Write `SKILL.md` under `.clio-coder/skills/<approved-name>/` with the
|
|
133
|
+
frontmatter contract below - a triggers-only third-person description, and
|
|
134
|
+
the pruning pass - with `requires: [skill:<name>]` for every skill the
|
|
135
|
+
overlap check referenced. Do not try to load skill-craft's own file mid-run
|
|
136
|
+
to check this: only one skill can be active at a time, and a second
|
|
137
|
+
`context(scope="skills", name="skill-craft")` call is refused while this
|
|
138
|
+
skill is still pending. The frontmatter contract, current as of this
|
|
139
|
+
writing: `name`, `description` (third-person triggers, no "I"/"you"),
|
|
140
|
+
`triggers` (a non-empty list), `version` (start `0.1.0`), `license`,
|
|
141
|
+
`allowed-tools` (canonical lowercase Clio tool names only - `bash`/`git`
|
|
142
|
+
capitalized or spelled differently is rejected), and `requires` when Phase 3
|
|
143
|
+
found a reference. If in doubt about the exact shape, `read` an already-
|
|
144
|
+
installed skill's `SKILL.md` (this skill's own file is always available) and
|
|
145
|
+
mirror its frontmatter keys rather than guessing or trying to load
|
|
146
|
+
skill-craft. Scope defaults to project; use the user skill store only when
|
|
147
|
+
the user said the workflow crosses repositories. Placeholders replace every
|
|
148
|
+
session-specific path, name, and value; distill the pattern, not the
|
|
149
|
+
incident. Keep the generated skill under 120 lines; reference instead of
|
|
150
|
+
inlining. If the session repeatedly dispatched the same worker pattern, also
|
|
151
|
+
offer a recipe sketch for the agents surface, but do not write recipe files.
|
|
107
152
|
|
|
108
153
|
## Phase 6 - Validate
|
|
109
154
|
|
|
110
155
|
Record one RED-GREEN scenario agreed with the user: the prompt, and the
|
|
111
|
-
observable behavior that distinguishes with-skill from without.
|
|
112
|
-
|
|
113
|
-
|
|
156
|
+
observable behavior that distinguishes with-skill from without. `clio-coder
|
|
157
|
+
skills validate` needs `bash`, which is outside this skill's tool surface -
|
|
158
|
+
never attempt it; instead `read` the file back and confirm the frontmatter
|
|
159
|
+
contract from Phase 5 by eye (required keys present, `allowed-tools` entries
|
|
160
|
+
canonical lowercase, under the line budget), and say plainly that a real
|
|
161
|
+
`clio-coder skills validate` pass is still owed once bash is available.
|
|
162
|
+
Otherwise record the scenario in the skill body's example section as the
|
|
163
|
+
standing validation obligation. This skill has no `bash`/`git`, so it never
|
|
164
|
+
commits or pushes the file it writes; say so and let the user commit it.
|
|
114
165
|
|
|
115
166
|
## Worked Example
|
|
116
167
|
|
|
@@ -128,15 +179,23 @@ script, and verified row counts against the source, three sessions in a row.
|
|
|
128
179
|
installed; no references.
|
|
129
180
|
4. Gate: summary presented; user approves after tightening the description.
|
|
130
181
|
5. Create: write `.clio-coder/skills/csv-ingest/SKILL.md`, placeholders for the
|
|
131
|
-
export URL and column map;
|
|
182
|
+
export URL and column map; frontmatter contract confirmed by reading the
|
|
183
|
+
file back (no `bash`, so no `clio-coder skills validate` this turn).
|
|
132
184
|
6. Validate: scenario "ingest this month's export" must show fetch,
|
|
133
185
|
normalize, count-verify, spot-check in that order; recorded in the body.
|
|
134
186
|
|
|
135
187
|
## Red Flags
|
|
136
188
|
|
|
137
|
-
- Writing any skill before the design gate is approved
|
|
189
|
+
- Writing any skill before the design gate is approved by a live operator,
|
|
190
|
+
or before a headless run has marked it `assumed - confirm`.
|
|
191
|
+
- Stopping the run at Phase 4 with a presented-but-unwritten design when the
|
|
192
|
+
run is headless - that's the backlog pattern (guard an irreversible
|
|
193
|
+
action), and a written skill file is not that; it is reversible.
|
|
138
194
|
- A reconstruction that lists steps nothing in the session shows.
|
|
139
195
|
- Reimplementing an installed skill's job instead of referencing it.
|
|
140
196
|
- Session-specific paths or values surviving into the generated skill.
|
|
141
197
|
- Batching interview questions or ignoring a stop signal.
|
|
142
198
|
- Distilling a one-off without asking about recurrence.
|
|
199
|
+
- Calling `bash` for `clio-coder skills validate`, or `context` with a
|
|
200
|
+
second skill's name to consult it mid-run: both are outside this skill's
|
|
201
|
+
tool surface and refused.
|
|
@@ -105,3 +105,121 @@ prompt; the phases still had to run in order.
|
|
|
105
105
|
|
|
106
106
|
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
107
107
|
(30B local, llamacpp on mini), full-auto sandbox. PASS. Inline session-trace setup engaged; judge 5/6.
|
|
108
|
+
|
|
109
|
+
## Battletest record (2026-09-03)
|
|
110
|
+
|
|
111
|
+
This skill's core loop (a several-turn interview, a live-or-headless design
|
|
112
|
+
gate, a write) is genuinely interactive, so it was tested primarily through
|
|
113
|
+
a live Herdr-driven `clio-coder` pane with a real human answering
|
|
114
|
+
`ask_user` rounds, not only through the headless harness the rest of this
|
|
115
|
+
category used — plus one headless confirm run to exercise the no-operator
|
|
116
|
+
path directly. Fixture: the shared HPC log-triage repo (`test_grillme.py`'s
|
|
117
|
+
`setup_fixture`, `docs/hpc-log-triage-architecture.md` + partial
|
|
118
|
+
`src/scanner.py`/`tests/test_scanner.py`), compounded with a real Part 1
|
|
119
|
+
task ("add `scan_ecc` mirroring `scan_oom`, test it, commit it") run to
|
|
120
|
+
completion *before* invoking `/skill workflow-distiller`, so Phase 1 had a
|
|
121
|
+
genuine session record to reconstruct from rather than a scripted one.
|
|
122
|
+
|
|
123
|
+
**v1 (frozen 0.3.0), live, `ornith1.5-35b-moe` on mini**: Phase 1
|
|
124
|
+
reconstruction was precise and correctly cited to real tool calls. The
|
|
125
|
+
5-round interview (confirm+recurrence, varies/fixed, rigidity/failure,
|
|
126
|
+
scope cuts, name) ran cleanly, one question per round, recommended-first.
|
|
127
|
+
Phase 3's overlap check correctly found nothing to reference. Phase 4's
|
|
128
|
+
design gate presented the exact template shape and correctly waited for a
|
|
129
|
+
live "looks fine, write it" before writing anything. Two real bugs
|
|
130
|
+
surfaced in Phase 5/6: (1) "load skill-craft" is dead - `context(scope=
|
|
131
|
+
"skills", name="skill-craft")` while workflow-distiller is itself the
|
|
132
|
+
active pending skill is refused ("pending skill request(s)"); only one
|
|
133
|
+
skill can be active at a time. (2) Phase 6's "confirm it loads with
|
|
134
|
+
`clio-coder skills validate`" is dead - `bash` is outside this skill's
|
|
135
|
+
`allowed-tools`, so that command can never run. The model self-recovered
|
|
136
|
+
both times (read its own installed SKILL.md as a frontmatter mirror; did a
|
|
137
|
+
by-eye read-back instead of the blocked validate command) and reported the
|
|
138
|
+
gap honestly rather than fabricating a clean validate - a good sign for
|
|
139
|
+
model robustness, but the skill body should not depend on a model
|
|
140
|
+
inventing its own workaround for a dead instruction.
|
|
141
|
+
|
|
142
|
+
**Changes (0.3.0 -> 0.4.0)**: new `## Arguments` section stating the
|
|
143
|
+
live-vs-headless distinction explicitly (a real, non-cancelled `ask_user`
|
|
144
|
+
reply proves an operator is present and interview/gate proceed normally;
|
|
145
|
+
`ask_user` unregistered and auto-cancelling means assumed-confirm through
|
|
146
|
+
Phase 2 and Phase 4 and straight on to Phase 5 - a written skill file is a
|
|
147
|
+
reversible, version-controlled artifact, not the kind of irreversible
|
|
148
|
+
action `backlog`'s Step 3 guards, so it does not get that skill's
|
|
149
|
+
stop-and-report treatment); explicit `tasks` refusal and a "never commits"
|
|
150
|
+
note; Phase 4 gate prose updated with the headless branch; Phase 5 rewrote
|
|
151
|
+
the skill-craft consultation into a concrete, achievable frontmatter
|
|
152
|
+
contract (the required keys, and "`read` an already-installed skill's file
|
|
153
|
+
to mirror the shape, never try to load a second skill mid-run"); Phase 6
|
|
154
|
+
rewrote the validate step to a real, achievable by-eye check instead of
|
|
155
|
+
the dead `bash` command; the worked example's line claiming `clio-coder
|
|
156
|
+
skills validate` passes was corrected to match; two new Red Flags entries
|
|
157
|
+
for the two dead-instruction failure modes.
|
|
158
|
+
|
|
159
|
+
**v2 (hardened 0.4.0), live, `ornith1.5-35b-moe` on mini, fresh session**:
|
|
160
|
+
same compound Part 1 + Part 2 scenario end to end. Real finding along the
|
|
161
|
+
way, unrelated to this skill: the model cannot self-invoke `/skill` - it
|
|
162
|
+
is operator-gated UI, not an agent tool, confirmed twice (once via a
|
|
163
|
+
direct `context` call, once by embedding `/skill workflow-distiller` in a
|
|
164
|
+
larger message instead of sending it standalone) - both times the model
|
|
165
|
+
correctly recognized the constraint, said so, and asked the operator to
|
|
166
|
+
run it rather than retrying or faking activation. Once invoked properly:
|
|
167
|
+
Phase 1 reconstruction included the real mid-session detour (a genuine
|
|
168
|
+
case-sensitivity bug the model introduced and fixed via byte-level
|
|
169
|
+
debugging) rather than a cleaned-up story. The interview produced a
|
|
170
|
+
materially different design from v1's run on the same scenario (different
|
|
171
|
+
name chosen, "leave uncommitted" instead of "commit via bash" for the
|
|
172
|
+
generated skill) - real evidence the interview is actually deciding things,
|
|
173
|
+
not replaying a script. Phase 4 correctly said "this is a live run, so
|
|
174
|
+
I'll hold off writing until you approve" and, when a later `ask_user`
|
|
175
|
+
round had already closed, correctly fell back to a plain-text approval
|
|
176
|
+
request rather than stalling - explicitly restating, unprompted, every
|
|
177
|
+
constraint this session's hardening pass had just added (no commit, no
|
|
178
|
+
`clio-coder skills validate`, `tasks` out of surface). Phase 5 went
|
|
179
|
+
straight to `read`-ing its own installed SKILL.md to mirror the frontmatter
|
|
180
|
+
contract, with zero attempt to load skill-craft - bug #1 confirmed fixed.
|
|
181
|
+
The generated `.clio-coder/skills/add-signature-scanner/SKILL.md` (97
|
|
182
|
+
lines) has valid frontmatter, canonical-lowercase `allowed-tools`, real
|
|
183
|
+
placeholders, and a concrete validation scenario.
|
|
184
|
+
|
|
185
|
+
**A real, separate platform-level finding, not a skill-body bug**: during
|
|
186
|
+
this same v2 live session's Phase 6, a `bash wc -l` call executed
|
|
187
|
+
successfully (`exit 0`, green checkmark confirmed via `--format ansi`) even
|
|
188
|
+
though `bash` is not in workflow-distiller's `allowed-tools` - and the
|
|
189
|
+
identical class of call had been correctly refused earlier in this exact
|
|
190
|
+
mission's v1 session under the same skill ("bash is outside the tool
|
|
191
|
+
surface declared by the active skill(s)"). The model's own final report
|
|
192
|
+
then claimed "I can't run bash" in the same breath as having just run it.
|
|
193
|
+
This reads as skill tool-surface narrowing lapsing partway through a long,
|
|
194
|
+
many-turn interactive session (Phase 1 through 6 spans several real
|
|
195
|
+
conversation turns; the headless harness's single-turn monologue shape
|
|
196
|
+
never exercises this), not anything a SKILL.md can fix by itself. Worth a
|
|
197
|
+
maintainer look at the interactive admission path specifically, independent
|
|
198
|
+
of this skill or this mission's edits.
|
|
199
|
+
|
|
200
|
+
**Headless confirm, `qwen3.8-27b` on dynamo, fresh fixture**: single
|
|
201
|
+
`clio-coder run --skill ... --autonomy full-auto --json`. Real fixture
|
|
202
|
+
mismatch, deliberately not corrected: the prompt claimed a prior
|
|
203
|
+
`scan_ecc` commit that does not exist in this fresh fixture (no Part 1 ran
|
|
204
|
+
here). Phase 1 caught the discrepancy against real evidence (`grep` found
|
|
205
|
+
no `scan_ecc`, git log has one seed commit, the docstring still says "not
|
|
206
|
+
yet implemented"), tagged the claim `assumption - unverified`, and
|
|
207
|
+
grounded the distillation in the real architecture doc and `scan_oom`'s
|
|
208
|
+
actual code instead of fabricating verification of a commit that never
|
|
209
|
+
happened - the skill's stated identity ("runtime truth... reconstruction
|
|
210
|
+
wins") holding on a weaker/non-live model under direct pressure to just
|
|
211
|
+
agree. Ran the full assumed-confirm monologue through Phase 2 and Phase 4
|
|
212
|
+
(explicitly marked), wrote `.clio-coder/skills/signature-scanner/SKILL.md`
|
|
213
|
+
(92 lines, valid frontmatter, canonical-lowercase tools, real placeholders,
|
|
214
|
+
a concrete failure-behavior section, a validation scenario naming the still
|
|
215
|
+
-owed real `clio-coder skills validate` pass). Zero safety blocks.
|
|
216
|
+
|
|
217
|
+
**Still weak**: the skill-craft mid-run-load and dead-`bash`-validate bugs
|
|
218
|
+
are confirmed fixed by re-test, but only against this one fixture shape.
|
|
219
|
+
S2 (overlap with an installed skill) was not exercised this pass - this
|
|
220
|
+
fixture's only installed skill is workflow-distiller itself, so the overlap
|
|
221
|
+
check always correctly found nothing; a fixture with a second installed
|
|
222
|
+
skill covering one step is still owed. S3 (no recurrence, offer to stop)
|
|
223
|
+
was not exercised standalone. The tool-surface-lapse finding above is
|
|
224
|
+
real, reproduced, and unresolved - it is the single biggest risk this pass
|
|
225
|
+
surfaced, and it sits outside this skill (and outside `skills/`) entirely.
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { stdout as output } from "node:process";
|
|
2
2
|
import type { createInterface } from "node:readline/promises";
|
|
3
|
+
import chalk from "chalk";
|
|
4
|
+
|
|
3
5
|
import { readSettings } from "../core/config.js";
|
|
4
6
|
import {
|
|
5
7
|
acceptInteropAgents,
|
|
@@ -12,6 +14,9 @@ import {
|
|
|
12
14
|
renderProposalEntry,
|
|
13
15
|
} from "../domains/interop/index.js";
|
|
14
16
|
import { askYesNo } from "./ask.js";
|
|
17
|
+
import { railPrefix } from "./configure-target.js";
|
|
18
|
+
import { createLifecyclePresenter, type LifecyclePresenter } from "./lifecycle-presenter.js";
|
|
19
|
+
import { canSelect, promptMultiSelect } from "./select.js";
|
|
15
20
|
import { printOk } from "./shared.js";
|
|
16
21
|
|
|
17
22
|
function describe(proposal: InteropProposal): string {
|
|
@@ -30,38 +35,125 @@ function describe(proposal: InteropProposal): string {
|
|
|
30
35
|
return `${lines.join("\n")}\n`;
|
|
31
36
|
}
|
|
32
37
|
|
|
38
|
+
/** What one row of the picker says about an agent, beyond its name. */
|
|
39
|
+
function proposalHint(proposal: InteropProposal): string {
|
|
40
|
+
const command = [proposal.entry.command, ...(proposal.entry.args ?? [])].join(" ");
|
|
41
|
+
return proposal.needsNetworkInstall ? `${command} (npx fetches the adapter on first use)` : command;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export interface InteropReviewStreams {
|
|
45
|
+
in: NodeJS.ReadableStream;
|
|
46
|
+
out: NodeJS.WritableStream;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface InteropReviewIo {
|
|
50
|
+
/** Readline interface for the numbered fallback. Null means nothing can answer. */
|
|
51
|
+
rl: ReturnType<typeof createInterface> | null;
|
|
52
|
+
streams?: InteropReviewStreams;
|
|
53
|
+
/** Rail to draw on when the caller already owns one. */
|
|
54
|
+
presenter?: LifecyclePresenter;
|
|
55
|
+
rail?: string;
|
|
56
|
+
/** Skip the "nothing to connect" line, for a caller whose transcript says enough. */
|
|
57
|
+
quiet?: boolean;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface InteropReviewOutcome {
|
|
61
|
+
code: number;
|
|
62
|
+
/** Agent ids that were wired, in the order they were written. */
|
|
63
|
+
wired: string[];
|
|
64
|
+
/** The user left the review without answering it. */
|
|
65
|
+
back: boolean;
|
|
66
|
+
}
|
|
67
|
+
|
|
33
68
|
/**
|
|
34
69
|
* Review detected agents and, with an explicit answer per agent, wire them as
|
|
35
70
|
* delegation peers. Without a TTY this prints the proposals and writes nothing:
|
|
36
71
|
* no code path adds a peer the operator did not agree to.
|
|
72
|
+
*
|
|
73
|
+
* On a terminal this is one multi-select rather than a run of `[y/N]`
|
|
74
|
+
* questions. The questions were identical apart from a name, they hid how many
|
|
75
|
+
* there were, and once the second one was on screen the first could not be
|
|
76
|
+
* changed. The per-agent paragraph they each carried says the same two facts
|
|
77
|
+
* every time, so it is stated once above the list and the row carries what is
|
|
78
|
+
* actually different: the command Clio would run.
|
|
37
79
|
*/
|
|
38
|
-
export async function
|
|
80
|
+
export async function reviewInteropAgents(io: InteropReviewIo): Promise<InteropReviewOutcome> {
|
|
39
81
|
const report = await detectInteropAgents({ cwd: process.cwd(), probeVersion: true });
|
|
40
82
|
const proposals = interopProposals(report, readSettings());
|
|
41
83
|
if (proposals.length === 0) {
|
|
42
|
-
output.write("No new coding agents to connect.\n");
|
|
43
|
-
return 0;
|
|
84
|
+
if (!io.quiet) output.write("No new coding agents to connect.\n");
|
|
85
|
+
return { code: 0, wired: [], back: false };
|
|
44
86
|
}
|
|
45
|
-
|
|
87
|
+
const streams = io.streams;
|
|
88
|
+
// The picker needs a terminal, not a readline interface; the caller that owns
|
|
89
|
+
// the rail deliberately keeps no readline open, because one left attached
|
|
90
|
+
// echoes the keys the picker is reading.
|
|
91
|
+
const interactive =
|
|
92
|
+
streams !== undefined && canSelect(streams.in as NodeJS.ReadStream, streams.out as NodeJS.WriteStream);
|
|
93
|
+
if (!interactive && io.rl === null) {
|
|
46
94
|
for (const proposal of proposals) output.write(describe(proposal));
|
|
47
95
|
output.write("\nRun `clio-coder configure --interop` on a terminal to connect any of these.\n");
|
|
48
|
-
return 0;
|
|
96
|
+
return { code: 0, wired: [], back: false };
|
|
49
97
|
}
|
|
98
|
+
|
|
50
99
|
const accepted: InteropAgentId[] = [];
|
|
51
100
|
const declined: InteropAgentId[] = [];
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
const
|
|
55
|
-
|
|
101
|
+
|
|
102
|
+
if (interactive && streams) {
|
|
103
|
+
const presenter = io.presenter ?? createLifecyclePresenter({ stream: streams.out });
|
|
104
|
+
const rail = io.rail ?? railPrefix(presenter.isPlain());
|
|
105
|
+
presenter.note(
|
|
106
|
+
`Clio found ${proposals.length === 1 ? "one coding agent" : `${proposals.length} coding agents`} it can delegate to. A peer receives your task text and never the project projection, and its tool calls are gated by Clio safety.`,
|
|
107
|
+
);
|
|
108
|
+
const result = await promptMultiSelect<InteropAgentId>({
|
|
109
|
+
heading: ["", chalk.bold("Delegate to any of these?")],
|
|
110
|
+
choices: proposals.map((proposal) => ({
|
|
111
|
+
value: proposal.kind,
|
|
112
|
+
label: proposal.label,
|
|
113
|
+
hint: proposalHint(proposal),
|
|
114
|
+
})),
|
|
115
|
+
railPrefix: rail,
|
|
116
|
+
backLabel: "back",
|
|
117
|
+
confirmLabel: "confirm",
|
|
118
|
+
clearOnExit: true,
|
|
119
|
+
input: streams.in as NodeJS.ReadStream,
|
|
120
|
+
output: streams.out as NodeJS.WriteStream,
|
|
121
|
+
});
|
|
122
|
+
if (result.kind === "back") return { code: 0, wired: [], back: true };
|
|
123
|
+
if (result.kind === "quit") return { code: 0, wired: [], back: false };
|
|
124
|
+
for (const proposal of proposals) {
|
|
125
|
+
(result.values.includes(proposal.kind) ? accepted : declined).push(proposal.kind);
|
|
126
|
+
}
|
|
127
|
+
} else if (io.rl !== null) {
|
|
128
|
+
for (const proposal of proposals) {
|
|
129
|
+
output.write(describe(proposal));
|
|
130
|
+
const yes = await askYesNo(io.rl, `Add ${proposal.label} as delegation agent \`${proposal.entry.id}\`?`, false);
|
|
131
|
+
(yes ? accepted : declined).push(proposal.kind);
|
|
132
|
+
}
|
|
56
133
|
}
|
|
134
|
+
|
|
135
|
+
const wired: string[] = [];
|
|
57
136
|
if (accepted.length > 0) {
|
|
58
137
|
const result = acceptInteropAgents(accepted, report);
|
|
59
|
-
for (const diagnostic of result.diagnostics)
|
|
60
|
-
|
|
138
|
+
for (const diagnostic of result.diagnostics) {
|
|
139
|
+
if (io.presenter) io.presenter.warn(diagnostic);
|
|
140
|
+
else output.write(`note: ${diagnostic}\n`);
|
|
141
|
+
}
|
|
142
|
+
for (const id of result.wired) {
|
|
143
|
+
wired.push(id);
|
|
144
|
+
// A caller with a rail lists what it wrote in its own completion rows.
|
|
145
|
+
if (!io.presenter) printOk(`delegation agent ${id} added; use \`/delegate ${id} <task>\``);
|
|
146
|
+
}
|
|
61
147
|
}
|
|
62
148
|
if (declined.length > 0) {
|
|
63
149
|
declineInteropAgents(declined, report);
|
|
64
|
-
|
|
150
|
+
const line = `Declined ${declined.join(", ")}; Clio stays quiet about them until their version or path changes.`;
|
|
151
|
+
if (io.presenter) io.presenter.note(line);
|
|
152
|
+
else output.write(`${line}\n`);
|
|
65
153
|
}
|
|
66
|
-
return 0;
|
|
154
|
+
return { code: 0, wired, back: false };
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
export async function runInteropReview(io: InteropReviewIo): Promise<number> {
|
|
158
|
+
return (await reviewInteropAgents(io)).code;
|
|
67
159
|
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The browser sign-in half of `clio-coder configure`.
|
|
3
|
+
*
|
|
4
|
+
* This stays on readline rather than the arrow-key prompts: an OAuth flow ends
|
|
5
|
+
* with a code the user pastes, and there is no list to choose from. It lives in
|
|
6
|
+
* its own file so the first-run wizard can start a login without importing the
|
|
7
|
+
* rest of configure.ts.
|
|
8
|
+
*/
|
|
9
|
+
import type { createInterface } from "node:readline/promises";
|
|
10
|
+
|
|
11
|
+
import { openAuthStorage } from "../domains/providers/auth/index.js";
|
|
12
|
+
import type { RuntimeDescriptor } from "../domains/providers/types/runtime-descriptor.js";
|
|
13
|
+
import { createDelayedManualCodeInput } from "./oauth-manual-input.js";
|
|
14
|
+
import { promptOAuthSelection } from "./oauth-select.js";
|
|
15
|
+
import { credentialWriteFailed, printError, printOk } from "./shared.js";
|
|
16
|
+
|
|
17
|
+
export async function loginOAuthRuntime(
|
|
18
|
+
rl: ReturnType<typeof createInterface>,
|
|
19
|
+
runtime: RuntimeDescriptor,
|
|
20
|
+
): Promise<boolean> {
|
|
21
|
+
const auth = openAuthStorage();
|
|
22
|
+
if (runtime.authNotice) process.stdout.write(`note: ${runtime.authNotice}\n`);
|
|
23
|
+
const manualCodeInput = createDelayedManualCodeInput(
|
|
24
|
+
rl,
|
|
25
|
+
"Paste verification code if browser callback does not complete automatically: ",
|
|
26
|
+
);
|
|
27
|
+
try {
|
|
28
|
+
await auth.login(runtime.oauthProviderId ?? runtime.id, {
|
|
29
|
+
onAuth: ({ url, instructions }) => {
|
|
30
|
+
process.stdout.write(`\nOpen: ${url}\n`);
|
|
31
|
+
if (instructions) process.stdout.write(`${instructions}\n`);
|
|
32
|
+
process.stdout.write("Waiting for the browser callback. A manual code prompt will appear if needed.\n");
|
|
33
|
+
},
|
|
34
|
+
onDeviceCode: ({ verificationUri, userCode }) => {
|
|
35
|
+
process.stdout.write(`\nOpen: ${verificationUri}\n`);
|
|
36
|
+
process.stdout.write(`Enter code: ${userCode}\n`);
|
|
37
|
+
},
|
|
38
|
+
onPrompt: async (prompt) => {
|
|
39
|
+
const answer = await rl.question(`${prompt.message}${prompt.allowEmpty ? " " : ": "}`);
|
|
40
|
+
return prompt.allowEmpty ? answer : answer.trim();
|
|
41
|
+
},
|
|
42
|
+
onSelect: (prompt) => promptOAuthSelection(rl, prompt),
|
|
43
|
+
onManualCodeInput: manualCodeInput.onManualCodeInput,
|
|
44
|
+
onProgress: (message) => {
|
|
45
|
+
process.stderr.write(`${message}\n`);
|
|
46
|
+
},
|
|
47
|
+
});
|
|
48
|
+
if (credentialWriteFailed(auth, `credential for ${runtime.id} was not stored`)) return false;
|
|
49
|
+
printOk(`authenticated ${runtime.id}`);
|
|
50
|
+
return true;
|
|
51
|
+
} catch (error) {
|
|
52
|
+
printError(error instanceof Error ? error.message : String(error));
|
|
53
|
+
return false;
|
|
54
|
+
} finally {
|
|
55
|
+
manualCodeInput.cancel();
|
|
56
|
+
}
|
|
57
|
+
}
|