@iowarp/clio-coder 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +4 -4
- package/CHANGELOG.md +35 -0
- package/CONTRIBUTING.md +9 -0
- package/README.md +12 -9
- package/ROADMAP.md +30 -5
- package/dist/{acp-KS7ARGV6.js → acp-F6PGOFZ4.js} +2 -2
- package/dist/{agents-YOGJ2YEJ.js → agents-ZHP4R6EX.js} +22 -21
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-CEDCO6O6.js → auth-N75U2H3Z.js} +4 -4
- package/dist/{chunk-D3KWFOUT.js → chunk-23UIXH6Z.js} +3 -3
- package/dist/{chunk-ZR5YA6ZF.js → chunk-36XVK5P5.js} +5 -5
- package/dist/{chunk-AVSR2HKI.js → chunk-3DZ4LPLS.js} +6 -6
- package/dist/{chunk-LYJJ6546.js → chunk-3GSPCWFX.js} +12 -12
- package/dist/{chunk-CRHPCPZP.js → chunk-4KC4CHKJ.js} +2 -2
- package/dist/{chunk-4VDHLCF5.js → chunk-4WJBWKMI.js} +2 -2
- package/dist/{chunk-KBUTSJSR.js → chunk-4XBYHXGK.js} +2 -2
- package/dist/{chunk-6W75D37F.js → chunk-53P775GE.js} +6 -3
- package/dist/{chunk-XZWSP67B.js → chunk-5APYAXCB.js} +2 -2
- package/dist/{chunk-D65E25JG.js → chunk-67HPYNTJ.js} +2 -2
- package/dist/{chunk-5JQR5B4D.js → chunk-6KV4KCQZ.js} +4 -4
- package/dist/{chunk-V4XJWDHR.js → chunk-6UKFRTXI.js} +2 -1
- package/dist/{chunk-NDPF2N2L.js → chunk-7DW7CJW2.js} +2 -2
- package/dist/{chunk-5KQEUIWM.js → chunk-7LUEQODV.js} +4 -4
- package/dist/{chunk-LNOXCGN2.js → chunk-7UFDIOBT.js} +3 -3
- package/dist/{chunk-DXSETN72.js → chunk-7WUGNN43.js} +2 -2
- package/dist/{chunk-V35KZDUZ.js → chunk-AA3GRZWS.js} +2 -2
- package/dist/{chunk-X4CEFOSG.js → chunk-ADO3OQIC.js} +4 -4
- package/dist/{chunk-UX2L5CUD.js → chunk-AI36V3ON.js} +3 -3
- package/dist/{chunk-B7GY6QER.js → chunk-B26OUMJW.js} +2 -2
- package/dist/{chunk-LAR7DX5E.js → chunk-B5TASTCK.js} +153 -62
- package/dist/{chunk-TGWZXJ7L.js → chunk-BEMQK5Q7.js} +92 -84
- package/dist/{chunk-RNDCROSG.js → chunk-CNXU7TR2.js} +3 -3
- package/dist/{chunk-3JLTRPN7.js → chunk-CVMUXT64.js} +35 -66
- package/dist/{chunk-733ZBT5P.js → chunk-E74MJF6G.js} +4 -4
- package/dist/{chunk-T2MDZES3.js → chunk-EBMTBE76.js} +3 -3
- package/dist/{chunk-OZ76TLGU.js → chunk-EEITU7S6.js} +2 -2
- package/dist/{chunk-GRHOXDKB.js → chunk-ELDJWJKM.js} +97 -14
- package/dist/{chunk-GCQFN7YI.js → chunk-EMYEQLWO.js} +2 -2
- package/dist/{chunk-LTWH2ACR.js → chunk-EQPFEDFV.js} +222 -127
- package/dist/{chunk-TDGYTWM6.js → chunk-F4I3H56H.js} +2 -2
- package/dist/chunk-F6KCJO3U.js +523 -0
- package/dist/{chunk-FGSQHUCK.js → chunk-F6VKWTTY.js} +5 -5
- package/dist/{chunk-UJD5MPP6.js → chunk-FLNXQQ5B.js} +2 -2
- package/dist/{chunk-EX57VVWW.js → chunk-FOSRJ2VZ.js} +4 -4
- package/dist/{chunk-ZARZ4POR.js → chunk-G5ZDQQE3.js} +3 -3
- package/dist/{chunk-HQVN7G4F.js → chunk-G62JBPXN.js} +7 -7
- package/dist/{chunk-ZJPORZTC.js → chunk-GIAI7A2K.js} +2 -2
- package/dist/chunk-GPTR6OTU.js +2551 -0
- package/dist/{chunk-EZD2BXGP.js → chunk-GUFQPTQK.js} +2 -2
- package/dist/{chunk-OPD7GL6F.js → chunk-H6FQVNZC.js} +8 -7
- package/dist/chunk-H7REJ7L5.js +80 -0
- package/dist/{chunk-EYTLV3W3.js → chunk-HTWTORF7.js} +2 -2
- package/dist/{chunk-DF6PI6GN.js → chunk-I6UKQPGS.js} +3 -3
- package/dist/{chunk-ISHUS7HC.js → chunk-JRW236JT.js} +3 -3
- package/dist/{chunk-DE2LU267.js → chunk-KKAJGT3B.js} +3 -2
- package/dist/{chunk-OV62D6KO.js → chunk-KQMYL5GR.js} +3 -3
- package/dist/{chunk-QDUUOZ3S.js → chunk-KWMFC7FN.js} +3 -3
- package/dist/{chunk-R7FGRZNF.js → chunk-LXYHCCIZ.js} +9 -9
- package/dist/{chunk-NFSK2VNU.js → chunk-LZMTVVN4.js} +2 -2
- package/dist/{chunk-62UK7AGW.js → chunk-M5PMRZSR.js} +397 -44
- package/dist/chunk-MMZMM6SW.js +131 -0
- package/dist/{chunk-FFYTC2KQ.js → chunk-MNABNEBX.js} +2 -2
- package/dist/{chunk-XMDBTYCH.js → chunk-N3CWOGVF.js} +3 -3
- package/dist/{chunk-VYRQHORQ.js → chunk-NMQSO6Z6.js} +4 -3
- package/dist/{chunk-R2UUNSNQ.js → chunk-O3ZAF7UY.js} +18 -5
- package/dist/{chunk-4SYHGVDE.js → chunk-OQNCLZI4.js} +4 -4
- package/dist/{chunk-EG4ARXPC.js → chunk-OX6QTA4N.js} +3 -3
- package/dist/{chunk-P4GNZPXO.js → chunk-OYUS3UZP.js} +2 -2
- package/dist/{chunk-OQ2EOAHA.js → chunk-PF36KOGI.js} +291 -112
- package/dist/{chunk-55RMDDFM.js → chunk-PGQG7ZMW.js} +2 -2
- package/dist/{chunk-CBAQPTDU.js → chunk-PIU6BXXW.js} +2 -2
- package/dist/{chunk-EKV4UBCM.js → chunk-PZBQUJ2F.js} +2 -2
- package/dist/{chunk-7RET77VB.js → chunk-Q2HKY32Y.js} +2 -2
- package/dist/{chunk-U6YTUVOX.js → chunk-Q4RMNWMZ.js} +4 -4
- package/dist/chunk-Q6FRR3AQ.js +48 -0
- package/dist/{chunk-6RJMYYIE.js → chunk-QJSIAJBY.js} +3 -3
- package/dist/{chunk-DOUISWEE.js → chunk-QOJ4ERMU.js} +5 -5
- package/dist/{chunk-IFPDAD7R.js → chunk-RWSRKXLT.js} +3 -3
- package/dist/{chunk-T33IYTZM.js → chunk-SAASVMIY.js} +3 -3
- package/dist/{chunk-2DTIWSAW.js → chunk-SEQDH6JC.js} +3 -3
- package/dist/{chunk-BRCOSE7O.js → chunk-SZUO5BSE.js} +2 -2
- package/dist/{chunk-HOEDA42N.js → chunk-T7HZ7KJF.js} +2 -2
- package/dist/{chunk-OJXOA6YU.js → chunk-TL6LNERH.js} +2 -2
- package/dist/{chunk-UCINIRHW.js → chunk-TPCZXWLT.js} +3 -3
- package/dist/{chunk-ZUKUCZYZ.js → chunk-TQ2KTH4A.js} +2 -2
- package/dist/{chunk-4WOOTKFD.js → chunk-U3FETNQB.js} +341 -20
- package/dist/{chunk-5B3RTYAK.js → chunk-UGDSB4AN.js} +2 -2
- package/dist/{chunk-VKHLUZNO.js → chunk-UZ7YBL43.js} +3 -3
- package/dist/{chunk-QNUQ7K7D.js → chunk-VKMSEO7Y.js} +4 -4
- package/dist/{chunk-IN7DGBVS.js → chunk-VLX5VZ35.js} +5 -5
- package/dist/{chunk-SJGS3GDI.js → chunk-WAGBMMNX.js} +5 -5
- package/dist/{chunk-BFOSV5EZ.js → chunk-WX2YCH7F.js} +2 -2
- package/dist/{chunk-3LT34CAM.js → chunk-XBIGUILU.js} +32 -17
- package/dist/{chunk-XDHUDE5K.js → chunk-XKYBFRWR.js} +5 -5
- package/dist/{chunk-U5QU5ZOD.js → chunk-XYPWFSU5.js} +3 -3
- package/dist/{chunk-YBUECLAF.js → chunk-YEJQDODA.js} +2 -2
- package/dist/{chunk-YQ6XEFVK.js → chunk-Z74OGONW.js} +39 -7
- package/dist/{chunk-I4ELN5BX.js → chunk-ZTNOEJRI.js} +2 -2
- package/dist/cli/index.js +26 -26
- package/dist/{clio-K2PBVCEM.js → clio-4IKFE25F.js} +2 -2
- package/dist/{clio-context-tools-GR4OHFSO.js → clio-context-tools-OHJXYX4R.js} +25 -23
- package/dist/{code-nav-4X3OI6DK.js → code-nav-XPM3MY5I.js} +7 -7
- package/dist/{config-Q5J7UBAB.js → config-343YS25T.js} +43 -41
- package/dist/{config-graph-P3555574.js → config-graph-WI263KDX.js} +43 -41
- package/dist/{configure-KIJ32ZO6.js → configure-GZDG4HER.js} +21 -21
- package/dist/{context-LFBQOSXT.js → context-4OFO7N7Y.js} +12 -12
- package/dist/{context-WBH3KSVR.js → context-JLK2RLOG.js} +42 -40
- package/dist/{context-NNLBES2Y.js → context-O6DIWCF6.js} +23 -21
- package/dist/{context-clear-YOHMQ6SQ.js → context-clear-MR7KOGQI.js} +42 -40
- package/dist/{context-working-set-MRAEVDNV.js → context-working-set-YZAD3KWO.js} +14 -13
- package/dist/{data-tool-JJPBGMQJ.js → data-tool-EKD7DFHG.js} +7 -7
- package/dist/{detail-RHJOTGN4.js → detail-YZPHFEVJ.js} +43 -41
- package/dist/{dispatch-runner-NC7PYNIO.js → dispatch-runner-VKVXVTWJ.js} +44 -42
- package/dist/{doctor-S5KZ2GTN.js → doctor-53AVQXAO.js} +17 -16
- package/dist/{doctor-deep-2EJQSHIZ.js → doctor-deep-BLRB2NCG.js} +5 -5
- package/dist/{eval-3D6M36N7.js → eval-P3BQSLQE.js} +21 -20
- package/dist/{evidence-XNVPPLVK.js → evidence-6LLI6SZH.js} +44 -42
- package/dist/{evidence-6IYN52ET.js → evidence-HTN3JLUO.js} +42 -40
- package/dist/{evidence-JVECPI3F.js → evidence-NJRE3R5H.js} +42 -40
- package/dist/{evolve-TLPUDJDM.js → evolve-WWZHJVJP.js} +42 -40
- package/dist/{fleet-WKVBF5SZ.js → fleet-GGZ4BVJE.js} +64 -62
- package/dist/{fleet-6RJWIOEO.js → fleet-QG3IIWWM.js} +44 -42
- package/dist/{fleet-commands-L6WFT7KZ.js → fleet-commands-J2ZIKISQ.js} +7 -7
- package/dist/{fleet-decisions-BTQMWU5A.js → fleet-decisions-Y2Q7375Q.js} +8 -8
- package/dist/{fleet-graph-553DZ67Q.js → fleet-graph-TCBCCCBP.js} +11 -11
- package/dist/{fleet-inspect-236FZPRB.js → fleet-inspect-KOLKBC56.js} +44 -42
- package/dist/{fleet-preflight-NZQHABWO.js → fleet-preflight-44UNCC4Y.js} +27 -25
- package/dist/{fleet-validate-7NLAE26R.js → fleet-validate-K2FXRZVA.js} +13 -13
- package/dist/{fleet-verify-E3RO6A5W.js → fleet-verify-AS444WDW.js} +42 -40
- package/dist/{fleet-view-KMZQY7YJ.js → fleet-view-X4JU4J6O.js} +43 -41
- package/dist/gui/ops-worker.js +9 -9
- package/dist/gui/reads-worker.js +9 -8
- package/dist/{init-DV5B6HQG.js → init-VSOWFHVL.js} +54 -52
- package/dist/{interactive-OXGYZYAS.js → interactive-JRCBENLZ.js} +1733 -577
- package/dist/{interop-J7UOBOSF.js → interop-JR6BCY7R.js} +10 -10
- package/dist/{inventory-OEGIL6PY.js → inventory-3JUGLVJI.js} +43 -41
- package/dist/{library-ROZAMUIL.js → library-7BOH3Y7N.js} +15 -15
- package/dist/{library-B3UPGY4V.js → library-7PFONXUI.js} +9 -9
- package/dist/{library-MXQA5DLT.js → library-TUYZMFIA.js} +7 -7
- package/dist/{library-import-2LTWOVBD.js → library-import-TJVXKLDK.js} +10 -10
- package/dist/{library-inventory-LJ3XHLWO.js → library-inventory-MD3AYTP4.js} +9 -9
- package/dist/{library-validation-DARHHUNW.js → library-validation-ZDLITEA3.js} +7 -7
- package/dist/{mcp-2GSPYQ54.js → mcp-EFCWUPUR.js} +3 -3
- package/dist/{memory-J3STUUIK.js → memory-Q2YVOKK5.js} +42 -40
- package/dist/{models-Y3GMFJ2V.js → models-M7BOCW7F.js} +17 -16
- package/dist/{monitor-LHNH577J.js → monitor-2HD3MC67.js} +47 -45
- package/dist/{orchestrator-DTAPXULA.js → orchestrator-OUW3ZN6Y.js} +1894 -226
- package/dist/{panes-A743ONZ5.js → panes-TTPPSYCL.js} +3 -3
- package/dist/{preload-OJBMAIXF.js → preload-A47C2NUF.js} +42 -40
- package/dist/{providers-CMETPZNS.js → providers-T43W7EL4.js} +4 -4
- package/dist/{resources-FY6XYFN2.js → resources-TP6X2V6W.js} +20 -11
- package/dist/{run-H64MEVTA.js → run-LDSBPMCS.js} +61 -59
- package/dist/{share-CAFMMQLM.js → share-URHKKMHR.js} +9 -9
- package/dist/{skills-ERKQUC7L.js → skills-E477MOOO.js} +13 -11
- package/dist/{skills-eval-CQ23MYQK.js → skills-eval-AIBR2L7H.js} +583 -93
- package/dist/{skills-inventory-TVUYJDVN.js → skills-inventory-HLMKY3R5.js} +13 -11
- package/dist/{slash-commands-ROG3KIFJ.js → slash-commands-LIIZKA5Y.js} +28 -26
- package/dist/{startup-background-KDWCQMWT.js → startup-background-JR2KDK6R.js} +43 -41
- package/dist/{steer-KKYXJSH4.js → steer-D7RTAPCM.js} +3 -3
- package/dist/{system-ZDW44ZGX.js → system-X2P5N4IY.js} +14 -13
- package/dist/{targets-AJWA6E7Z.js → targets-VFFAKUFM.js} +25 -24
- package/dist/{tasks-6L2JJRC5.js → tasks-ERAWZA22.js} +6 -6
- package/dist/{terminal-lease-YFUSDPAD.js → terminal-lease-V2N2BIKQ.js} +2 -2
- package/dist/{trace-LA5MVR4Y.js → trace-MA5BT7XY.js} +3 -3
- package/dist/{usage-XSILDHKF.js → usage-WYFY2L6M.js} +50 -48
- package/dist/{verifiers-ECA5GIHL.js → verifiers-TYIW2XKF.js} +7 -7
- package/dist/{verify-R6QQCCEB.js → verify-242UFYTP.js} +6 -6
- package/dist/{web-fetch-PVZNMBOB.js → web-fetch-LRAGYU6Y.js} +3 -3
- package/dist/{wiki-generate-RP22WP5K.js → wiki-generate-OFKOX2T6.js} +53 -51
- package/dist/worker/entry.js +31 -29
- package/docs/architecture/architecture.md +1 -0
- package/docs/architecture/context-engine.md +4 -2
- package/docs/architecture/observability.md +6 -4
- package/docs/architecture/prompt-envelope-and-tools.md +1 -1
- package/docs/architecture/session-lifecycle.md +1 -1
- package/docs/architecture/tui-design.md +12 -9
- package/docs/gui/parity/02-slash-and-surfaces.md +1 -1
- package/docs/guide/commands-and-modes.md +91 -5
- package/docs/guide/configuration-and-targets.md +2 -2
- package/docs/guide/context-continuity.md +49 -0
- package/docs/guide/proactive-memory.md +6 -3
- package/library/registry.yaml +292 -32
- package/library/skills/README.md +27 -2
- package/library/skills/context/context-handoff/SKILL.md +10 -4
- package/library/skills/meta/clio-coder-dev/SKILL.md +92 -88
- package/library/skills/meta/clio-coder-dev/evals.md +51 -46
- package/library/skills/meta/clio-coder-dev/plugin.json +2 -2
- package/library/skills/meta/clio-coder-dev/references/change-map.md +49 -0
- package/library/skills/meta/clio-coder-test/SKILL.md +89 -142
- package/library/skills/meta/clio-coder-test/evals.md +58 -62
- package/library/skills/meta/clio-coder-test/plugin.json +2 -2
- package/library/skills/meta/clio-coder-test/references/harness.md +3 -3
- package/library/skills/meta/clio-coder-test/references/lifecycle-validation.md +30 -0
- package/library/skills/meta/clio-coder-test/references/test-map.md +60 -89
- package/library/skills/registry.yaml +5 -5
- package/library/skills/skill-marketplace.json +12 -14
- package/package.json +1 -1
- package/src/cli/skills-eval.ts +1064 -54
- package/src/cli/usage.ts +2 -2
- package/src/core/clio-repo.ts +3 -0
- package/src/core/tool-names.ts +1 -0
- package/src/domains/context/budget/inspection.ts +9 -0
- package/src/domains/context/budget/live-view.ts +418 -0
- package/src/domains/context/budget/pressure.ts +393 -0
- package/src/domains/context/budget/request-fit.ts +15 -0
- package/src/domains/evidence/build.ts +24 -0
- package/src/domains/gateway/mcp/client.ts +1 -1
- package/src/domains/memory/commit-state.ts +160 -0
- package/src/domains/memory/operations.ts +18 -10
- package/src/domains/memory/prompt-cache.ts +95 -0
- package/src/domains/memory/prompt-section.ts +50 -7
- package/src/domains/memory/relevance.ts +93 -0
- package/src/domains/memory/restoration.ts +78 -0
- package/src/domains/memory/store.ts +39 -1
- package/src/domains/middleware/memory-intervention.ts +116 -24
- package/src/domains/observability/background-memory-usage.ts +1 -1
- package/src/domains/observability/cost.ts +3 -3
- package/src/domains/observability/extension.ts +1 -1
- package/src/domains/observability/metrics.ts +1 -1
- package/src/domains/prompts/compiler.ts +1 -1
- package/src/domains/prompts/extension.ts +38 -1
- package/src/domains/quota/anthropic-max-provider.ts +134 -0
- package/src/domains/quota/anthropic-usage.ts +220 -0
- package/src/domains/quota/antigravity-provider.ts +339 -0
- package/src/domains/quota/cache.ts +87 -0
- package/src/domains/quota/claude-code-provider.ts +169 -0
- package/src/domains/quota/codex-provider.ts +237 -0
- package/src/domains/quota/presentation.ts +240 -0
- package/src/domains/quota/registry.ts +23 -0
- package/src/domains/quota/service.ts +93 -0
- package/src/domains/quota/summary-feed.ts +86 -0
- package/src/domains/quota/types.ts +84 -0
- package/src/domains/resources/index.ts +11 -0
- package/src/domains/resources/skills/catalog-view.ts +571 -0
- package/src/domains/resources/skills/lexical-match.ts +136 -0
- package/src/domains/resources/skills/loader.ts +38 -0
- package/src/domains/resources/skills/promotion.ts +1 -55
- package/src/domains/resources/skills/provenance-pin.ts +50 -20
- package/src/domains/safety/action-classifier.ts +1 -0
- package/src/domains/session/compaction/branch-summary.ts +4 -1
- package/src/domains/session/compaction/compact.ts +18 -2
- package/src/domains/session/compaction/cut-point.ts +20 -1
- package/src/domains/session/compaction/tokens.ts +48 -3
- package/src/domains/session/context-accounting.ts +8 -1
- package/src/domains/session/continuity/carry.ts +59 -0
- package/src/domains/session/continuity/contract.ts +592 -0
- package/src/domains/session/continuity/evidence.ts +290 -0
- package/src/domains/session/continuity/fold.ts +1075 -0
- package/src/domains/session/continuity/note.ts +79 -0
- package/src/domains/session/continuity/operator-request.ts +104 -0
- package/src/domains/session/continuity/persistence.ts +408 -0
- package/src/domains/session/continuity/ports.ts +231 -0
- package/src/domains/session/continuity/projection.ts +538 -0
- package/src/domains/session/continuity/validate.ts +352 -0
- package/src/domains/session/entries.ts +50 -3
- package/src/domains/session/index.ts +48 -0
- package/src/domains/session/migrations/index.ts +7 -3
- package/src/domains/session/tree/fork.ts +15 -1
- package/src/domains/session/usage.ts +2 -2
- package/src/engine/acp/commands.ts +1 -1
- package/src/engine/agent.ts +133 -9
- package/src/engine/session.ts +13 -6
- package/src/entry/orchestrator.ts +116 -33
- package/src/interactive/chat-loop-messages.ts +2 -2
- package/src/interactive/chat-loop.ts +219 -63
- package/src/interactive/chat-renderer.ts +99 -8
- package/src/interactive/context-overlay.ts +24 -3
- package/src/interactive/continuity-controller.ts +526 -0
- package/src/interactive/dispatch-board.ts +21 -3
- package/src/interactive/footer/dashboard.ts +25 -1
- package/src/interactive/footer/key-hints.ts +2 -2
- package/src/interactive/footer/pages.ts +70 -27
- package/src/interactive/footer/widgets.ts +32 -7
- package/src/interactive/footer-panel.ts +1 -1
- package/src/interactive/interactive-application.ts +7 -2
- package/src/interactive/interactive-input-runtime.ts +2 -2
- package/src/interactive/interactive-presentation.ts +15 -0
- package/src/interactive/interactive-slash-runtime.ts +45 -8
- package/src/interactive/interactive-tickers.ts +7 -1
- package/src/interactive/model-session-replay.ts +130 -3
- package/src/interactive/output-reserve.ts +35 -0
- package/src/interactive/overlay-general-openers.ts +12 -8
- package/src/interactive/overlay-key-routing.ts +6 -9
- package/src/interactive/overlay-lifecycle.ts +10 -6
- package/src/interactive/overlay-session-lifecycle.ts +21 -14
- package/src/interactive/quota-view.ts +229 -0
- package/src/interactive/session-last-turn.ts +1 -1
- package/src/interactive/session-transcript.ts +2 -1
- package/src/interactive/session-usage-reseed.ts +2 -2
- package/src/interactive/side-question.ts +2 -2
- package/src/interactive/slash-commands.ts +28 -9
- package/src/interactive/turn-context.ts +686 -63
- package/src/interactive/turn-middleware.ts +39 -6
- package/src/interactive/turn-persistence.ts +15 -2
- package/src/interactive/turn-prewarm.ts +1 -1
- package/src/interactive/turn-runtime.ts +48 -5
- package/src/interactive/{cost-overlay.ts → usage-overlay.ts} +121 -33
- package/src/interactive/welcome-dashboard.ts +26 -3
- package/src/tools/agent-tools.ts +3 -1
- package/src/tools/bootstrap.ts +6 -0
- package/src/tools/builtin-tool-catalog.ts +9 -0
- package/src/tools/context/index.ts +148 -113
- package/src/tools/context/surface.ts +11 -7
- package/src/tools/core-bootstrap.ts +4 -0
- package/src/tools/observation.ts +8 -2
- package/src/tools/policy.ts +3 -0
- package/src/tools/self-compact.ts +31 -0
- package/src/tools/surface.ts +1 -0
- package/src/tools/tasks.ts +1 -1
- package/dist/chunk-CTFPFW3H.js +0 -44
- package/dist/chunk-EOJPDNUP.js +0 -1325
- package/dist/chunk-I4Y4WKZR.js +0 -386
package/src/cli/skills-eval.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import { createHash, randomBytes } from "node:crypto";
|
|
3
|
-
import { existsSync } from "node:fs";
|
|
4
|
-
import { cp, mkdir, mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
|
|
3
|
+
import { existsSync, realpathSync } from "node:fs";
|
|
4
|
+
import { cp, mkdir, mkdtemp, readdir, readFile, readlink, realpath, rm, stat, writeFile } from "node:fs/promises";
|
|
5
5
|
import { tmpdir } from "node:os";
|
|
6
|
-
import { join, resolve } from "node:path";
|
|
6
|
+
import { isAbsolute, join, relative as relativePath, resolve } from "node:path";
|
|
7
7
|
import { performance } from "node:perf_hooks";
|
|
8
8
|
import { combineBashOutput, runBashCommand } from "../core/bash-exec.js";
|
|
9
9
|
import { HEADLESS_PERMISSION_DENIED_MARKER } from "../core/headless-permission.js";
|
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
} from "../domains/eval/index.js";
|
|
21
21
|
import { buildEvalEvidence } from "../domains/evidence/index.js";
|
|
22
22
|
import {
|
|
23
|
+
checkSkillDrift,
|
|
23
24
|
discoverMarketplaceSkills,
|
|
24
25
|
loadSkills,
|
|
25
26
|
parseSkillEvals,
|
|
@@ -44,6 +45,22 @@ import { formatColumns, printError } from "./shared.js";
|
|
|
44
45
|
* command lists, receipt-backed token/cost totals when headless main-agent
|
|
45
46
|
* receipts are present, and per-bullet detail in a `skill-eval.json` sidecar
|
|
46
47
|
* registered in the bundle's `overview.json` files list.
|
|
48
|
+
*
|
|
49
|
+
* Two things the sidecar records beyond the rubric result.
|
|
50
|
+
*
|
|
51
|
+
* `subject` names the exact artifact the run measured: the resolved base
|
|
52
|
+
* directory, how it was resolved, both hashes of the SKILL.md, and whether that
|
|
53
|
+
* content still matches whatever hash was recorded for it. A bundle that names
|
|
54
|
+
* only a skill and an evals.md cannot be compared against another bundle for
|
|
55
|
+
* the same skill, because nothing in either says whether the skill changed.
|
|
56
|
+
*
|
|
57
|
+
* `attribution` answers "did the skill change anything?" by scoring the
|
|
58
|
+
* baseline arm too. That arm was always executed and always thrown away: the
|
|
59
|
+
* treatment judge is told the baseline exists only for context. A second,
|
|
60
|
+
* isolated judge scores it against the same bullets, and the two verdict sets
|
|
61
|
+
* are paired per bullet. It is advisory in the strict sense: `pass`,
|
|
62
|
+
* `exitCode` and `failureClass` keep their treatment-only meaning, and no
|
|
63
|
+
* attribution value reaches the exit code. `--no-attribution` skips it.
|
|
47
64
|
*/
|
|
48
65
|
|
|
49
66
|
const DEFAULT_RUN_TIMEOUT_MS = 600_000;
|
|
@@ -82,6 +99,14 @@ export interface SkillsEvalOptions {
|
|
|
82
99
|
* fetch the open web is measuring something else as well.
|
|
83
100
|
*/
|
|
84
101
|
allowNetwork: boolean;
|
|
102
|
+
/**
|
|
103
|
+
* Skip the baseline judge, and with it the paired attribution.
|
|
104
|
+
*
|
|
105
|
+
* Attribution costs one extra judge run per scenario. Turning it off is a
|
|
106
|
+
* cost decision, not a result: a skipped comparison is recorded
|
|
107
|
+
* `not-attempted` with the reason, never `no-change`.
|
|
108
|
+
*/
|
|
109
|
+
noAttribution: boolean;
|
|
85
110
|
scenario?: string;
|
|
86
111
|
target?: string;
|
|
87
112
|
timeoutSeconds?: number;
|
|
@@ -104,6 +129,18 @@ export interface ScoredBullet {
|
|
|
104
129
|
reason: string;
|
|
105
130
|
}
|
|
106
131
|
|
|
132
|
+
/**
|
|
133
|
+
* One skill activation this run actually performed, as the activation contract
|
|
134
|
+
* reported it. `hash` is the sha256 of the bytes the child read.
|
|
135
|
+
*
|
|
136
|
+
* @internal Exported for contract tests.
|
|
137
|
+
*/
|
|
138
|
+
export interface ObservedActivation {
|
|
139
|
+
name: string;
|
|
140
|
+
hash: string;
|
|
141
|
+
path: string;
|
|
142
|
+
}
|
|
143
|
+
|
|
107
144
|
/** Exported for contracts tests. */
|
|
108
145
|
export interface CapturedRun {
|
|
109
146
|
sessionId: string | null;
|
|
@@ -113,6 +150,8 @@ export interface CapturedRun {
|
|
|
113
150
|
timedOut: boolean;
|
|
114
151
|
wallTimeMs: number;
|
|
115
152
|
stderr: string;
|
|
153
|
+
/** Skill loads this run completed successfully; empty when none did. */
|
|
154
|
+
activations: ObservedActivation[];
|
|
116
155
|
}
|
|
117
156
|
|
|
118
157
|
interface ScenarioUsage {
|
|
@@ -121,12 +160,141 @@ interface ScenarioUsage {
|
|
|
121
160
|
harness: EvalHarnessMetrics;
|
|
122
161
|
}
|
|
123
162
|
|
|
163
|
+
/**
|
|
164
|
+
* The exact artifact a run measured.
|
|
165
|
+
*
|
|
166
|
+
* A bundle used to name only its `evals.md` by hash, so two runs of the same
|
|
167
|
+
* skill could not be told apart when the SKILL.md between them had changed.
|
|
168
|
+
* Everything here is already computed elsewhere in the run: the loader hashes
|
|
169
|
+
* the file, `resolveSkillBaseDir` resolves which copy activation would pick,
|
|
170
|
+
* and `checkSkillDrift` compares it against whatever recorded hash speaks for
|
|
171
|
+
* it. Discarding all of it was the whole gap.
|
|
172
|
+
*
|
|
173
|
+
* @internal Exported for contract tests.
|
|
174
|
+
*/
|
|
175
|
+
export interface SkillEvalSubject {
|
|
176
|
+
name: string;
|
|
177
|
+
baseDir: string;
|
|
178
|
+
/** How the copy was resolved: `path`, `<source>/<scope>`, or `catalog`. */
|
|
179
|
+
origin: string;
|
|
180
|
+
/** sha256 of the SKILL.md exactly as read. */
|
|
181
|
+
sha256: string;
|
|
182
|
+
/** sha256 with install-lifecycle provenance stripped; this is what pins compare. */
|
|
183
|
+
normalizedHash: string;
|
|
184
|
+
/**
|
|
185
|
+
* Digest over every file under `baseDir`, not just SKILL.md.
|
|
186
|
+
*
|
|
187
|
+
* A skill is a directory: the body can point at `references/` and scripts
|
|
188
|
+
* the run will read. Naming the artifact by its SKILL.md hash alone would
|
|
189
|
+
* claim an identity for content that hash does not cover, and would miss an
|
|
190
|
+
* edit to a reference file entirely. Null when the tree could not be read.
|
|
191
|
+
*/
|
|
192
|
+
treeSha256: string | null;
|
|
193
|
+
/**
|
|
194
|
+
* Whether the artifact could be snapshotted for the run. False when the body
|
|
195
|
+
* carries package references, which resolve against an owning package root
|
|
196
|
+
* above the skill directory: a copy of the directory alone would deliver
|
|
197
|
+
* different instructions from the live path.
|
|
198
|
+
*/
|
|
199
|
+
pinnable: boolean;
|
|
200
|
+
evalsPath: string;
|
|
201
|
+
evalsSha256: string;
|
|
202
|
+
/** Null when nothing on this machine recorded a hash for the skill. */
|
|
203
|
+
drift: { verdict: "match" | "mismatch"; authority: string; expected: string } | null;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Whether the artifact on disk still matched {@link SkillEvalSubject} across
|
|
208
|
+
* one scenario's treatment arm.
|
|
209
|
+
*
|
|
210
|
+
* `verified` states exactly one thing: the tree digest taken immediately before
|
|
211
|
+
* the treatment arm and again immediately after both equalled the run-level
|
|
212
|
+
* subject. It is not proof of what the child activated. The treatment child
|
|
213
|
+
* loads the skill from the live source directory by path, so an edit landing
|
|
214
|
+
* between resolution and the arm, or between two scenarios, would otherwise be
|
|
215
|
+
* measured under the previous artifact's recorded identity. This check closes
|
|
216
|
+
* that window; it cannot see a change reverted entirely inside the arm, and the
|
|
217
|
+
* sidecar says so rather than implying activation was witnessed.
|
|
218
|
+
*/
|
|
219
|
+
export type SubjectVerification =
|
|
220
|
+
| "verified"
|
|
221
|
+
| "mismatch"
|
|
222
|
+
| "unreadable"
|
|
223
|
+
| "not-pinned"
|
|
224
|
+
| "not-activated"
|
|
225
|
+
| "activation-mismatch"
|
|
226
|
+
| "not-checked";
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* Whether the skill changed the outcome for one bullet, relative to the same
|
|
230
|
+
* bullet in the baseline arm.
|
|
231
|
+
*
|
|
232
|
+
* `unmeasured` is not a comparison that came out even. It is the absence of a
|
|
233
|
+
* comparison, and it is returned whenever either arm failed to produce a
|
|
234
|
+
* verdict for that bullet.
|
|
235
|
+
*
|
|
236
|
+
* @internal Exported for contract tests.
|
|
237
|
+
*/
|
|
238
|
+
export type BulletAttribution = "helped" | "regressed" | "no-change-pass" | "no-change-fail" | "unmeasured";
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* The scenario-level rollup.
|
|
242
|
+
*
|
|
243
|
+
* `not-attempted` and `unmeasured` are deliberately separate. The first means
|
|
244
|
+
* the comparison was never run, the second means it was impossible. Collapsing
|
|
245
|
+
* either into `no-change` would report "the skill made no difference" about a
|
|
246
|
+
* measurement that never happened.
|
|
247
|
+
*
|
|
248
|
+
* @internal Exported for contract tests.
|
|
249
|
+
*/
|
|
250
|
+
export type ScenarioAttributionVerdict =
|
|
251
|
+
| "helped"
|
|
252
|
+
| "regressed"
|
|
253
|
+
| "mixed"
|
|
254
|
+
| "no-change"
|
|
255
|
+
| "unmeasured"
|
|
256
|
+
| "not-attempted";
|
|
257
|
+
|
|
258
|
+
/** @internal Exported for contract tests. */
|
|
259
|
+
export interface AttributedBullet {
|
|
260
|
+
index: number;
|
|
261
|
+
text: string;
|
|
262
|
+
baseline: BulletVerdict;
|
|
263
|
+
treatment: BulletVerdict;
|
|
264
|
+
attribution: BulletAttribution;
|
|
265
|
+
/** Why this bullet is unmeasured; absent once both arms supplied evidence. */
|
|
266
|
+
reason?: string;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/** @internal Exported for contract tests. */
|
|
270
|
+
export interface ScenarioAttribution {
|
|
271
|
+
verdict: ScenarioAttributionVerdict;
|
|
272
|
+
/** Why the verdict is `unmeasured` or `not-attempted`; null once a real comparison ran. */
|
|
273
|
+
reason: string | null;
|
|
274
|
+
bullets: AttributedBullet[];
|
|
275
|
+
counts: {
|
|
276
|
+
helped: number;
|
|
277
|
+
regressed: number;
|
|
278
|
+
noChangePass: number;
|
|
279
|
+
noChangeFail: number;
|
|
280
|
+
unmeasured: number;
|
|
281
|
+
};
|
|
282
|
+
}
|
|
283
|
+
|
|
124
284
|
interface ScenarioOutcome {
|
|
125
285
|
scenario: SkillEvalScenario;
|
|
126
286
|
bullets: ScoredBullet[];
|
|
127
287
|
baseline: CapturedRun | null;
|
|
128
288
|
treatment: CapturedRun | null;
|
|
129
289
|
judge: CapturedRun | null;
|
|
290
|
+
/** The isolated judge run that scored the baseline arm; null when none ran. */
|
|
291
|
+
baselineJudge: CapturedRun | null;
|
|
292
|
+
/** Whether the artifact on disk still matched the subject across this scenario. */
|
|
293
|
+
subjectVerification: SubjectVerification;
|
|
294
|
+
/** Tree digest observed after this scenario's treatment arm; null when unread. */
|
|
295
|
+
observedTreeSha256: string | null;
|
|
296
|
+
/** Advisory paired comparison. Never a gate, never folded into `pass`. */
|
|
297
|
+
attribution: ScenarioAttribution;
|
|
130
298
|
/** Seed workspace cloned for both arms; removed after the run, kept as a record. */
|
|
131
299
|
workspace: string;
|
|
132
300
|
wallTimeMs: number;
|
|
@@ -134,6 +302,211 @@ interface ScenarioOutcome {
|
|
|
134
302
|
usage: ScenarioUsage;
|
|
135
303
|
}
|
|
136
304
|
|
|
305
|
+
/**
|
|
306
|
+
* Did the skill change this bullet's outcome?
|
|
307
|
+
*
|
|
308
|
+
* Either arm failing to produce a verdict makes the pair unmeasurable, and that
|
|
309
|
+
* check comes first: a bullet the baseline judge never scored carries no
|
|
310
|
+
* information about the treatment, whatever the treatment did.
|
|
311
|
+
*
|
|
312
|
+
* @internal Exported for contract tests.
|
|
313
|
+
*/
|
|
314
|
+
export function deriveBulletAttribution(baseline: BulletVerdict, treatment: BulletVerdict): BulletAttribution {
|
|
315
|
+
if (baseline === "unmeasured" || baseline === "error") return "unmeasured";
|
|
316
|
+
if (treatment === "unmeasured" || treatment === "error") return "unmeasured";
|
|
317
|
+
if (baseline === "fail" && treatment === "pass") return "helped";
|
|
318
|
+
if (baseline === "pass" && treatment === "fail") return "regressed";
|
|
319
|
+
return baseline === "pass" ? "no-change-pass" : "no-change-fail";
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Roll per-bullet attributions into one scenario verdict.
|
|
324
|
+
*
|
|
325
|
+
* Order matters and is deliberate. An unmeasured bullet poisons the scenario,
|
|
326
|
+
* because a rollup that ignored it would report a comparison over a subset
|
|
327
|
+
* while naming the whole. `mixed` outranks both single-direction verdicts, so a
|
|
328
|
+
* skill that fixed three bullets and broke one is never reported as simply
|
|
329
|
+
* having helped.
|
|
330
|
+
*
|
|
331
|
+
* @internal Exported for contract tests.
|
|
332
|
+
*/
|
|
333
|
+
export function deriveScenarioAttributionVerdict(
|
|
334
|
+
bullets: ReadonlyArray<AttributedBullet>,
|
|
335
|
+
): Exclude<ScenarioAttributionVerdict, "not-attempted"> {
|
|
336
|
+
if (bullets.length === 0) return "unmeasured";
|
|
337
|
+
if (bullets.some((bullet) => bullet.attribution === "unmeasured")) return "unmeasured";
|
|
338
|
+
const helped = bullets.some((bullet) => bullet.attribution === "helped");
|
|
339
|
+
const regressed = bullets.some((bullet) => bullet.attribution === "regressed");
|
|
340
|
+
if (helped && regressed) return "mixed";
|
|
341
|
+
if (regressed) return "regressed";
|
|
342
|
+
if (helped) return "helped";
|
|
343
|
+
return "no-change";
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
function attributionCounts(bullets: ReadonlyArray<AttributedBullet>): ScenarioAttribution["counts"] {
|
|
347
|
+
const counts = { helped: 0, regressed: 0, noChangePass: 0, noChangeFail: 0, unmeasured: 0 };
|
|
348
|
+
for (const bullet of bullets) {
|
|
349
|
+
if (bullet.attribution === "helped") counts.helped += 1;
|
|
350
|
+
else if (bullet.attribution === "regressed") counts.regressed += 1;
|
|
351
|
+
else if (bullet.attribution === "no-change-pass") counts.noChangePass += 1;
|
|
352
|
+
else if (bullet.attribution === "no-change-fail") counts.noChangeFail += 1;
|
|
353
|
+
else counts.unmeasured += 1;
|
|
354
|
+
}
|
|
355
|
+
return counts;
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/**
|
|
359
|
+
* Pair two arms' strict verdicts into a per-bullet comparison.
|
|
360
|
+
*
|
|
361
|
+
* Both sides have to have supplied unambiguous evidence for the same bullet.
|
|
362
|
+
* When either did not, the bullet is `unmeasured` and carries the reason it was
|
|
363
|
+
* refused, so a reader can tell "the judge omitted it" from "the judge answered
|
|
364
|
+
* with a string" from "the judge contradicted itself". The reason used to be
|
|
365
|
+
* dropped during pairing, which left an unmeasured scenario with nothing saying
|
|
366
|
+
* why.
|
|
367
|
+
*
|
|
368
|
+
* @internal Exported for contract tests.
|
|
369
|
+
*/
|
|
370
|
+
export function attributionFromStrict(
|
|
371
|
+
scenario: SkillEvalScenario,
|
|
372
|
+
baseline: StrictJudgeVerdicts,
|
|
373
|
+
treatment: StrictJudgeVerdicts,
|
|
374
|
+
): ScenarioAttribution {
|
|
375
|
+
const verdictOf = (pass: boolean | undefined): BulletVerdict =>
|
|
376
|
+
pass === undefined ? "unmeasured" : pass ? "pass" : "fail";
|
|
377
|
+
const refusal = (side: string, strict: StrictJudgeVerdicts, index: number): string | null => {
|
|
378
|
+
if (strict.verdicts.has(index)) return null;
|
|
379
|
+
if (strict.absent !== null) return `${side} judge: ${strict.absent}`;
|
|
380
|
+
return `${side} judge: ${strict.rejected.get(index) ?? `no verdict for bullet ${index}`}`;
|
|
381
|
+
};
|
|
382
|
+
const bullets: AttributedBullet[] = scenario.expected.map((text, i) => {
|
|
383
|
+
const index = i + 1;
|
|
384
|
+
const baselinePass = baseline.verdicts.get(index);
|
|
385
|
+
const treatmentPass = treatment.verdicts.get(index);
|
|
386
|
+
const baselineVerdict = verdictOf(baselinePass);
|
|
387
|
+
const treatmentVerdict = verdictOf(treatmentPass);
|
|
388
|
+
const reason = refusal("baseline", baseline, index) ?? refusal("treatment", treatment, index);
|
|
389
|
+
return {
|
|
390
|
+
index,
|
|
391
|
+
text,
|
|
392
|
+
baseline: baselineVerdict,
|
|
393
|
+
treatment: treatmentVerdict,
|
|
394
|
+
attribution: deriveBulletAttribution(baselineVerdict, treatmentVerdict),
|
|
395
|
+
...(reason !== null ? { reason } : {}),
|
|
396
|
+
};
|
|
397
|
+
});
|
|
398
|
+
const verdict = deriveScenarioAttributionVerdict(bullets);
|
|
399
|
+
const firstReason = bullets.find((bullet) => bullet.reason !== undefined)?.reason ?? null;
|
|
400
|
+
return {
|
|
401
|
+
verdict,
|
|
402
|
+
reason: verdict === "unmeasured" ? firstReason : null,
|
|
403
|
+
bullets,
|
|
404
|
+
counts: attributionCounts(bullets),
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
/**
|
|
409
|
+
* A digest of the whole skill directory, path-sensitive and order-independent.
|
|
410
|
+
*
|
|
411
|
+
* Sorted relative paths are hashed alongside their contents, so adding, moving
|
|
412
|
+
* or removing a reference file changes the digest as surely as editing one
|
|
413
|
+
* does. Null on any read failure rather than a partial digest, because a digest
|
|
414
|
+
* over some of a tree would compare unequal for a reason that is not a change.
|
|
415
|
+
*
|
|
416
|
+
* @internal Exported for contract tests.
|
|
417
|
+
*/
|
|
418
|
+
export async function skillTreeDigest(baseDir: string): Promise<string | null> {
|
|
419
|
+
interface Entry {
|
|
420
|
+
relative: string;
|
|
421
|
+
kind: "file" | "symlink";
|
|
422
|
+
target?: string;
|
|
423
|
+
}
|
|
424
|
+
const root = resolve(baseDir);
|
|
425
|
+
const entries: Entry[] = [];
|
|
426
|
+
const seenDirs = new Set<string>();
|
|
427
|
+
const walk = async (dir: string): Promise<void> => {
|
|
428
|
+
// A directory symlink pointing at an ancestor would otherwise walk forever.
|
|
429
|
+
const real = await realpath(dir);
|
|
430
|
+
if (seenDirs.has(real)) throw new Error("cycle");
|
|
431
|
+
seenDirs.add(real);
|
|
432
|
+
const found = await readdir(dir, { withFileTypes: true });
|
|
433
|
+
for (const entry of found.sort((a, b) => a.name.localeCompare(b.name))) {
|
|
434
|
+
const full = join(dir, entry.name);
|
|
435
|
+
if (entry.isSymbolicLink()) {
|
|
436
|
+
// A link is part of the artifact's identity: retargeting one changes
|
|
437
|
+
// what the skill reads without changing any regular file. The link's
|
|
438
|
+
// own target string is hashed, and a link leaving the tree makes the
|
|
439
|
+
// artifact unverifiable rather than silently half-covered.
|
|
440
|
+
const target = await readlink(full);
|
|
441
|
+
const resolved = resolve(dir, target);
|
|
442
|
+
const relative = relativePath(root, resolved);
|
|
443
|
+
if (relative.startsWith("..") || isAbsolute(relative)) throw new Error("escaping symlink");
|
|
444
|
+
entries.push({ relative: full.slice(root.length), kind: "symlink", target });
|
|
445
|
+
continue;
|
|
446
|
+
}
|
|
447
|
+
if (entry.isDirectory()) await walk(full);
|
|
448
|
+
else if (entry.isFile()) entries.push({ relative: full.slice(root.length), kind: "file" });
|
|
449
|
+
}
|
|
450
|
+
};
|
|
451
|
+
try {
|
|
452
|
+
await walk(root);
|
|
453
|
+
const hash = createHash("sha256");
|
|
454
|
+
for (const entry of entries.sort((a, b) => a.relative.localeCompare(b.relative))) {
|
|
455
|
+
hash.update(entry.relative, "utf8");
|
|
456
|
+
hash.update("\0");
|
|
457
|
+
hash.update(entry.kind, "utf8");
|
|
458
|
+
hash.update("\0");
|
|
459
|
+
if (entry.kind === "symlink") hash.update(entry.target ?? "", "utf8");
|
|
460
|
+
else hash.update(await readFile(join(root, entry.relative)));
|
|
461
|
+
hash.update("\0");
|
|
462
|
+
}
|
|
463
|
+
return hash.digest("hex");
|
|
464
|
+
} catch {
|
|
465
|
+
return null;
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
/**
|
|
470
|
+
* Whether the artifact can be snapshotted for the run.
|
|
471
|
+
*
|
|
472
|
+
* The treatment arm runs against a private copy rather than the live source
|
|
473
|
+
* directory. Hashing the source before and after the arm catches an edit that
|
|
474
|
+
* persists and nothing else: an A to B to A change inside the arm leaves both
|
|
475
|
+
* observations equal while the child read B. A copy under a per-run temp root
|
|
476
|
+
* is not reachable from the source path, so an ordinary edit there cannot
|
|
477
|
+
* affect the run.
|
|
478
|
+
*
|
|
479
|
+
* What that is, precisely: an isolated snapshot, not an immutable pin. The
|
|
480
|
+
* copy sits in a writable temp directory, and the arm runs at full-auto, so a
|
|
481
|
+
* model that chose to edit the copy, read it, and restore it would not be
|
|
482
|
+
* detected. The harness has no mechanism that would make the copy read-only to
|
|
483
|
+
* its own child, and building one is out of scope here. These results are
|
|
484
|
+
* evidence about a cooperative model, which is the same caveat
|
|
485
|
+
* `materializeSkillEvalWorkspaces` already records about arm isolation.
|
|
486
|
+
*
|
|
487
|
+
* Returns false for the one case a copy would misrepresent: a body carrying
|
|
488
|
+
* package references. Those resolve against the owning package root, which
|
|
489
|
+
* sits above the skill directory and is not part of the copy, so a copied
|
|
490
|
+
* skill would deliver different instructions from the ones the live path
|
|
491
|
+
* delivers. A tree whose digest cannot be taken is a separate failure and is
|
|
492
|
+
* reported by `skillTreeDigest` returning null.
|
|
493
|
+
*
|
|
494
|
+
* @internal Exported for contract tests.
|
|
495
|
+
*/
|
|
496
|
+
export function artifactIsPinnable(skillBody: string): boolean {
|
|
497
|
+
return !skillBody.includes("${pluginRoot}") && !skillBody.includes("${component:");
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
/** No comparison was attempted. Records why, and never reads as `no-change`. */
|
|
501
|
+
function notAttemptedAttribution(reason: string): ScenarioAttribution {
|
|
502
|
+
return { verdict: "not-attempted", reason, bullets: [], counts: attributionCounts([]) };
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
/** A comparison was impossible. Distinct from one that was never started. */
|
|
506
|
+
function unmeasurableAttribution(reason: string): ScenarioAttribution {
|
|
507
|
+
return { verdict: "unmeasured", reason, bullets: [], counts: attributionCounts([]) };
|
|
508
|
+
}
|
|
509
|
+
|
|
137
510
|
async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptions): Promise<number> {
|
|
138
511
|
const resolved = resolveSkillBaseDir(nameOrPath, process.cwd());
|
|
139
512
|
if (resolved.baseDir === null) {
|
|
@@ -164,6 +537,42 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
|
|
|
164
537
|
for (const diagnostic of parsed.diagnostics) {
|
|
165
538
|
process.stderr.write(`clio-coder eval skill: ${diagnostic}\n`);
|
|
166
539
|
}
|
|
540
|
+
// Coherence check on the subject itself. `skill` was loaded, and its hashes
|
|
541
|
+
// taken, before the tree digest; an edit landing in that gap would produce a
|
|
542
|
+
// subject naming one SKILL.md revision and a tree containing another. The
|
|
543
|
+
// SKILL.md is re-read here and compared, so the recorded identity describes
|
|
544
|
+
// one state of the directory or admits it could not.
|
|
545
|
+
const subjectTree = await skillTreeDigest(resolved.baseDir);
|
|
546
|
+
const reread = await readFile(join(resolved.baseDir, "SKILL.md"), "utf8").catch(() => null);
|
|
547
|
+
const coherent = reread !== null && createHash("sha256").update(reread, "utf8").digest("hex") === skill.hash;
|
|
548
|
+
const pinnable = reread !== null && artifactIsPinnable(reread);
|
|
549
|
+
const driftReport = checkSkillDrift(skill, process.cwd());
|
|
550
|
+
const subject: SkillEvalSubject = {
|
|
551
|
+
name: skill.name,
|
|
552
|
+
baseDir: resolved.baseDir,
|
|
553
|
+
origin: resolved.origin ?? "unknown",
|
|
554
|
+
sha256: skill.hash,
|
|
555
|
+
normalizedHash: skill.normalizedHash,
|
|
556
|
+
treeSha256: coherent ? subjectTree : null,
|
|
557
|
+
pinnable,
|
|
558
|
+
evalsPath,
|
|
559
|
+
evalsSha256: createHash("sha256").update(evalsRaw, "utf8").digest("hex"),
|
|
560
|
+
drift:
|
|
561
|
+
driftReport === null
|
|
562
|
+
? null
|
|
563
|
+
: { verdict: driftReport.verdict, authority: driftReport.authority, expected: driftReport.expected },
|
|
564
|
+
};
|
|
565
|
+
// Said once, before any arm runs. A bundle measured against content that no
|
|
566
|
+
// longer matches its recorded form is evidence about a different artifact
|
|
567
|
+
// than the one it names, and the reader has to know that up front. It does
|
|
568
|
+
// not block: drift never gates activation either.
|
|
569
|
+
if (subject.drift?.verdict === "mismatch") {
|
|
570
|
+
process.stderr.write(
|
|
571
|
+
`clio-coder eval skill: WARNING skill_drift: ${skill.name} content (sha256 ${skill.normalizedHash.slice(0, 12)}…) ` +
|
|
572
|
+
`does not match the hash recorded for it by the ${subject.drift.authority} (expected ${subject.drift.expected.slice(0, 12)}…); ` +
|
|
573
|
+
"this run measures the copy on disk, not the recorded one\n",
|
|
574
|
+
);
|
|
575
|
+
}
|
|
167
576
|
const matcher = options.scenario === undefined ? null : scenarioMatcher(options.scenario);
|
|
168
577
|
if (options.scenario !== undefined && matcher === null) {
|
|
169
578
|
printError(`invalid --scenario "${options.scenario}": use a scenario id like S1 or a bare number`);
|
|
@@ -189,24 +598,39 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
|
|
|
189
598
|
}
|
|
190
599
|
const startedAt = new Date().toISOString();
|
|
191
600
|
const outcomes: ScenarioOutcome[] = [];
|
|
601
|
+
const attributionEnabled = !options.noAttribution;
|
|
192
602
|
for (const scenario of scenarios) {
|
|
193
|
-
process.stderr.write(
|
|
603
|
+
process.stderr.write(
|
|
604
|
+
`clio-coder eval skill: ${skill.name} ${scenario.id} baseline/treatment/judge${attributionEnabled ? "/baseline-judge" : ""}...\n`,
|
|
605
|
+
);
|
|
194
606
|
outcomes.push(
|
|
195
|
-
await runScenario(
|
|
196
|
-
skill.name,
|
|
197
|
-
resolved.baseDir,
|
|
607
|
+
await runScenario({
|
|
608
|
+
skillName: skill.name,
|
|
609
|
+
skillBaseDir: resolved.baseDir,
|
|
198
610
|
scenario,
|
|
199
|
-
options.target,
|
|
611
|
+
target: options.target,
|
|
200
612
|
timeoutMs,
|
|
201
613
|
workspaceOverride,
|
|
202
|
-
options.trustFixtures,
|
|
614
|
+
trustFixtures: options.trustFixtures,
|
|
203
615
|
childEnv,
|
|
204
|
-
|
|
616
|
+
attributionEnabled,
|
|
617
|
+
subjectTreeSha256: subject.treeSha256,
|
|
618
|
+
subjectSha256: subject.sha256,
|
|
619
|
+
pinnable: subject.pinnable,
|
|
620
|
+
}),
|
|
621
|
+
);
|
|
622
|
+
}
|
|
623
|
+
const unverified = outcomes.filter((outcome) => outcome.subjectVerification === "mismatch");
|
|
624
|
+
if (unverified.length > 0) {
|
|
625
|
+
process.stderr.write(
|
|
626
|
+
`clio-coder eval skill: WARNING subject_changed: the skill tree at ${resolved.baseDir} changed during ${unverified
|
|
627
|
+
.map((outcome) => outcome.scenario.id)
|
|
628
|
+
.join(", ")}; those scenarios' rubric results stand but their baseline comparison is unmeasured\n`,
|
|
205
629
|
);
|
|
206
630
|
}
|
|
207
631
|
const endedAt = new Date().toISOString();
|
|
208
632
|
|
|
209
|
-
const artifact = synthesizeArtifact(
|
|
633
|
+
const artifact = synthesizeArtifact(subject, evalsPath, evalsRaw, startedAt, endedAt, outcomes, options.target);
|
|
210
634
|
let evidenceId: string | null = null;
|
|
211
635
|
let evidenceDirectory: string | null = null;
|
|
212
636
|
const evidenceErrors: string[] = [];
|
|
@@ -222,7 +646,7 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
|
|
|
222
646
|
evidenceDirectory = built.directory;
|
|
223
647
|
await writeFile(
|
|
224
648
|
join(built.directory, SKILL_EVAL_SIDECAR),
|
|
225
|
-
`${JSON.stringify(sidecar(
|
|
649
|
+
`${JSON.stringify(sidecar(subject, artifact.evalId, outcomes, options.allowNetwork, attributionEnabled), null, 2)}\n`,
|
|
226
650
|
"utf8",
|
|
227
651
|
);
|
|
228
652
|
} catch (error) {
|
|
@@ -234,23 +658,35 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
|
|
|
234
658
|
|
|
235
659
|
if (options.json) {
|
|
236
660
|
for (const outcome of outcomes) {
|
|
661
|
+
const attributed = new Map(outcome.attribution.bullets.map((item) => [item.index, item]));
|
|
237
662
|
for (const bullet of outcome.bullets) {
|
|
663
|
+
const pair = attributed.get(bullet.index);
|
|
238
664
|
process.stdout.write(
|
|
239
665
|
`${JSON.stringify({
|
|
240
666
|
schema: "experimental",
|
|
241
667
|
kind: "skill-eval-bullet",
|
|
242
668
|
skill: skill.name,
|
|
669
|
+
skillSha256: subject.normalizedHash,
|
|
670
|
+
skillOrigin: subject.origin,
|
|
671
|
+
skillDrift: subject.drift?.verdict ?? null,
|
|
243
672
|
scenario: outcome.scenario.id,
|
|
244
673
|
title: outcome.scenario.title,
|
|
245
674
|
bullet: bullet.index,
|
|
246
675
|
expected: bullet.text,
|
|
247
676
|
verdict: bullet.verdict,
|
|
248
677
|
reason: bullet.reason,
|
|
678
|
+
// Advisory, never a gate: `verdict` above is the rubric result and
|
|
679
|
+
// is unchanged by anything here.
|
|
680
|
+
baselineVerdict: pair?.baseline ?? null,
|
|
681
|
+
attribution: pair?.attribution ?? null,
|
|
682
|
+
scenarioAttribution: outcome.attribution.verdict,
|
|
683
|
+
attributionReason: outcome.attribution.reason,
|
|
249
684
|
network: networkPolicyLabel(options.allowNetwork),
|
|
250
685
|
autonomy: ARM_AUTONOMY,
|
|
251
686
|
baselineSessionId: outcome.baseline?.sessionId ?? null,
|
|
252
687
|
treatmentSessionId: outcome.treatment?.sessionId ?? null,
|
|
253
688
|
judgeSessionId: outcome.judge?.sessionId ?? null,
|
|
689
|
+
baselineJudgeSessionId: outcome.baselineJudge?.sessionId ?? null,
|
|
254
690
|
evalId: artifact.evalId,
|
|
255
691
|
evidenceId,
|
|
256
692
|
})}\n`,
|
|
@@ -258,7 +694,7 @@ async function runSkillsEvalCommand(nameOrPath: string, options: SkillsEvalOptio
|
|
|
258
694
|
}
|
|
259
695
|
}
|
|
260
696
|
} else {
|
|
261
|
-
printHumanReport(
|
|
697
|
+
printHumanReport(subject, outcomes, artifact.evalId, evidenceId, evidenceDirectory, options.allowNetwork);
|
|
262
698
|
}
|
|
263
699
|
const anyFailure = outcomes.some((outcome) =>
|
|
264
700
|
outcome.bullets.some((bullet) => bullet.verdict === "fail" || bullet.verdict === "error"),
|
|
@@ -300,7 +736,16 @@ function describeArmPolicyOutcome(allowNetwork: boolean): string {
|
|
|
300
736
|
return `policy: the baseline and treatment arms ran at autonomy ${ARM_AUTONOMY}; ${network}`;
|
|
301
737
|
}
|
|
302
738
|
|
|
303
|
-
type EvalArm = "baseline" | "treatment" | "judge";
|
|
739
|
+
type EvalArm = "baseline" | "treatment" | "judge" | "baseline-judge";
|
|
740
|
+
|
|
741
|
+
/**
|
|
742
|
+
* A judge arm scores text and is told to call no tools, so granting it
|
|
743
|
+
* unattended write and exec buys nothing. Both judge arms are excluded from
|
|
744
|
+
* `--autonomy full-auto` for that reason.
|
|
745
|
+
*/
|
|
746
|
+
function isJudgeArm(arm: EvalArm): boolean {
|
|
747
|
+
return arm === "judge" || arm === "baseline-judge";
|
|
748
|
+
}
|
|
304
749
|
|
|
305
750
|
/**
|
|
306
751
|
* The argv for one arm's child `clio-coder run`. Every arm streams the full JSON
|
|
@@ -318,7 +763,7 @@ function armRunArgs(
|
|
|
318
763
|
options: { target?: string | undefined; skillBaseDir?: string | undefined } = {},
|
|
319
764
|
): string[] {
|
|
320
765
|
const args = ["run", "--json", "--json-events", "full", "--no-skills"];
|
|
321
|
-
if (arm
|
|
766
|
+
if (!isJudgeArm(arm)) args.push("--autonomy", ARM_AUTONOMY);
|
|
322
767
|
if (options.skillBaseDir !== undefined) args.push("--skill", options.skillBaseDir);
|
|
323
768
|
if (options.target !== undefined) args.push("--target", options.target);
|
|
324
769
|
args.push(prompt);
|
|
@@ -432,22 +877,81 @@ async function resolveWorkspaceOverride(workspace: string | undefined): Promise<
|
|
|
432
877
|
}
|
|
433
878
|
}
|
|
434
879
|
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
880
|
+
/**
|
|
881
|
+
* One arm's child process, behind a seam.
|
|
882
|
+
*
|
|
883
|
+
* The default is {@link captureHeadlessRun}, which spawns a real
|
|
884
|
+
* `clio-coder run`. Contract tests substitute a scripted runner so the arm
|
|
885
|
+
* sequencing, the skip gates and the recorded reasons are exercised through the
|
|
886
|
+
* production orchestration rather than by constructing already-correct typed
|
|
887
|
+
* data, and without paying for an inference.
|
|
888
|
+
*
|
|
889
|
+
* @internal
|
|
890
|
+
*/
|
|
891
|
+
export type SkillEvalArmRunner = (
|
|
892
|
+
args: ReadonlyArray<string>,
|
|
893
|
+
cwd: string,
|
|
440
894
|
timeoutMs: number,
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
895
|
+
env: NodeJS.ProcessEnv,
|
|
896
|
+
) => Promise<CapturedRun>;
|
|
897
|
+
|
|
898
|
+
/** @internal Exported for contract tests. */
|
|
899
|
+
export interface RunScenarioInput {
|
|
900
|
+
skillName: string;
|
|
901
|
+
skillBaseDir: string;
|
|
902
|
+
scenario: SkillEvalScenario;
|
|
903
|
+
target: string | undefined;
|
|
904
|
+
timeoutMs: number;
|
|
905
|
+
workspaceOverride: string | null;
|
|
906
|
+
trustFixtures: boolean;
|
|
907
|
+
childEnv: NodeJS.ProcessEnv;
|
|
908
|
+
attributionEnabled: boolean;
|
|
909
|
+
/** Tree digest recorded at resolution; the scenario re-checks against it. */
|
|
910
|
+
subjectTreeSha256: string | null;
|
|
911
|
+
/** sha256 of the subject's SKILL.md, compared against the activation receipt. */
|
|
912
|
+
subjectSha256: string;
|
|
913
|
+
/** False when the artifact cannot be snapshotted faithfully; see artifactIsPinnable. */
|
|
914
|
+
pinnable: boolean;
|
|
915
|
+
runner?: SkillEvalArmRunner;
|
|
916
|
+
}
|
|
917
|
+
|
|
918
|
+
/** @internal Exported for contract tests. */
|
|
919
|
+
export async function runScenario(input: RunScenarioInput): Promise<ScenarioOutcome> {
|
|
920
|
+
const {
|
|
921
|
+
skillName,
|
|
922
|
+
skillBaseDir,
|
|
923
|
+
scenario,
|
|
924
|
+
target,
|
|
925
|
+
timeoutMs,
|
|
926
|
+
workspaceOverride,
|
|
927
|
+
trustFixtures,
|
|
928
|
+
childEnv,
|
|
929
|
+
attributionEnabled,
|
|
930
|
+
subjectTreeSha256,
|
|
931
|
+
subjectSha256,
|
|
932
|
+
pinnable,
|
|
933
|
+
} = input;
|
|
934
|
+
const runArm = input.runner ?? captureHeadlessRun;
|
|
445
935
|
// Published per-scenario figure: monotonic so a clock correction during a
|
|
446
936
|
// long sweep cannot land in one row's wall time.
|
|
447
937
|
const scenarioStart = performance.now();
|
|
448
938
|
const workspace = await mkdtemp(join(tmpdir(), "clio-coder-skill-eval-seed-"));
|
|
449
939
|
let runWorkspaces: MaterializedSkillEvalWorkspaces | null = null;
|
|
940
|
+
let subjectVerification: SubjectVerification = "not-checked";
|
|
941
|
+
let observedTreeSha256: string | null = null;
|
|
450
942
|
try {
|
|
943
|
+
// A scenario that ended before the comparison could run records why. When
|
|
944
|
+
// attribution is off the operator's own choice is the operative reason,
|
|
945
|
+
// because the baseline judge would have been skipped either way.
|
|
946
|
+
const skipAttribution = (reason: string): ScenarioAttribution =>
|
|
947
|
+
attributionEnabled ? unmeasurableAttribution(reason) : notAttemptedAttribution(ATTRIBUTION_DISABLED_REASON);
|
|
948
|
+
const outcomeBase = () => ({
|
|
949
|
+
workspace,
|
|
950
|
+
subjectVerification,
|
|
951
|
+
observedTreeSha256,
|
|
952
|
+
wallTimeMs: Math.round(performance.now() - scenarioStart),
|
|
953
|
+
});
|
|
954
|
+
|
|
451
955
|
if (workspaceOverride !== null) await copyWorkspace(workspaceOverride, workspace);
|
|
452
956
|
const fixtureError = await runFixtureCommands(scenario, workspace, timeoutMs, trustFixtures);
|
|
453
957
|
if (fixtureError !== null) {
|
|
@@ -457,24 +961,43 @@ async function runScenario(
|
|
|
457
961
|
baseline: null,
|
|
458
962
|
treatment: null,
|
|
459
963
|
judge: null,
|
|
460
|
-
|
|
461
|
-
|
|
964
|
+
baselineJudge: null,
|
|
965
|
+
attribution: skipAttribution(fixtureError),
|
|
966
|
+
...outcomeBase(),
|
|
462
967
|
infraError: fixtureError,
|
|
463
968
|
});
|
|
464
969
|
}
|
|
465
|
-
runWorkspaces = await materializeSkillEvalWorkspaces(workspace);
|
|
466
|
-
const baseline = await
|
|
970
|
+
runWorkspaces = await materializeSkillEvalWorkspaces(workspace, pinnable ? skillBaseDir : null);
|
|
971
|
+
const baseline = await runArm(
|
|
467
972
|
armRunArgs("baseline", scenario.setup, { target }),
|
|
468
973
|
runWorkspaces.baseline,
|
|
469
974
|
timeoutMs,
|
|
470
975
|
childEnv,
|
|
471
976
|
);
|
|
472
|
-
|
|
473
|
-
|
|
977
|
+
// The arm runs against a private copy, so an edit to the source tree
|
|
978
|
+
// cannot reach the child mid-run. The copy is still digested either side
|
|
979
|
+
// of the arm, so the claim rests on a measurement rather than on the
|
|
980
|
+
// assumption that nothing touched it.
|
|
981
|
+
const pinned = pinnable ? runWorkspaces.pin : null;
|
|
982
|
+
const armSkillDir = pinned ?? skillBaseDir;
|
|
983
|
+
const before = await skillTreeDigest(armSkillDir);
|
|
984
|
+
const treatment = await runArm(
|
|
985
|
+
armRunArgs("treatment", `/skill ${skillName} ${scenario.setup}`, { target, skillBaseDir: armSkillDir }),
|
|
474
986
|
runWorkspaces.treatment,
|
|
475
987
|
timeoutMs,
|
|
476
988
|
childEnv,
|
|
477
989
|
);
|
|
990
|
+
const after = await skillTreeDigest(armSkillDir);
|
|
991
|
+
observedTreeSha256 = after;
|
|
992
|
+
// Three conditions, in the order they can fail. The artifact must be
|
|
993
|
+
// pinnable, the pin must have held, and the child must have actually
|
|
994
|
+
// activated it. Disk equality alone is satisfied by a run that activated
|
|
995
|
+
// nothing, so it is necessary and not sufficient.
|
|
996
|
+
subjectVerification = !pinnable
|
|
997
|
+
? "not-pinned"
|
|
998
|
+
: verifySubjectTree(subjectTreeSha256, before, after) === "verified"
|
|
999
|
+
? verifyObservedActivation(skillName, subjectSha256, join(armSkillDir, "SKILL.md"), treatment.activations)
|
|
1000
|
+
: verifySubjectTree(subjectTreeSha256, before, after);
|
|
478
1001
|
const infra = runInfraError("baseline", baseline) ?? runInfraError("treatment", treatment);
|
|
479
1002
|
if (infra !== null) {
|
|
480
1003
|
return await completeScenarioOutcome({
|
|
@@ -483,8 +1006,9 @@ async function runScenario(
|
|
|
483
1006
|
baseline,
|
|
484
1007
|
treatment,
|
|
485
1008
|
judge: null,
|
|
486
|
-
|
|
487
|
-
|
|
1009
|
+
baselineJudge: null,
|
|
1010
|
+
attribution: skipAttribution(infra),
|
|
1011
|
+
...outcomeBase(),
|
|
488
1012
|
infraError: infra,
|
|
489
1013
|
});
|
|
490
1014
|
}
|
|
@@ -499,8 +1023,9 @@ async function runScenario(
|
|
|
499
1023
|
baseline,
|
|
500
1024
|
treatment,
|
|
501
1025
|
judge: null,
|
|
502
|
-
|
|
503
|
-
|
|
1026
|
+
baselineJudge: null,
|
|
1027
|
+
attribution: skipAttribution(wall),
|
|
1028
|
+
...outcomeBase(),
|
|
504
1029
|
infraError: wall,
|
|
505
1030
|
});
|
|
506
1031
|
}
|
|
@@ -508,7 +1033,7 @@ async function runScenario(
|
|
|
508
1033
|
// terminating tool (artifact plan/review/report) prints only that tool's
|
|
509
1034
|
// result line in text mode, while the event stream carries the artifact
|
|
510
1035
|
// content the verdict may live in.
|
|
511
|
-
const judge = await
|
|
1036
|
+
const judge = await runArm(
|
|
512
1037
|
armRunArgs("judge", judgePrompt(scenario, baseline.transcript, treatment.transcript), { target }),
|
|
513
1038
|
runWorkspaces.judge,
|
|
514
1039
|
timeoutMs,
|
|
@@ -522,20 +1047,38 @@ async function runScenario(
|
|
|
522
1047
|
baseline,
|
|
523
1048
|
treatment,
|
|
524
1049
|
judge,
|
|
525
|
-
|
|
526
|
-
|
|
1050
|
+
baselineJudge: null,
|
|
1051
|
+
attribution: skipAttribution(judgeInfra),
|
|
1052
|
+
...outcomeBase(),
|
|
527
1053
|
infraError: judgeInfra,
|
|
528
1054
|
});
|
|
529
1055
|
}
|
|
530
1056
|
const bullets = parseJudgeVerdicts(scenario, judge);
|
|
1057
|
+
// The rubric result above stands whatever the comparison decides. The
|
|
1058
|
+
// comparison, unlike the rubric, is refused when the artifact measured is
|
|
1059
|
+
// not demonstrably the artifact the subject names.
|
|
1060
|
+
const attributed = await attributeScenario({
|
|
1061
|
+
scenario,
|
|
1062
|
+
bullets,
|
|
1063
|
+
baseline,
|
|
1064
|
+
judge,
|
|
1065
|
+
target,
|
|
1066
|
+
timeoutMs,
|
|
1067
|
+
childEnv,
|
|
1068
|
+
attributionEnabled,
|
|
1069
|
+
subjectVerification,
|
|
1070
|
+
workspace: runWorkspaces.baselineJudge,
|
|
1071
|
+
runner: runArm,
|
|
1072
|
+
});
|
|
531
1073
|
return await completeScenarioOutcome({
|
|
532
1074
|
scenario,
|
|
533
1075
|
bullets,
|
|
534
1076
|
baseline,
|
|
535
1077
|
treatment,
|
|
536
1078
|
judge,
|
|
537
|
-
|
|
538
|
-
|
|
1079
|
+
baselineJudge: attributed.baselineJudge,
|
|
1080
|
+
attribution: attributed.attribution,
|
|
1081
|
+
...outcomeBase(),
|
|
539
1082
|
infraError: null,
|
|
540
1083
|
});
|
|
541
1084
|
} finally {
|
|
@@ -547,10 +1090,189 @@ async function runScenario(
|
|
|
547
1090
|
}
|
|
548
1091
|
}
|
|
549
1092
|
|
|
1093
|
+
/** The exact reason recorded when the operator turned the comparison off. */
|
|
1094
|
+
const ATTRIBUTION_DISABLED_REASON = "disabled by --no-attribution";
|
|
1095
|
+
|
|
1096
|
+
/**
|
|
1097
|
+
* Compare the tree digests taken either side of the treatment arm against the
|
|
1098
|
+
* one the subject was resolved from.
|
|
1099
|
+
*
|
|
1100
|
+
* All three must agree. `verified` therefore means "the artifact on disk was
|
|
1101
|
+
* the subject's artifact before the arm and still was after it", which is a
|
|
1102
|
+
* narrower claim than "the child activated the subject", and the sidecar states
|
|
1103
|
+
* that distinction rather than letting the word imply more.
|
|
1104
|
+
*
|
|
1105
|
+
* @internal Exported for contract tests.
|
|
1106
|
+
*/
|
|
1107
|
+
export function verifySubjectTree(
|
|
1108
|
+
subject: string | null,
|
|
1109
|
+
before: string | null,
|
|
1110
|
+
after: string | null,
|
|
1111
|
+
): SubjectVerification {
|
|
1112
|
+
if (subject === null || before === null || after === null) return "unreadable";
|
|
1113
|
+
return subject === before && before === after ? "verified" : "mismatch";
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
/**
|
|
1117
|
+
* Did the treatment child actually activate the artifact the subject names?
|
|
1118
|
+
*
|
|
1119
|
+
* Disk equality answers a different question. It says the directory looked the
|
|
1120
|
+
* same either side of the arm, which is true of a run that activated nothing at
|
|
1121
|
+
* all, and true of a run that loaded a revision written and reverted inside the
|
|
1122
|
+
* arm. Neither is a measurement of the subject, and reporting `helped` about
|
|
1123
|
+
* either would attribute a difference to an artifact that was never read.
|
|
1124
|
+
*
|
|
1125
|
+
* So the positive evidence is the activation receipt: the child reports the
|
|
1126
|
+
* sha256 of the bytes it read, and that has to equal the pinned SKILL.md's.
|
|
1127
|
+
* Several activations of the same name are ambiguous and are refused rather
|
|
1128
|
+
* than resolved by picking one.
|
|
1129
|
+
*
|
|
1130
|
+
* @internal Exported for contract tests.
|
|
1131
|
+
*/
|
|
1132
|
+
export function verifyObservedActivation(
|
|
1133
|
+
skillName: string,
|
|
1134
|
+
expectedSha256: string,
|
|
1135
|
+
expectedPath: string,
|
|
1136
|
+
activations: ReadonlyArray<ObservedActivation>,
|
|
1137
|
+
): SubjectVerification {
|
|
1138
|
+
const matching = activations.filter((entry) => entry.name === skillName);
|
|
1139
|
+
if (matching.length === 0) return "not-activated";
|
|
1140
|
+
if (matching.length > 1 && new Set(matching.map((entry) => `${entry.hash}|${entry.path}`)).size > 1) {
|
|
1141
|
+
return "activation-mismatch";
|
|
1142
|
+
}
|
|
1143
|
+
const observed = matching[0];
|
|
1144
|
+
if (observed === undefined || observed.hash !== expectedSha256) return "activation-mismatch";
|
|
1145
|
+
// The hash says which bytes; the path says which copy. Two directories can
|
|
1146
|
+
// hold the same SKILL.md and different reference files beside it, and the
|
|
1147
|
+
// body's own pointers resolve against the directory it was loaded from, so a
|
|
1148
|
+
// matching hash from somewhere else is not the artifact under test.
|
|
1149
|
+
return samePath(observed.path, expectedPath) ? "verified" : "activation-mismatch";
|
|
1150
|
+
}
|
|
1151
|
+
|
|
1152
|
+
/** Compare two paths as the filesystem resolves them, falling back to lexical. */
|
|
1153
|
+
function samePath(left: string, right: string): boolean {
|
|
1154
|
+
if (left.length === 0 || right.length === 0) return false;
|
|
1155
|
+
try {
|
|
1156
|
+
return realpathSync.native(left) === realpathSync.native(right);
|
|
1157
|
+
} catch {
|
|
1158
|
+
return resolve(left) === resolve(right);
|
|
1159
|
+
}
|
|
1160
|
+
}
|
|
1161
|
+
|
|
1162
|
+
/** Why the comparison was refused, named after what was not established. */
|
|
1163
|
+
const SUBJECT_REFUSAL: Record<Exclude<SubjectVerification, "verified">, string> = {
|
|
1164
|
+
mismatch:
|
|
1165
|
+
"the measured skill tree changed on disk during this scenario, so its two arms did not run against one artifact",
|
|
1166
|
+
unreadable: "the measured skill tree could not be re-read, so the artifact under test is unverified",
|
|
1167
|
+
"not-pinned":
|
|
1168
|
+
"the skill body carries package references, which resolve against an owning package root outside the skill directory, so the artifact could not be copied faithfully for this run",
|
|
1169
|
+
"not-activated":
|
|
1170
|
+
"the treatment arm recorded no successful activation of this skill, so nothing establishes that the measured artifact ran",
|
|
1171
|
+
"activation-mismatch":
|
|
1172
|
+
"the treatment arm activated content whose hash is not the pinned artifact's, so the comparison would name the wrong revision",
|
|
1173
|
+
"not-checked": "the artifact identity was never checked for this scenario",
|
|
1174
|
+
};
|
|
1175
|
+
|
|
1176
|
+
interface AttributeScenarioInput {
|
|
1177
|
+
scenario: SkillEvalScenario;
|
|
1178
|
+
/** Treatment verdicts from the legacy rubric parser, used only as a skip gate. */
|
|
1179
|
+
bullets: ReadonlyArray<ScoredBullet>;
|
|
1180
|
+
baseline: CapturedRun;
|
|
1181
|
+
/** The treatment judge run, re-read under strict rules for the comparison. */
|
|
1182
|
+
judge: CapturedRun;
|
|
1183
|
+
target: string | undefined;
|
|
1184
|
+
timeoutMs: number;
|
|
1185
|
+
childEnv: NodeJS.ProcessEnv;
|
|
1186
|
+
attributionEnabled: boolean;
|
|
1187
|
+
subjectVerification: SubjectVerification;
|
|
1188
|
+
workspace: string;
|
|
1189
|
+
runner: SkillEvalArmRunner;
|
|
1190
|
+
}
|
|
1191
|
+
|
|
1192
|
+
/**
|
|
1193
|
+
* Score the baseline arm and pair it with the treatment, or say why not.
|
|
1194
|
+
*
|
|
1195
|
+
* The baseline run is already paid for by the time this is reached; only the
|
|
1196
|
+
* judge that reads it is new. Five states end the comparison before it produces
|
|
1197
|
+
* a verdict, and each is recorded with its own reason rather than collapsed
|
|
1198
|
+
* into one: the operator disabled it, the artifact measured was not
|
|
1199
|
+
* demonstrably the artifact the subject names, the treatment produced nothing
|
|
1200
|
+
* to compare against, the baseline judge failed to run, or the baseline judge
|
|
1201
|
+
* hit the harness's own permission wall.
|
|
1202
|
+
*
|
|
1203
|
+
* The verdicts it pairs come from {@link strictJudgeVerdicts}, not from the
|
|
1204
|
+
* rubric parser, so a judge that answered with a missing, null or string `pass`
|
|
1205
|
+
* contributes nothing instead of contributing an invented `fail`.
|
|
1206
|
+
*/
|
|
1207
|
+
async function attributeScenario(
|
|
1208
|
+
input: AttributeScenarioInput,
|
|
1209
|
+
): Promise<{ attribution: ScenarioAttribution; baselineJudge: CapturedRun | null }> {
|
|
1210
|
+
if (!input.attributionEnabled) {
|
|
1211
|
+
return { attribution: notAttemptedAttribution(ATTRIBUTION_DISABLED_REASON), baselineJudge: null };
|
|
1212
|
+
}
|
|
1213
|
+
// Nothing establishes which artifact ran, so nothing can be attributed to
|
|
1214
|
+
// one. Each refusal names what was not established rather than collapsing
|
|
1215
|
+
// into one verdict.
|
|
1216
|
+
if (input.subjectVerification !== "verified") {
|
|
1217
|
+
return { attribution: unmeasurableAttribution(SUBJECT_REFUSAL[input.subjectVerification]), baselineJudge: null };
|
|
1218
|
+
}
|
|
1219
|
+
// The same comparative eligibility the baseline judge gets. A treatment judge
|
|
1220
|
+
// that collected the harness's own denial and then answered was scoring under
|
|
1221
|
+
// a constraint the other arm did not have. The rubric result it produced
|
|
1222
|
+
// stands; only the comparison is refused.
|
|
1223
|
+
const treatmentWall = permissionWallReason("judge", input.judge);
|
|
1224
|
+
if (treatmentWall !== null) return { attribution: unmeasurableAttribution(treatmentWall), baselineJudge: null };
|
|
1225
|
+
// Nothing on the treatment side was scored, so no pair can be formed. Running
|
|
1226
|
+
// the baseline judge anyway would spend an inference to learn nothing.
|
|
1227
|
+
if (!input.bullets.some((bullet) => bullet.verdict === "pass" || bullet.verdict === "fail")) {
|
|
1228
|
+
return {
|
|
1229
|
+
attribution: unmeasurableAttribution(
|
|
1230
|
+
"the treatment arm produced no scored bullet, so there is nothing to compare a baseline against",
|
|
1231
|
+
),
|
|
1232
|
+
baselineJudge: null,
|
|
1233
|
+
};
|
|
1234
|
+
}
|
|
1235
|
+
const baselineJudge = await input.runner(
|
|
1236
|
+
armRunArgs("baseline-judge", baselineJudgePrompt(input.scenario, input.baseline.transcript), {
|
|
1237
|
+
target: input.target,
|
|
1238
|
+
}),
|
|
1239
|
+
input.workspace,
|
|
1240
|
+
input.timeoutMs,
|
|
1241
|
+
input.childEnv,
|
|
1242
|
+
);
|
|
1243
|
+
const infra = runInfraError("baseline-judge", baselineJudge);
|
|
1244
|
+
if (infra !== null) {
|
|
1245
|
+
// The treatment verdicts stand: this failure is about the comparison, not
|
|
1246
|
+
// about the skill, so it never touches `bullets` or the exit code.
|
|
1247
|
+
return { attribution: unmeasurableAttribution(infra), baselineJudge };
|
|
1248
|
+
}
|
|
1249
|
+
// The prompt tells the judge to use no tools, which is an instruction and not
|
|
1250
|
+
// enforcement. A judge that tried anyway and collected the harness's denial
|
|
1251
|
+
// was scoring under a constraint the treatment judge did not have.
|
|
1252
|
+
const wall = permissionWallReason("baseline-judge", baselineJudge);
|
|
1253
|
+
if (wall !== null) return { attribution: unmeasurableAttribution(wall), baselineJudge };
|
|
1254
|
+
return {
|
|
1255
|
+
attribution: attributionFromStrict(
|
|
1256
|
+
input.scenario,
|
|
1257
|
+
strictJudgeVerdicts(input.scenario, baselineJudge),
|
|
1258
|
+
strictJudgeVerdicts(input.scenario, input.judge),
|
|
1259
|
+
),
|
|
1260
|
+
baselineJudge,
|
|
1261
|
+
};
|
|
1262
|
+
}
|
|
1263
|
+
|
|
550
1264
|
export interface MaterializedSkillEvalWorkspaces {
|
|
551
1265
|
baseline: string;
|
|
552
1266
|
treatment: string;
|
|
553
1267
|
judge: string;
|
|
1268
|
+
/** The baseline judge gets its own root for the same reason the other arms do. */
|
|
1269
|
+
baselineJudge: string;
|
|
1270
|
+
/**
|
|
1271
|
+
* Private per-run copy of the artifact the treatment arm loads; null when the
|
|
1272
|
+
* artifact cannot be copied faithfully. Isolated from the source directory,
|
|
1273
|
+
* not read-only to the arm itself.
|
|
1274
|
+
*/
|
|
1275
|
+
pin: string | null;
|
|
554
1276
|
cleanup(): Promise<void>;
|
|
555
1277
|
}
|
|
556
1278
|
|
|
@@ -581,17 +1303,33 @@ async function armWorkspace(created: string[]): Promise<string> {
|
|
|
581
1303
|
}
|
|
582
1304
|
|
|
583
1305
|
/** @internal Exported for contract tests. */
|
|
584
|
-
async function materializeSkillEvalWorkspaces(
|
|
1306
|
+
async function materializeSkillEvalWorkspaces(
|
|
1307
|
+
seedWorkspace: string,
|
|
1308
|
+
pinSource: string | null,
|
|
1309
|
+
): Promise<MaterializedSkillEvalWorkspaces> {
|
|
585
1310
|
const created: string[] = [];
|
|
586
1311
|
try {
|
|
587
1312
|
const baseline = await armWorkspace(created);
|
|
588
1313
|
const treatment = await armWorkspace(created);
|
|
589
1314
|
const judge = await armWorkspace(created);
|
|
1315
|
+
const baselineJudge = await armWorkspace(created);
|
|
1316
|
+
let pin: string | null = null;
|
|
1317
|
+
if (pinSource !== null) {
|
|
1318
|
+
const root = await mkdtemp(join(tmpdir(), "clio-coder-skill-eval-pin-"));
|
|
1319
|
+
created.push(root);
|
|
1320
|
+
pin = join(root, "skill");
|
|
1321
|
+
await cp(pinSource, pin, { recursive: true, preserveTimestamps: true, verbatimSymlinks: true });
|
|
1322
|
+
}
|
|
1323
|
+
// Only the acting arms get the fixture. A judge scores text and is told to
|
|
1324
|
+
// call no tools, so seeding its workspace would only give it the artifacts
|
|
1325
|
+
// it is supposed to read about.
|
|
590
1326
|
await Promise.all([copyWorkspace(seedWorkspace, baseline), copyWorkspace(seedWorkspace, treatment)]);
|
|
591
1327
|
return {
|
|
592
1328
|
baseline,
|
|
593
1329
|
treatment,
|
|
594
1330
|
judge,
|
|
1331
|
+
baselineJudge,
|
|
1332
|
+
pin,
|
|
595
1333
|
cleanup: async () => {
|
|
596
1334
|
await Promise.all(created.map((path) => rm(path, { recursive: true, force: true })));
|
|
597
1335
|
},
|
|
@@ -613,7 +1351,7 @@ async function copyWorkspace(source: string, destination: string): Promise<void>
|
|
|
613
1351
|
async function completeScenarioOutcome(outcome: Omit<ScenarioOutcome, "usage">): Promise<ScenarioOutcome> {
|
|
614
1352
|
return {
|
|
615
1353
|
...outcome,
|
|
616
|
-
usage: await usageForCapturedRuns([outcome.baseline, outcome.treatment, outcome.judge]),
|
|
1354
|
+
usage: await usageForCapturedRuns([outcome.baseline, outcome.treatment, outcome.judge, outcome.baselineJudge]),
|
|
617
1355
|
};
|
|
618
1356
|
}
|
|
619
1357
|
|
|
@@ -816,12 +1554,36 @@ function loadsSkillBody(tool: string, args: unknown): boolean {
|
|
|
816
1554
|
return args.scope === "skills" && typeof args.name === "string" && args.name.trim().length > 0;
|
|
817
1555
|
}
|
|
818
1556
|
|
|
1557
|
+
/**
|
|
1558
|
+
* The activation contract inside a successful skill load's result details.
|
|
1559
|
+
*
|
|
1560
|
+
* `runSkillsScope` records `name`, `filePath` and `hash` on every activation,
|
|
1561
|
+
* and the agent loop puts the tool result's details on the execution-end event.
|
|
1562
|
+
* Anything missing either field is not an activation receipt and is ignored
|
|
1563
|
+
* rather than half-read.
|
|
1564
|
+
*/
|
|
1565
|
+
function activationFromResult(result: unknown): ObservedActivation | null {
|
|
1566
|
+
if (!isRecord(result)) return null;
|
|
1567
|
+
const details = isRecord(result.details) ? result.details : null;
|
|
1568
|
+
if (details === null) return null;
|
|
1569
|
+
const name = readString(details.name);
|
|
1570
|
+
const hash = readString(details.hash);
|
|
1571
|
+
if (name === null || hash === null) return null;
|
|
1572
|
+
return { name, hash, path: readString(details.filePath) ?? readString(details.path) ?? "" };
|
|
1573
|
+
}
|
|
1574
|
+
|
|
819
1575
|
/** @internal Exported for contract tests. */
|
|
820
|
-
function parseRunStdout(stdout: string): {
|
|
1576
|
+
export function parseRunStdout(stdout: string): {
|
|
1577
|
+
sessionId: string | null;
|
|
1578
|
+
transcript: string;
|
|
1579
|
+
finalText: string;
|
|
1580
|
+
activations: ObservedActivation[];
|
|
1581
|
+
} {
|
|
821
1582
|
let sessionId: string | null = null;
|
|
822
1583
|
const lines: string[] = [];
|
|
823
1584
|
let finalText = "";
|
|
824
1585
|
let sawJson = false;
|
|
1586
|
+
const activations: ObservedActivation[] = [];
|
|
825
1587
|
const streamedText = new Map<number, string>();
|
|
826
1588
|
// Tool calls whose result is the skill's own SKILL.md. Correlated by
|
|
827
1589
|
// toolCallId, which both the start and end events carry.
|
|
@@ -867,6 +1629,15 @@ function parseRunStdout(stdout: string): { sessionId: string | null; transcript:
|
|
|
867
1629
|
const tool = readString(event.toolName) ?? readString(event.tool) ?? "tool";
|
|
868
1630
|
const status = event.isError === true ? "error" : "ok";
|
|
869
1631
|
const callId = readString(event.toolCallId);
|
|
1632
|
+
// A successful skill load carries the activation contract in its result
|
|
1633
|
+
// details: the name, the file and the sha256 of the bytes the child
|
|
1634
|
+
// actually read. That is the only evidence in this stream about which
|
|
1635
|
+
// artifact ran, and it is collected here and kept out of the transcript
|
|
1636
|
+
// so the judge still never sees the body or its identity.
|
|
1637
|
+
if (event.isError !== true && callId !== null && skillBodyCallIds.has(callId)) {
|
|
1638
|
+
const activation = activationFromResult(event.result);
|
|
1639
|
+
if (activation !== null) activations.push(activation);
|
|
1640
|
+
}
|
|
870
1641
|
// The skill body is the instructions, not the behavior. Left in the
|
|
871
1642
|
// transcript it is the easiest thing in the run for a judge to quote,
|
|
872
1643
|
// and a 30B judge scored bullets as passing from SKILL.md prose that
|
|
@@ -903,9 +1674,9 @@ function parseRunStdout(stdout: string): { sessionId: string | null; transcript:
|
|
|
903
1674
|
}
|
|
904
1675
|
if (!sawJson) {
|
|
905
1676
|
const text = stdout.trim();
|
|
906
|
-
return { sessionId: null, transcript: text, finalText: text };
|
|
1677
|
+
return { sessionId: null, transcript: text, finalText: text, activations: [] };
|
|
907
1678
|
}
|
|
908
|
-
return { sessionId, transcript: elide(lines.join("\n")), finalText };
|
|
1679
|
+
return { sessionId, transcript: elide(lines.join("\n")), finalText, activations };
|
|
909
1680
|
}
|
|
910
1681
|
|
|
911
1682
|
function judgePrompt(scenario: SkillEvalScenario, baselineTranscript: string, treatmentTranscript: string): string {
|
|
@@ -932,6 +1703,41 @@ function judgePrompt(scenario: SkillEvalScenario, baselineTranscript: string, tr
|
|
|
932
1703
|
].join("\n");
|
|
933
1704
|
}
|
|
934
1705
|
|
|
1706
|
+
/**
|
|
1707
|
+
* Score the baseline arm alone, against the same bullets.
|
|
1708
|
+
*
|
|
1709
|
+
* Isolated on purpose. The treatment judge above sees both transcripts and is
|
|
1710
|
+
* told to score only the treatment; giving the baseline judge the treatment
|
|
1711
|
+
* transcript as well would let a strong treatment run colour the baseline's
|
|
1712
|
+
* verdicts, and the comparison those verdicts feed would then be measuring the
|
|
1713
|
+
* judge. The bullet contract, the strict-JSON shape and the no-tools rule are
|
|
1714
|
+
* identical, so `parseJudgeVerdicts` reads both without branching.
|
|
1715
|
+
*
|
|
1716
|
+
* The treatment prompt is deliberately left byte-identical to what it was
|
|
1717
|
+
* before attribution existed, so treatment verdicts stay comparable with every
|
|
1718
|
+
* bundle already on disk.
|
|
1719
|
+
*/
|
|
1720
|
+
function baselineJudgePrompt(scenario: SkillEvalScenario, baselineTranscript: string): string {
|
|
1721
|
+
const bullets = scenario.expected.map((text, index) => `${index + 1}. ${text}`).join("\n");
|
|
1722
|
+
return [
|
|
1723
|
+
"You are scoring an agent run. One transcript follows: it ran WITHOUT any skill loaded.",
|
|
1724
|
+
"Score each EXPECTED bullet strictly against the TRANSCRIPT.",
|
|
1725
|
+
"A bullet passes only if the transcript observably satisfies it; anything unverifiable from the transcript fails.",
|
|
1726
|
+
'Reply with STRICT JSON only, no prose and no code fences, exactly: {"bullets":[{"index":1,"pass":true,"reason":"<= 25 words"}]}',
|
|
1727
|
+
`Include one entry per bullet, indexes 1 through ${scenario.expected.length} in order.`,
|
|
1728
|
+
"Do not use any tools. Respond with the JSON verdict directly.",
|
|
1729
|
+
"",
|
|
1730
|
+
`SCENARIO ${scenario.id} - ${scenario.title}`,
|
|
1731
|
+
`SETUP: ${scenario.setup}`,
|
|
1732
|
+
"",
|
|
1733
|
+
"EXPECTED BULLETS:",
|
|
1734
|
+
bullets,
|
|
1735
|
+
"",
|
|
1736
|
+
"TRANSCRIPT:",
|
|
1737
|
+
baselineTranscript.length > 0 ? baselineTranscript : "(empty)",
|
|
1738
|
+
].join("\n");
|
|
1739
|
+
}
|
|
1740
|
+
|
|
935
1741
|
/**
|
|
936
1742
|
* Why a judge response carried no verdict, named after the thing that failed.
|
|
937
1743
|
* A response that opened a bullets object and never closed it is the observed
|
|
@@ -984,6 +1790,85 @@ function parseJudgeVerdicts(scenario: SkillEvalScenario, judge: CapturedRun): Sc
|
|
|
984
1790
|
});
|
|
985
1791
|
}
|
|
986
1792
|
|
|
1793
|
+
/**
|
|
1794
|
+
* Verdicts strict enough to compare two arms against each other.
|
|
1795
|
+
*
|
|
1796
|
+
* {@link parseJudgeVerdicts} above is the legacy rubric gate and its coercions
|
|
1797
|
+
* are load-bearing for `pass` / `exitCode` / `failureClass`, so it is left
|
|
1798
|
+
* exactly as it is. It is, however, forgiving in ways that are fine for a
|
|
1799
|
+
* one-armed verdict and wrong for a comparison: `pass: item.pass === true`
|
|
1800
|
+
* turns a missing, null or string field into a genuine `fail`, and
|
|
1801
|
+
* `Number.parseInt(String(item.index))` reads `"1oops"` as bullet 1. Pairing a
|
|
1802
|
+
* real treatment `pass` against an invented baseline `fail` reports `helped`
|
|
1803
|
+
* about a measurement the judge never made.
|
|
1804
|
+
*
|
|
1805
|
+
* So comparison reads the raw judge output again under stricter rules, and
|
|
1806
|
+
* anything that does not clear them stays out of the comparison rather than
|
|
1807
|
+
* entering it as a verdict. Rejecting evidence here cannot change the rubric
|
|
1808
|
+
* result: the two parsers have separate callers on purpose.
|
|
1809
|
+
*
|
|
1810
|
+
* @internal Exported for contract tests.
|
|
1811
|
+
*/
|
|
1812
|
+
export interface StrictJudgeVerdicts {
|
|
1813
|
+
/** Bullet index to its unambiguous boolean verdict. */
|
|
1814
|
+
verdicts: Map<number, boolean>;
|
|
1815
|
+
/** Bullet index to why its evidence was refused. */
|
|
1816
|
+
rejected: Map<number, string>;
|
|
1817
|
+
/** Set when the run produced no parseable verdict object at all. */
|
|
1818
|
+
absent: string | null;
|
|
1819
|
+
}
|
|
1820
|
+
|
|
1821
|
+
/** @internal Exported for contract tests. */
|
|
1822
|
+
export function strictJudgeVerdicts(scenario: SkillEvalScenario, judge: CapturedRun): StrictJudgeVerdicts {
|
|
1823
|
+
const verdicts = new Map<number, boolean>();
|
|
1824
|
+
const rejected = new Map<number, string>();
|
|
1825
|
+
const parsed = extractBulletsObject(judge.finalText) ?? extractBulletsObject(judge.transcript);
|
|
1826
|
+
if (parsed === null) return { verdicts, rejected, absent: judgeVerdictAbsenceReason(judge) };
|
|
1827
|
+
if (!Array.isArray(parsed.bullets)) {
|
|
1828
|
+
return { verdicts, rejected, absent: "judge verdict object carried no bullets array" };
|
|
1829
|
+
}
|
|
1830
|
+
const count = scenario.expected.length;
|
|
1831
|
+
const seen = new Map<number, boolean>();
|
|
1832
|
+
for (const item of parsed.bullets) {
|
|
1833
|
+
if (!isRecord(item)) continue;
|
|
1834
|
+
// A numeric index and nothing else. A string that happens to start with
|
|
1835
|
+
// digits is not an index the judge chose; it is a parse accident.
|
|
1836
|
+
const index = item.index;
|
|
1837
|
+
if (typeof index !== "number" || !Number.isInteger(index) || index < 1 || index > count) {
|
|
1838
|
+
continue;
|
|
1839
|
+
}
|
|
1840
|
+
// A rejection for an index is final and order-independent. An entry that
|
|
1841
|
+
// arrives after a valid one still poisons it: the judge emitted two
|
|
1842
|
+
// answers for one bullet and only one of them is usable, so which one it
|
|
1843
|
+
// "meant" is a guess, and a guess is not comparison evidence.
|
|
1844
|
+
if (typeof item.pass !== "boolean") {
|
|
1845
|
+
rejected.set(index, `judge gave no boolean pass for bullet ${index}: comparison evidence refused`);
|
|
1846
|
+
verdicts.delete(index);
|
|
1847
|
+
continue;
|
|
1848
|
+
}
|
|
1849
|
+
if (rejected.has(index)) {
|
|
1850
|
+
verdicts.delete(index);
|
|
1851
|
+
continue;
|
|
1852
|
+
}
|
|
1853
|
+
const previous = seen.get(index);
|
|
1854
|
+
if (previous !== undefined && previous !== item.pass) {
|
|
1855
|
+
// Two contradictory verdicts for one bullet. Taking either would pick a
|
|
1856
|
+
// winner the judge never picked.
|
|
1857
|
+
rejected.set(index, `judge gave contradictory verdicts for bullet ${index}: comparison evidence refused`);
|
|
1858
|
+
verdicts.delete(index);
|
|
1859
|
+
continue;
|
|
1860
|
+
}
|
|
1861
|
+
seen.set(index, item.pass);
|
|
1862
|
+
verdicts.set(index, item.pass);
|
|
1863
|
+
}
|
|
1864
|
+
for (let index = 1; index <= count; index += 1) {
|
|
1865
|
+
if (!verdicts.has(index) && !rejected.has(index)) {
|
|
1866
|
+
rejected.set(index, `judge omitted bullet ${index}: comparison evidence absent`);
|
|
1867
|
+
}
|
|
1868
|
+
}
|
|
1869
|
+
return { verdicts, rejected, absent: null };
|
|
1870
|
+
}
|
|
1871
|
+
|
|
987
1872
|
/**
|
|
988
1873
|
* Find the JSON object carrying the judge's `bullets` array anywhere in the
|
|
989
1874
|
* text: models wrap verdicts in prose, code fences, or terminating tool
|
|
@@ -1043,7 +1928,7 @@ function balancedJsonSlice(text: string, start: number): string | null {
|
|
|
1043
1928
|
}
|
|
1044
1929
|
|
|
1045
1930
|
function synthesizeArtifact(
|
|
1046
|
-
|
|
1931
|
+
subject: SkillEvalSubject,
|
|
1047
1932
|
evalsPath: string,
|
|
1048
1933
|
evalsRaw: string,
|
|
1049
1934
|
startedAt: string,
|
|
@@ -1051,6 +1936,7 @@ function synthesizeArtifact(
|
|
|
1051
1936
|
outcomes: ReadonlyArray<ScenarioOutcome>,
|
|
1052
1937
|
target: string | undefined,
|
|
1053
1938
|
): EvalRunArtifact {
|
|
1939
|
+
const skillName = subject.name;
|
|
1054
1940
|
const contentHash = createHash("sha256").update(evalsRaw, "utf8").digest("hex");
|
|
1055
1941
|
const stamp = startedAt.replace(/[-:.]/g, "");
|
|
1056
1942
|
// Random suffix for the same reason createEvalId carries one: the stamp plus
|
|
@@ -1070,6 +1956,14 @@ function synthesizeArtifact(
|
|
|
1070
1956
|
tags: [
|
|
1071
1957
|
"skill-eval",
|
|
1072
1958
|
`skill:${skillName}`,
|
|
1959
|
+
// The artifact identity, on every record. Two bundles for one skill
|
|
1960
|
+
// name are only comparable when both say which content they ran.
|
|
1961
|
+
`skill-sha:${subject.normalizedHash.slice(0, 12)}`,
|
|
1962
|
+
`skill-origin:${subject.origin}`,
|
|
1963
|
+
...(subject.drift !== null ? [`skill-drift:${subject.drift.verdict}`] : []),
|
|
1964
|
+
// Advisory. `pass` and `exitCode` below keep their treatment-only
|
|
1965
|
+
// meaning; this tag is read by nothing that gates.
|
|
1966
|
+
`attribution:${outcome.attribution.verdict}`,
|
|
1073
1967
|
...(unmeasured ? ["scenario:unmeasured"] : []),
|
|
1074
1968
|
...outcome.bullets.map((bullet) => `bullet-${bullet.index}:${bullet.verdict}`),
|
|
1075
1969
|
],
|
|
@@ -1107,25 +2001,46 @@ function synthesizeArtifact(
|
|
|
1107
2001
|
};
|
|
1108
2002
|
}
|
|
1109
2003
|
|
|
1110
|
-
|
|
1111
|
-
|
|
2004
|
+
/** @internal Exported for contract tests. */
|
|
2005
|
+
export function sidecar(
|
|
2006
|
+
subject: SkillEvalSubject,
|
|
1112
2007
|
evalId: string,
|
|
1113
2008
|
outcomes: ReadonlyArray<ScenarioOutcome>,
|
|
1114
2009
|
allowNetwork: boolean,
|
|
2010
|
+
attributionEnabled: boolean,
|
|
1115
2011
|
): unknown {
|
|
1116
2012
|
return {
|
|
1117
|
-
version:
|
|
2013
|
+
version: 2,
|
|
1118
2014
|
schema: "experimental",
|
|
1119
2015
|
kind: "skill-eval",
|
|
1120
|
-
skill:
|
|
2016
|
+
skill: subject.name,
|
|
2017
|
+
// What this run measured, by identity rather than by name. `version` moved
|
|
2018
|
+
// to 2 for this block; readers of version 1 keep working, they just never
|
|
2019
|
+
// learn which copy produced their numbers.
|
|
2020
|
+
subject,
|
|
1121
2021
|
evalId,
|
|
1122
2022
|
network: networkPolicyLabel(allowNetwork),
|
|
1123
2023
|
autonomy: ARM_AUTONOMY,
|
|
2024
|
+
attributionEnabled,
|
|
2025
|
+
attributionSummary: attributionSummary(outcomes),
|
|
1124
2026
|
deltas: [
|
|
1125
2027
|
"bullet verdicts are judge-scored from run transcripts, not command exit codes; the evals-domain artifact carries scenario-level records with empty command lists",
|
|
1126
2028
|
"a bullet the judge never scored is recorded unmeasured, not failed: its scenario record is pass:false with exitCode 3 and no failureClass",
|
|
1127
2029
|
"an arm whose transcript carries the headless permission wall is recorded unmeasured with an infraError: the harness's own gate is not a verdict about the skill",
|
|
1128
2030
|
"tokens and cost are rolled up from headless main-agent receipts when those receipts are present",
|
|
2031
|
+
"attribution is advisory and gates nothing: pass, exitCode and failureClass keep their treatment-only meaning",
|
|
2032
|
+
"attribution unmeasured means the comparison was impossible; not-attempted means it was never run. Neither is no-change",
|
|
2033
|
+
"the baseline judge scores the baseline transcript alone; the treatment judge prompt is unchanged from version 1 bundles",
|
|
2034
|
+
"comparison reads both judges' raw output under strict rules (boolean pass, integer in-range index, no contradictory duplicate); refused evidence stays unmeasured and never becomes a verdict",
|
|
2035
|
+
"the rubric bullets above keep the legacy parser's forgiving coercions: tightening comparison eligibility does not move the pass/exitCode gate",
|
|
2036
|
+
"the treatment arm runs against a private per-run copy of the skill directory, so an edit to the source during the arm cannot reach the child; subject.pinnable is false when the body carries package references, which resolve above the skill directory and cannot be snapshotted faithfully",
|
|
2037
|
+
"that copy is an isolated snapshot, not an immutable pin: it lives in a writable temp directory and the arm runs at full-auto, so a model that edited the copy, read it and restored it would not be detected. These are results about a cooperative model, the same caveat the arm workspaces already carry",
|
|
2038
|
+
"subjectVerification is verified only when the snapshot held (tree digest equal before and after the arm) AND the treatment arm reported a successful activation whose sha256 equals subject.sha256 and whose file is the snapshot's own SKILL.md by canonical path. not-activated, activation-mismatch, not-pinned, mismatch and unreadable each name what was not established",
|
|
2039
|
+
"a scenario whose subjectVerification is not verified keeps its rubric result and records its comparison unmeasured",
|
|
2040
|
+
"treeSha256 covers regular files and in-tree symlink targets; a symlink leaving the directory makes the tree unverifiable rather than partially hashed",
|
|
2041
|
+
"both judges are checked for the headless permission wall before their verdicts are eligible for comparison; the legacy rubric result is unaffected either way",
|
|
2042
|
+
"subject.drift records whether the measured content still matches its recorded hash; a mismatch never blocks the run",
|
|
2043
|
+
"subject hashes cover SKILL.md (sha256, normalizedHash) and the whole skill directory (treeSha256); neither covers resources the skill reads from outside its own directory",
|
|
1129
2044
|
"this sidecar is additive and is registered in overview.json files[]",
|
|
1130
2045
|
],
|
|
1131
2046
|
scenarios: outcomes.map((outcome) => ({
|
|
@@ -1136,13 +2051,31 @@ function sidecar(
|
|
|
1136
2051
|
infraError: outcome.infraError,
|
|
1137
2052
|
usage: outcome.usage,
|
|
1138
2053
|
bullets: outcome.bullets,
|
|
2054
|
+
subjectVerification: outcome.subjectVerification,
|
|
2055
|
+
observedTreeSha256: outcome.observedTreeSha256,
|
|
2056
|
+
attribution: outcome.attribution,
|
|
1139
2057
|
baseline: sidecarRun(outcome.baseline),
|
|
1140
2058
|
treatment: sidecarRun(outcome.treatment),
|
|
1141
2059
|
judge: sidecarRun(outcome.judge),
|
|
2060
|
+
baselineJudge: sidecarRun(outcome.baselineJudge),
|
|
1142
2061
|
})),
|
|
1143
2062
|
};
|
|
1144
2063
|
}
|
|
1145
2064
|
|
|
2065
|
+
/** Scenario counts per attribution verdict. A rollup, never collapsed to one score. */
|
|
2066
|
+
function attributionSummary(outcomes: ReadonlyArray<ScenarioOutcome>): Record<ScenarioAttributionVerdict, number> {
|
|
2067
|
+
const summary: Record<ScenarioAttributionVerdict, number> = {
|
|
2068
|
+
helped: 0,
|
|
2069
|
+
regressed: 0,
|
|
2070
|
+
mixed: 0,
|
|
2071
|
+
"no-change": 0,
|
|
2072
|
+
unmeasured: 0,
|
|
2073
|
+
"not-attempted": 0,
|
|
2074
|
+
};
|
|
2075
|
+
for (const outcome of outcomes) summary[outcome.attribution.verdict] += 1;
|
|
2076
|
+
return summary;
|
|
2077
|
+
}
|
|
2078
|
+
|
|
1146
2079
|
function sidecarRun(run: CapturedRun | null): unknown {
|
|
1147
2080
|
if (run === null) return null;
|
|
1148
2081
|
return {
|
|
@@ -1156,17 +2089,27 @@ function sidecarRun(run: CapturedRun | null): unknown {
|
|
|
1156
2089
|
}
|
|
1157
2090
|
|
|
1158
2091
|
function printHumanReport(
|
|
1159
|
-
|
|
2092
|
+
subject: SkillEvalSubject,
|
|
1160
2093
|
outcomes: ReadonlyArray<ScenarioOutcome>,
|
|
1161
2094
|
evalId: string,
|
|
1162
2095
|
evidenceId: string | null,
|
|
1163
2096
|
evidenceDirectory: string | null,
|
|
1164
2097
|
allowNetwork: boolean,
|
|
1165
2098
|
): void {
|
|
1166
|
-
const
|
|
2099
|
+
const skillName = subject.name;
|
|
2100
|
+
// `vs base` is the same bullet in the arm that ran without the skill. Kept in
|
|
2101
|
+
// its own column so a reader never mistakes it for part of the rubric verdict.
|
|
2102
|
+
const rows: string[][] = [["scenario", "bullet", "verdict", "vs base", "expected"]];
|
|
1167
2103
|
for (const outcome of outcomes) {
|
|
2104
|
+
const attributed = new Map(outcome.attribution.bullets.map((item) => [item.index, item]));
|
|
1168
2105
|
for (const bullet of outcome.bullets) {
|
|
1169
|
-
rows.push([
|
|
2106
|
+
rows.push([
|
|
2107
|
+
outcome.scenario.id,
|
|
2108
|
+
String(bullet.index),
|
|
2109
|
+
bullet.verdict,
|
|
2110
|
+
attributed.get(bullet.index)?.attribution ?? "-",
|
|
2111
|
+
truncate(bullet.text, 64),
|
|
2112
|
+
]);
|
|
1170
2113
|
}
|
|
1171
2114
|
}
|
|
1172
2115
|
process.stdout.write(formatColumns(rows));
|
|
@@ -1204,13 +2147,78 @@ function printHumanReport(
|
|
|
1204
2147
|
);
|
|
1205
2148
|
}
|
|
1206
2149
|
}
|
|
2150
|
+
printAttributionReport(outcomes);
|
|
1207
2151
|
process.stdout.write(`${describeArmPolicyOutcome(allowNetwork)}\n`);
|
|
2152
|
+
process.stdout.write(
|
|
2153
|
+
`subject: ${skillName} from ${subject.origin} at ${subject.baseDir} (sha256 ${subject.normalizedHash.slice(0, 12)}…)\n`,
|
|
2154
|
+
);
|
|
2155
|
+
if (subject.drift !== null) {
|
|
2156
|
+
process.stdout.write(
|
|
2157
|
+
subject.drift.verdict === "mismatch"
|
|
2158
|
+
? `subject drift: MISMATCH against the ${subject.drift.authority} (expected ${subject.drift.expected.slice(0, 12)}…); this run measured the copy on disk\n`
|
|
2159
|
+
: `subject drift: matches the ${subject.drift.authority}\n`,
|
|
2160
|
+
);
|
|
2161
|
+
}
|
|
2162
|
+
const unverified = outcomes.filter((outcome) => outcome.subjectVerification !== "verified");
|
|
2163
|
+
if (unverified.length > 0) {
|
|
2164
|
+
process.stdout.write(
|
|
2165
|
+
`subject verification: ${unverified.map((o) => `${o.scenario.id}=${o.subjectVerification}`).join(", ")}; ` +
|
|
2166
|
+
"those scenarios kept their rubric result and recorded no baseline comparison\n",
|
|
2167
|
+
);
|
|
2168
|
+
}
|
|
1208
2169
|
process.stdout.write(`eval artifact: ${evalId}\n`);
|
|
1209
2170
|
if (evidenceId !== null && evidenceDirectory !== null) {
|
|
1210
2171
|
process.stdout.write(`evidence: ${evidenceId} at ${evidenceDirectory} (per-bullet detail in skill-eval.json)\n`);
|
|
1211
2172
|
}
|
|
1212
2173
|
}
|
|
1213
2174
|
|
|
2175
|
+
/**
|
|
2176
|
+
* The paired comparison, stated as scenario counts and never as one score.
|
|
2177
|
+
*
|
|
2178
|
+
* Deliberately separated from the rubric block above it. A reader who takes
|
|
2179
|
+
* "3/4 bullets passed" and "1 scenario regressed" as the same measurement will
|
|
2180
|
+
* draw the wrong conclusion from both: the first says whether the skill met its
|
|
2181
|
+
* rubric, the second says whether it changed anything relative to no skill at
|
|
2182
|
+
* all. Nothing here influences the exit code.
|
|
2183
|
+
*/
|
|
2184
|
+
function printAttributionReport(outcomes: ReadonlyArray<ScenarioOutcome>): void {
|
|
2185
|
+
if (outcomes.length === 0) return;
|
|
2186
|
+
const summary = attributionSummary(outcomes);
|
|
2187
|
+
const compared = summary.helped + summary.regressed + summary.mixed + summary["no-change"];
|
|
2188
|
+
if (compared === 0) {
|
|
2189
|
+
const reason = outcomes.find((outcome) => outcome.attribution.reason !== null)?.attribution.reason ?? null;
|
|
2190
|
+
process.stdout.write(
|
|
2191
|
+
`attribution: no scenario was compared against its baseline${reason !== null ? ` (${reason})` : ""}\n`,
|
|
2192
|
+
);
|
|
2193
|
+
return;
|
|
2194
|
+
}
|
|
2195
|
+
const parts = [
|
|
2196
|
+
`${summary.helped} helped`,
|
|
2197
|
+
`${summary.regressed} regressed`,
|
|
2198
|
+
`${summary.mixed} mixed`,
|
|
2199
|
+
`${summary["no-change"]} no-change`,
|
|
2200
|
+
];
|
|
2201
|
+
if (summary.unmeasured > 0) parts.push(`${summary.unmeasured} unmeasured`);
|
|
2202
|
+
if (summary["not-attempted"] > 0) parts.push(`${summary["not-attempted"]} not-attempted`);
|
|
2203
|
+
process.stdout.write(
|
|
2204
|
+
`attribution (advisory, vs the no-skill baseline; gates nothing): ${parts.join(", ")} of ${outcomes.length} scenario${outcomes.length === 1 ? "" : "s"}\n`,
|
|
2205
|
+
);
|
|
2206
|
+
for (const outcome of outcomes) {
|
|
2207
|
+
if (outcome.attribution.verdict === "regressed" || outcome.attribution.verdict === "mixed") {
|
|
2208
|
+
const regressed = outcome.attribution.bullets
|
|
2209
|
+
.filter((bullet) => bullet.attribution === "regressed")
|
|
2210
|
+
.map((bullet) => String(bullet.index));
|
|
2211
|
+
process.stdout.write(
|
|
2212
|
+
`${outcome.scenario.id}: ${outcome.attribution.verdict}; the baseline passed bullet${regressed.length === 1 ? "" : "s"} ${regressed.join(", ")} and the treatment did not\n`,
|
|
2213
|
+
);
|
|
2214
|
+
} else if (outcome.attribution.reason !== null) {
|
|
2215
|
+
process.stdout.write(
|
|
2216
|
+
`${outcome.scenario.id}: attribution ${outcome.attribution.verdict}: ${outcome.attribution.reason}\n`,
|
|
2217
|
+
);
|
|
2218
|
+
}
|
|
2219
|
+
}
|
|
2220
|
+
}
|
|
2221
|
+
|
|
1214
2222
|
function preview(value: unknown): string {
|
|
1215
2223
|
if (value === undefined) return "";
|
|
1216
2224
|
const text = typeof value === "string" ? value : safeStringify(value);
|
|
@@ -1249,7 +2257,7 @@ function isRecord(value: unknown): value is Record<string, unknown> {
|
|
|
1249
2257
|
|
|
1250
2258
|
/** The experimental evals.md lane stays under eval, alongside package suite evals. */
|
|
1251
2259
|
export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<number> {
|
|
1252
|
-
const options: SkillsEvalOptions = { json: false, trustFixtures: false, allowNetwork: false };
|
|
2260
|
+
const options: SkillsEvalOptions = { json: false, trustFixtures: false, allowNetwork: false, noAttribution: false };
|
|
1253
2261
|
let source: string | undefined;
|
|
1254
2262
|
try {
|
|
1255
2263
|
for (let i = 0; i < args.length; i++) {
|
|
@@ -1258,6 +2266,7 @@ export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<numb
|
|
|
1258
2266
|
if (arg === "--json") options.json = true;
|
|
1259
2267
|
else if (arg === "--trust-fixtures") options.trustFixtures = true;
|
|
1260
2268
|
else if (arg === "--allow-network") options.allowNetwork = true;
|
|
2269
|
+
else if (arg === "--no-attribution") options.noAttribution = true;
|
|
1261
2270
|
else if (["--scenario", "--target", "--workspace", "--timeout"].includes(arg)) {
|
|
1262
2271
|
const value = args[++i];
|
|
1263
2272
|
if (!value || value.startsWith("-")) throw new Error(`${arg} requires a value`);
|
|
@@ -1270,7 +2279,8 @@ export async function runSkillEvalCli(args: ReadonlyArray<string>): Promise<numb
|
|
|
1270
2279
|
else options.workspace = value;
|
|
1271
2280
|
} else if (arg === "--help" || arg === "-h") {
|
|
1272
2281
|
process.stdout.write(
|
|
1273
|
-
"clio-coder eval skill <name|path> [--scenario <id>] [--target <id>] [--workspace <path>] [--timeout <seconds>] [--trust-fixtures] [--allow-network] [--json]\n"
|
|
2282
|
+
"clio-coder eval skill <name|path> [--scenario <id>] [--target <id>] [--workspace <path>] [--timeout <seconds>] [--trust-fixtures] [--allow-network] [--no-attribution] [--json]\n" +
|
|
2283
|
+
" --no-attribution skip the baseline judge and the paired comparison; saves one judge run per scenario\n",
|
|
1274
2284
|
);
|
|
1275
2285
|
return 0;
|
|
1276
2286
|
} else if (arg.startsWith("-") || source) throw new Error(`unexpected skill eval argument: ${arg}`);
|