@iowarp/clio-coder 0.3.8 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +102 -0
- package/NOTICE +33 -0
- package/README.md +12 -3
- package/dist/{acp-U67UHUK2.js → acp-G5WJBNCT.js} +13 -12
- package/dist/{agents-YU6SGALZ.js → agents-FMV2Q5G4.js} +41 -36
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-3IDSJEIK.js} +18 -18
- package/dist/{builtins-C6JMZVV6.js → builtins-XCZWXSC7.js} +5 -5
- package/dist/{chunk-IFBNV6H6.js → chunk-2ANTL7MR.js} +3 -3
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/{chunk-XWSF374K.js → chunk-2OQE55CK.js} +3 -3
- package/dist/{chunk-5H3GB5BO.js → chunk-32KWKNSF.js} +8 -384
- package/dist/{chunk-KTYTFRMB.js → chunk-36EJLSQQ.js} +34 -36
- package/dist/{chunk-FHJEP5SW.js → chunk-3BT2XMV4.js} +19 -13
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-GU2UIAFZ.js → chunk-3URVFKWK.js} +7 -7
- package/dist/{chunk-WEH5XRJQ.js → chunk-3XML7CDN.js} +3 -3
- package/dist/chunk-42FMPA75.js +101 -0
- package/dist/{chunk-A2NJGIB3.js → chunk-5LXZXPKX.js} +2 -2
- package/dist/{chunk-VCBR6CU7.js → chunk-5PSMVOLM.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-6DB53AJS.js} +203 -25
- package/dist/{chunk-3BINW3FP.js → chunk-76ONBSIA.js} +2 -2
- package/dist/{chunk-26LEYJZH.js → chunk-7MCTRUCE.js} +2 -2
- package/dist/{chunk-TYPGUK6W.js → chunk-AB44T6BB.js} +111 -5
- package/dist/{chunk-K4XHGFR5.js → chunk-B7OBL7PK.js} +317 -721
- package/dist/{chunk-B5CSFE7B.js → chunk-BBVJUZHB.js} +2 -2
- package/dist/{chunk-EQ63NRB7.js → chunk-BBVYXMFO.js} +2 -2
- package/dist/{chunk-ZVJ5BLO2.js → chunk-BKFJHQCA.js} +154 -16
- package/dist/chunk-BKFM6EJV.js +462 -0
- package/dist/chunk-BMS5RKQY.js +27 -0
- package/dist/{chunk-ZNLWCMVZ.js → chunk-BPKCPIL7.js} +2 -2
- package/dist/{chunk-TLQJPP24.js → chunk-BUMFYQFY.js} +1394 -1349
- package/dist/{chunk-BNAZZHFG.js → chunk-BYP5D4HI.js} +1 -1
- package/dist/chunk-C2LTL2W6.js +2447 -0
- package/dist/{chunk-ME6CCNFO.js → chunk-CODPRO7Q.js} +8 -8
- package/dist/{chunk-E77JEWSD.js → chunk-CTJ4RNAA.js} +7 -37
- package/dist/{chunk-ODFEOB4F.js → chunk-CY6FY24N.js} +26 -8
- package/dist/{chunk-7RFXX52T.js → chunk-DQITNCXG.js} +642 -172
- package/dist/{chunk-TTHACPOM.js → chunk-DYJP44XW.js} +578 -117
- package/dist/{chunk-2HFZQUHL.js → chunk-F4CKPOEQ.js} +18 -8
- package/dist/{chunk-DGSYXYMX.js → chunk-FEFIFZTL.js} +3 -3
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-MV3K5QF2.js → chunk-GCSMB2KY.js} +2 -2
- package/dist/chunk-GKF55TAZ.js +391 -0
- package/dist/{chunk-VWZOAB7K.js → chunk-GR5G2PVF.js} +9 -8
- package/dist/{chunk-4DGYLA73.js → chunk-GXNLGKAB.js} +80 -9
- package/dist/chunk-HHV2GANA.js +88 -0
- package/dist/{chunk-IIZWH4XA.js → chunk-HI63TFOG.js} +5 -4
- package/dist/{chunk-WLFILSD5.js → chunk-HJJTYUHX.js} +113 -83
- package/dist/{chunk-TT36MB5S.js → chunk-HLW2MRKE.js} +3 -1
- package/dist/{chunk-PMDBGQSJ.js → chunk-HWHKMHUA.js} +7 -7
- package/dist/chunk-HZHHCK24.js +1631 -0
- package/dist/{chunk-WSB3FPX7.js → chunk-I5VEOC6I.js} +39 -143
- package/dist/chunk-IBEBSCYA.js +564 -0
- package/dist/chunk-IQ7KR472.js +362 -0
- package/dist/{chunk-A3WNZD3P.js → chunk-J4W7KFM7.js} +949 -972
- package/dist/{chunk-TB5666IT.js → chunk-JDG2WCRO.js} +5 -5
- package/dist/{chunk-N22QMJKY.js → chunk-K5C3NCBD.js} +4 -4
- package/dist/{chunk-CGKSTWHD.js → chunk-K6BSR66V.js} +2 -1
- package/dist/{chunk-5C3AQNDW.js → chunk-KFZI4NIL.js} +216 -38
- package/dist/chunk-KMVISBZR.js +132 -0
- package/dist/{chunk-WXY7KU3G.js → chunk-LDQ2ZF2M.js} +2 -2
- package/dist/{chunk-XN3L4EYL.js → chunk-LQ3DZAMX.js} +3 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-LY4S7GJC.js} +173 -144
- package/dist/{chunk-4SPRNWDE.js → chunk-MLKNTWH2.js} +19 -19
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/chunk-NEKRRTYW.js +56 -0
- package/dist/{chunk-SPULKLCF.js → chunk-NHCZP4K7.js} +3 -3
- package/dist/chunk-NHLBIGRH.js +1506 -0
- package/dist/chunk-NQQH3YT7.js +302 -0
- package/dist/chunk-NYS75XW5.js +15 -0
- package/dist/{chunk-GN57SG4G.js → chunk-O4XIVISU.js} +10 -8
- package/dist/{chunk-EMYUUSFG.js → chunk-O6TL7WWY.js} +6 -6
- package/dist/chunk-OQBA45DZ.js +97 -0
- package/dist/{chunk-LU7P4LHA.js → chunk-P3FOHJT4.js} +2 -2
- package/dist/chunk-PMZCIOCJ.js +25 -0
- package/dist/{chunk-I4HZDVNP.js → chunk-PQEFIJ36.js} +2 -2
- package/dist/{chunk-J3YUBZWY.js → chunk-QBJA7R7N.js} +62 -6
- package/dist/chunk-QDC3K2U3.js +262 -0
- package/dist/{chunk-2HEJ2F35.js → chunk-QLFS5GO2.js} +22 -10
- package/dist/{chunk-7RGZWPB6.js → chunk-QLL7ILRG.js} +95 -32
- package/dist/{chunk-YS5VLNH5.js → chunk-QREDIESB.js} +6 -6
- package/dist/chunk-QSNYB6ZV.js +195 -0
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-P43ETTHK.js → chunk-SJ5ZKQ4S.js} +2 -2
- package/dist/{chunk-GPIEI3LY.js → chunk-SP2RXXYO.js} +6 -54
- package/dist/chunk-SUCTJL45.js +45 -0
- package/dist/{chunk-DYIM5TJT.js → chunk-SUW5DORT.js} +263 -7
- package/dist/chunk-T56WDKA5.js +183 -0
- package/dist/chunk-TVHHYFHE.js +255 -0
- package/dist/{chunk-FJ3H4MN5.js → chunk-TZ3SGWZZ.js} +3 -3
- package/dist/{chunk-MXKJU4JB.js → chunk-U77AMWDL.js} +91 -10
- package/dist/{chunk-5DHKRSMQ.js → chunk-ULC6OTWO.js} +11 -7
- package/dist/{chunk-RWSI4YD7.js → chunk-UM7N4G5A.js} +33 -12
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-IGWKHNIQ.js → chunk-UXMFQ54G.js} +44 -37
- package/dist/{chunk-HFSBBKSQ.js → chunk-V5DHCITQ.js} +171 -3
- package/dist/{chunk-5WIGXA4T.js → chunk-VAZSBTKF.js} +111 -4
- package/dist/{chunk-VHN4MY6O.js → chunk-VEO4AP2K.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-VFA6GDY5.js} +65 -4
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/chunk-VO67MWHC.js +75 -0
- package/dist/{chunk-XK56QHLX.js → chunk-VPKWYKEY.js} +19 -5
- package/dist/{chunk-TANS5ZJS.js → chunk-VYMXRQI6.js} +36 -22
- package/dist/{chunk-WWCZ5F23.js → chunk-W5VSYASO.js} +77 -16
- package/dist/{chunk-WNIJTQQK.js → chunk-WZR7K7ZX.js} +72 -116
- package/dist/{chunk-5Q2VVUKB.js → chunk-X3YGUTOB.js} +4 -4
- package/dist/chunk-X75E3D2N.js +686 -0
- package/dist/{chunk-VAWNZU7Z.js → chunk-YDFRH54B.js} +4 -4
- package/dist/chunk-YJX4SHTD.js +40 -0
- package/dist/{chunk-ZI647VB5.js → chunk-YPI3QQCF.js} +2 -2
- package/dist/chunk-Z2RR6MAK.js +127 -0
- package/dist/{chunk-FBVTI2TJ.js → chunk-Z4TXYIEG.js} +12 -131
- package/dist/cli/index.js +47 -36
- package/dist/{clio-QVTYJ57A.js → clio-2JXHBBY5.js} +7 -7
- package/dist/{code-nav-FGGFIE7L.js → code-nav-3YYRMYNF.js} +8 -8
- package/dist/{compile-cache-CVJMMODC.js → compile-cache-7FPE6PS3.js} +3 -3
- package/dist/{components-ZFA3SAER.js → components-RYZV4JGP.js} +5 -5
- package/dist/{config-LW5IJFQN.js → config-QZPCMYSO.js} +99 -63
- package/dist/{configure-7XIZCOU4.js → configure-TEGEBYCA.js} +23 -22
- package/dist/{context-Y6Y7QPR6.js → context-AV7OEZ4D.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-E6H5RNMC.js} +56 -47
- package/dist/{context-N52ZA626.js → context-GSXUE4CT.js} +29 -27
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-SHIBYK6T.js} +55 -46
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-5ZGKPGZQ.js} +13 -13
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-EFMJT4LD.js} +93 -64
- package/dist/{docs-7LQ23DLM.js → docs-23KQS3XK.js} +5 -5
- package/dist/doctor-QOA5FNY5.js +313 -0
- package/dist/{eval-BEC2WHDA.js → eval-TFBYQH4H.js} +2032 -156
- package/dist/eval-inventory-SXH7PDKX.js +316 -0
- package/dist/{evidence-REJUMSKM.js → evidence-ERGESKGN.js} +203 -48
- package/dist/{evolve-PY5ZBA5K.js → evolve-VDXTSYCJ.js} +52 -43
- package/dist/{extensions-HVKU65YU.js → extensions-7BGBHN57.js} +13 -7
- package/dist/{fleet-7WZEWRFA.js → fleet-2RRVDF2V.js} +228 -108
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-VJ726XIA.js} +11 -11
- package/dist/fleet-decisions-EPAPM3XJ.js +157 -0
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-JF5QOATM.js} +18 -15
- package/dist/fleet-inspect-VLY4S7QM.js +442 -0
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-AIZUEJOY.js} +5 -5
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-AJRPDMDV.js} +22 -19
- package/dist/fleet-verify-JFEL2L3H.js +175 -0
- package/dist/fleet-view-ZCON35AG.js +102 -0
- package/dist/{init-OG3TPGQG.js → init-DN2WWLFE.js} +72 -62
- package/dist/install-XGLBQY5E.js +13 -0
- package/dist/interop-OZBKXAYL.js +114 -0
- package/dist/{library-CNTMPLRF.js → library-YWZG7IMW.js} +21 -18
- package/dist/{memory-6IS7F275.js → memory-I4C4HMLW.js} +54 -45
- package/dist/{models-ENRJDA5W.js → models-CEYXJBO6.js} +33 -30
- package/dist/{monitor-XLDVO7TN.js → monitor-NZ6GCI3P.js} +59 -52
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-GCGQ4N5I.js} +7827 -7328
- package/dist/panes-HMABYVO4.js +58 -0
- package/dist/panes-KY6W3V2E.js +103 -0
- package/dist/{paths-DBXMZMDU.js → paths-II4K7DNR.js} +5 -5
- package/dist/{reset-RZ4ER727.js → reset-DQ6FGCSH.js} +13 -11
- package/dist/resources-BB3MVJMD.js +111 -0
- package/dist/{run-Y2CNK5RU.js → run-H2GQDUER.js} +119 -87
- package/dist/{share-A55GYP6Z.js → share-GTJN6A5O.js} +20 -17
- package/dist/{skills-ALC5J6AT.js → skills-L55TEW6R.js} +33 -24
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-XVXPH2JI.js} +67 -56
- package/dist/skills-inventory-S4MXPJFV.js +126 -0
- package/dist/slash-commands-ZSGASKJC.js +77 -0
- package/dist/{steer-GGWFUJUD.js → steer-RZGSCY4R.js} +4 -4
- package/dist/{support-MIETYA5E.js → support-PKEUNNQL.js} +6 -6
- package/dist/{targets-VGNXIR3S.js → targets-NCPZ644J.js} +68 -40
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-44SV3YCN.js} +6 -4
- package/dist/tools-DAF3DI3C.js +27 -0
- package/dist/{trace-PNCASAXC.js → trace-FYVW2MQA.js} +207 -10
- package/dist/tui-primitives-2AKXQNZK.js +13 -0
- package/dist/{uninstall-ZJF5H5ZN.js → uninstall-DW2PNOIC.js} +5 -5
- package/dist/{upgrade-FUSUAGHR.js → upgrade-3XPP6OQL.js} +27 -24
- package/dist/{usage-N4MKVHKD.js → usage-3NLHGTU2.js} +114 -62
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-SSQONKRT.js} +172 -13
- package/dist/{verify-LTDHYBGY.js → verify-3U6J7FZI.js} +10 -10
- package/dist/{web-fetch-2YHJ3KTG.js → web-fetch-S7RR6GZ7.js} +3 -3
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-CEHYGPGQ.js} +75 -65
- package/dist/with-panes-MKB46MPQ.js +782 -0
- package/dist/worker/entry.js +106 -75
- package/docs/README.md +3 -2
- package/docs/acp.md +24 -3
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +2 -2
- package/docs/artifact-versions.md +9 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +62 -3
- package/docs/commands-and-modes.md +31 -2
- package/docs/configuration-and-targets.md +69 -12
- package/docs/context-engine.md +63 -4
- package/docs/development-pipeline.md +19 -0
- package/docs/dispatch-typed-intent.md +385 -0
- package/docs/documentation-coverage.md +5 -5
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +3 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +12 -11
- package/docs/evolution.md +1 -1
- package/docs/exit-codes-and-output.md +1 -1
- package/docs/extensions-and-sharing.md +27 -1
- package/docs/fleet-dispatch.md +25 -4
- package/docs/installation-and-lifecycle.md +15 -2
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +10 -1
- package/docs/observability.md +55 -4
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +2 -2
- package/docs/scientific-validation.md +3 -3
- package/docs/session-lifecycle.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +18 -8
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +88 -1
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +5 -2
- package/src/cli/acp.ts +6 -2
- package/src/cli/agents.ts +1 -1
- package/src/cli/argv.ts +25 -0
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/configure.ts +23 -21
- package/src/cli/doctor-panes.ts +124 -0
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor-toolchain.ts +57 -0
- package/src/cli/doctor.ts +22 -1
- package/src/cli/eval-inventory.ts +436 -0
- package/src/cli/eval.ts +93 -16
- package/src/cli/evidence-detail.ts +88 -0
- package/src/cli/evidence-inventory.ts +183 -0
- package/src/cli/evidence.ts +30 -5
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet-decisions.ts +69 -0
- package/src/cli/fleet-inspect.ts +334 -0
- package/src/cli/fleet-verify.ts +133 -0
- package/src/cli/fleet-view.ts +810 -0
- package/src/cli/fleet.ts +179 -39
- package/src/cli/index.ts +14 -2
- package/src/cli/interop-inspect.ts +128 -0
- package/src/cli/interop.ts +34 -0
- package/src/cli/panes.ts +35 -0
- package/src/cli/reset.ts +5 -2
- package/src/cli/run.ts +58 -0
- package/src/cli/skills-inventory.ts +185 -0
- package/src/cli/skills.ts +16 -13
- package/src/cli/targets.ts +44 -13
- package/src/cli/tools.ts +321 -0
- package/src/cli/trace-inspect.ts +252 -0
- package/src/cli/trace.ts +85 -5
- package/src/cli/usage.ts +63 -14
- package/src/cli/verifiers-inspect.ts +347 -0
- package/src/cli/verifiers.ts +9 -0
- package/src/core/bus-events.ts +33 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +67 -0
- package/src/core/defaults.ts +124 -8
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +80 -6
- package/src/core/theme-token-hex.ts +43 -0
- package/src/core/tool-names.ts +2 -1
- package/src/core/xdg.ts +1 -1
- package/src/domains/agents/fleets/build-review.md +0 -3
- package/src/domains/agents/fleets/build-test.md +0 -3
- package/src/domains/agents/result-contract-filesystem.ts +32 -0
- package/src/domains/agents/result-contract.ts +164 -35
- package/src/domains/config/classify.ts +6 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission-error.ts +9 -0
- package/src/domains/dispatch/admission.ts +52 -14
- package/src/domains/dispatch/capacity-lease.ts +118 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/council-topology.ts +398 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +378 -82
- package/src/domains/dispatch/fleet-node-prompt.ts +62 -0
- package/src/domains/dispatch/fleet-plan.ts +7 -2
- package/src/domains/dispatch/fleet-run.ts +87 -4
- package/src/domains/dispatch/gate-decisions.ts +11 -1
- package/src/domains/dispatch/gate-role-prompts.ts +9 -0
- package/src/domains/dispatch/gate-topology.ts +289 -0
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +22 -0
- package/src/domains/dispatch/intent-compatibility.ts +330 -0
- package/src/domains/dispatch/intent.ts +85 -1
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +139 -11
- package/src/domains/dispatch/run-event-journal-bridge.ts +149 -0
- package/src/domains/dispatch/run-event-journal.ts +598 -0
- package/src/domains/dispatch/state.ts +49 -1
- package/src/domains/dispatch/types.ts +13 -0
- package/src/domains/dispatch/validation.ts +33 -8
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +342 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/inventory.ts +113 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +105 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/store.ts +6 -0
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -1
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +82 -17
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +3 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +97 -21
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/mux/contract.ts +434 -0
- package/src/domains/mux/detect.ts +158 -0
- package/src/domains/mux/extension.ts +47 -0
- package/src/domains/mux/index.ts +96 -0
- package/src/domains/mux/manifest.ts +6 -0
- package/src/domains/mux/operations.ts +164 -0
- package/src/domains/mux/pane-registry.ts +90 -0
- package/src/domains/mux/protocol.ts +49 -0
- package/src/domains/mux/socket-client.ts +816 -0
- package/src/domains/mux/types.ts +222 -0
- package/src/domains/mux/viewer-command.ts +59 -0
- package/src/domains/mux/yazi/assets/init.lua +2 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/LICENSE +21 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/README.md +78 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/main.lua +255 -0
- package/src/domains/mux/yazi/assets/plugins/git.yazi/types.lua +12 -0
- package/src/domains/mux/yazi/assets/yazi.toml +17 -0
- package/src/domains/mux/yazi/event-stream.ts +180 -0
- package/src/domains/mux/yazi/profile.ts +299 -0
- package/src/domains/mux/yazi/session.ts +228 -0
- package/src/domains/mux/yazi/theme.ts +30 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +22 -1
- package/src/domains/observability/index.ts +9 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +234 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +228 -0
- package/src/domains/providers/endpoint-slots-store.ts +189 -0
- package/src/domains/providers/extension.ts +20 -3
- package/src/domains/providers/index.ts +32 -0
- package/src/domains/providers/model-runtime-capabilities.ts +32 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/boot-manifest.ts +1 -0
- package/src/domains/providers/runtimes/builtins.ts +2 -0
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/runtimes/protocol/litellm.ts +375 -0
- package/src/domains/providers/support.ts +1 -0
- package/src/domains/providers/target-model-cache.ts +124 -0
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/index.ts +3 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/resources/skills/loader.ts +33 -0
- package/src/domains/safety/action-classifier.ts +6 -0
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/domains/toolchain/archive.ts +175 -0
- package/src/domains/toolchain/contract.ts +28 -0
- package/src/domains/toolchain/extension.ts +47 -0
- package/src/domains/toolchain/index.ts +39 -0
- package/src/domains/toolchain/install.ts +327 -0
- package/src/domains/toolchain/manifest.ts +8 -0
- package/src/domains/toolchain/paths.ts +34 -0
- package/src/domains/toolchain/registry.ts +265 -0
- package/src/domains/toolchain/remove.ts +218 -0
- package/src/domains/toolchain/resolve.ts +182 -0
- package/src/domains/toolchain/types.ts +113 -0
- package/src/domains/toolchain/version.ts +88 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/acp/server.ts +413 -70
- package/src/engine/acp/types.ts +19 -1
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/claude/sdk-module.ts +98 -0
- package/src/engine/claude/sdk-runtime.ts +19 -11
- package/src/engine/provider-payload.ts +29 -1
- package/src/engine/tui-primitives.ts +21 -0
- package/src/engine/tui.ts +1 -0
- package/src/engine/worker-runtime.ts +2 -13
- package/src/entry/boot-options.ts +2 -0
- package/src/entry/orchestrator.ts +279 -35
- package/src/entry/panes-activation.ts +31 -0
- package/src/entry/with-panes.ts +20 -0
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +41 -10
- package/src/interactive/cost-overlay.ts +66 -6
- package/src/interactive/council-grid.ts +1 -3
- package/src/interactive/council.ts +11 -0
- package/src/interactive/dispatch-board.ts +102 -20
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +177 -7
- package/src/interactive/interactive-input-runtime.ts +15 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +63 -10
- package/src/interactive/memory-overlay.ts +9 -0
- package/src/interactive/modal-marker.ts +170 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/mux-bridge.ts +214 -0
- package/src/interactive/overlay-frame.ts +58 -2
- package/src/interactive/overlay-general-openers.ts +17 -0
- package/src/interactive/overlay-key-routing.ts +52 -3
- package/src/interactive/overlay-lifecycle.ts +56 -7
- package/src/interactive/overlay-model-selectors.ts +40 -3
- package/src/interactive/overlay-permission-lifecycle.ts +112 -24
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlay-transitions.ts +18 -4
- package/src/interactive/overlays/agents.ts +1 -0
- package/src/interactive/overlays/ask-user.ts +227 -49
- package/src/interactive/overlays/auth-dialog.ts +1 -0
- package/src/interactive/overlays/context-reset.ts +1 -0
- package/src/interactive/overlays/cwd-fallback.ts +1 -0
- package/src/interactive/overlays/decisions.ts +11 -11
- package/src/interactive/overlays/extensions.ts +1 -0
- package/src/interactive/overlays/fleet-run-approval.ts +1 -0
- package/src/interactive/overlays/handoff-review.ts +1 -0
- package/src/interactive/overlays/help-reference.ts +6 -0
- package/src/interactive/overlays/interop.ts +1 -0
- package/src/interactive/overlays/library-install-confirm.ts +1 -0
- package/src/interactive/overlays/library-tabs.ts +28 -0
- package/src/interactive/overlays/list-overlay.ts +10 -1
- package/src/interactive/overlays/message-picker.ts +1 -0
- package/src/interactive/overlays/model-scope.ts +86 -0
- package/src/interactive/overlays/model-selector.ts +1 -0
- package/src/interactive/overlays/prompts.ts +12 -1
- package/src/interactive/overlays/session-selector.ts +1 -0
- package/src/interactive/overlays/settings-sections.ts +30 -0
- package/src/interactive/overlays/settings.ts +614 -45
- package/src/interactive/overlays/side-question.ts +1 -0
- package/src/interactive/overlays/skills-hub.ts +3 -11
- package/src/interactive/overlays/tree-selector.ts +1 -0
- package/src/interactive/pane-policy.ts +46 -0
- package/src/interactive/panes-runtime.ts +292 -0
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/compaction-summary.ts +29 -0
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/renderers/worker-entry.ts +122 -14
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/slash-commands.ts +251 -15
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/tasks-overlay.ts +1 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/theme/tokens.ts +3 -14
- package/src/interactive/turn-context.ts +346 -33
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/artifacts.ts +109 -1
- package/src/interactive/view/view-overlay.ts +29 -3
- package/src/interactive/watch-pane.ts +152 -0
- package/src/interactive/worker-progress.ts +7 -1
- package/src/interactive/worker-receipts.ts +19 -1
- package/src/interactive/worker-stream.ts +5 -0
- package/src/interactive/yazi-bridge.ts +444 -0
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/bootstrap.ts +26 -2
- package/src/tools/builtin-tool-catalog.ts +15 -0
- package/src/tools/compete-worktrees.ts +83 -2
- package/src/tools/core-bootstrap.ts +2 -1
- package/src/tools/dispatch-admission.ts +14 -3
- package/src/tools/dispatch-arguments.ts +20 -20
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-run-events.ts +134 -19
- package/src/tools/dispatch-runner.ts +29 -7
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/dispatch-types.ts +15 -3
- package/src/tools/dispatch.ts +1 -1
- package/src/tools/executables.ts +17 -14
- package/src/tools/observation.ts +54 -4
- package/src/tools/panes-surface.ts +38 -0
- package/src/tools/panes.ts +112 -0
- package/src/tools/policy.ts +10 -1
- package/src/tools/presentation.ts +1 -0
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HLE42MG7.js +0 -37
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-JOZYP4GM.js +0 -279
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -1,32 +1,79 @@
|
|
|
1
1
|
import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
|
+
EVAL_SUITE_V2_VERSION
|
|
4
|
+
} from "./chunk-KMVISBZR.js";
|
|
5
|
+
import {
|
|
6
|
+
loadFragments,
|
|
3
7
|
renderCodewikiDigest
|
|
4
|
-
} from "./chunk-
|
|
8
|
+
} from "./chunk-VAZSBTKF.js";
|
|
5
9
|
import {
|
|
6
10
|
EvalTaskFileError,
|
|
7
11
|
TRUST_STATUS_AXES,
|
|
8
12
|
adaptRunReceiptTrustStatus,
|
|
9
|
-
|
|
13
|
+
assertComparableTrackedMetricSources,
|
|
10
14
|
evalClioProvenance,
|
|
11
15
|
evalEnvironmentProvenance,
|
|
16
|
+
evalServingConfiguration,
|
|
17
|
+
evalServingObservationFrom,
|
|
12
18
|
formatTrustSummary,
|
|
13
19
|
inspectRunReceiptTrustStatus,
|
|
14
|
-
loadEvalArtifactV4,
|
|
15
20
|
loadEvalTaskFile,
|
|
16
21
|
summarizeTrustStatus,
|
|
17
|
-
verifyReceiptIntegrity
|
|
22
|
+
verifyReceiptIntegrity
|
|
23
|
+
} from "./chunk-B7OBL7PK.js";
|
|
24
|
+
import {
|
|
25
|
+
listSessionLedgerRefs,
|
|
26
|
+
parseSessionEntries
|
|
27
|
+
} from "./chunk-QREDIESB.js";
|
|
28
|
+
import "./chunk-3DPEIQKN.js";
|
|
29
|
+
import {
|
|
30
|
+
EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
|
|
31
|
+
EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
32
|
+
EVAL_TRACKED_METRIC_NAMES,
|
|
33
|
+
EVAL_VERDICT_SCHEMA_V1,
|
|
34
|
+
assertEvalBehaviorReferencesVerdictV1,
|
|
35
|
+
buildEvalBehaviorMetricsV1,
|
|
36
|
+
createEvalId,
|
|
37
|
+
evalHarnessMetricsFromReceipt,
|
|
38
|
+
evalServingConfigurationOf,
|
|
39
|
+
judgeEvalBehaviorV1,
|
|
40
|
+
loadEvalArtifactV4,
|
|
41
|
+
parseEvalBehaviorScenarioV1,
|
|
42
|
+
parseEvalExecutionMatrixDimensionsV1,
|
|
43
|
+
parseEvalVerdictEnvelopeV1,
|
|
44
|
+
renderEvalServingConfiguration,
|
|
45
|
+
sameEvalServingConfiguration,
|
|
18
46
|
writeEvalArtifactV4
|
|
19
|
-
} from "./chunk-
|
|
47
|
+
} from "./chunk-NHLBIGRH.js";
|
|
48
|
+
import "./chunk-W6GROXXM.js";
|
|
20
49
|
import {
|
|
21
50
|
shellQuote
|
|
22
51
|
} from "./chunk-TXOTCRLG.js";
|
|
23
|
-
import "./chunk-
|
|
52
|
+
import "./chunk-HVDIIIQW.js";
|
|
53
|
+
import {
|
|
54
|
+
discoverAgentRecipes
|
|
55
|
+
} from "./chunk-Z4TXYIEG.js";
|
|
56
|
+
import "./chunk-VYMXRQI6.js";
|
|
57
|
+
import "./chunk-FEFIFZTL.js";
|
|
58
|
+
import "./chunk-RKKLTLYB.js";
|
|
59
|
+
import "./chunk-SUW5DORT.js";
|
|
60
|
+
import "./chunk-SJ5ZKQ4S.js";
|
|
61
|
+
import "./chunk-2JDWVJND.js";
|
|
62
|
+
import {
|
|
63
|
+
createSafetyPolicyEngine
|
|
64
|
+
} from "./chunk-32KWKNSF.js";
|
|
24
65
|
import "./chunk-H7IXIC72.js";
|
|
25
|
-
import "./chunk-
|
|
26
|
-
import "./chunk-
|
|
66
|
+
import "./chunk-5LXZXPKX.js";
|
|
67
|
+
import "./chunk-RAPCMZL4.js";
|
|
68
|
+
import "./chunk-HHV2GANA.js";
|
|
69
|
+
import {
|
|
70
|
+
agentSpecFingerprint,
|
|
71
|
+
normalizeAgentSpec
|
|
72
|
+
} from "./chunk-DYJP44XW.js";
|
|
73
|
+
import "./chunk-GCSMB2KY.js";
|
|
27
74
|
import "./chunk-UL3WSD3F.js";
|
|
28
75
|
import "./chunk-ECH6PKUQ.js";
|
|
29
|
-
import "./chunk-
|
|
76
|
+
import "./chunk-K6BSR66V.js";
|
|
30
77
|
import {
|
|
31
78
|
readCodewiki,
|
|
32
79
|
structuralCodewikiHash
|
|
@@ -38,16 +85,31 @@ import {
|
|
|
38
85
|
enumerateWorkspaceFiles
|
|
39
86
|
} from "./chunk-33YXPOE3.js";
|
|
40
87
|
import "./chunk-7CR24IG7.js";
|
|
41
|
-
import "./chunk-
|
|
88
|
+
import "./chunk-XPLRXC72.js";
|
|
89
|
+
import "./chunk-2ANTL7MR.js";
|
|
42
90
|
import {
|
|
43
91
|
printError
|
|
44
|
-
} from "./chunk-
|
|
92
|
+
} from "./chunk-VPKWYKEY.js";
|
|
45
93
|
import "./chunk-5TSRNF4G.js";
|
|
94
|
+
import "./chunk-CFGTUFWB.js";
|
|
95
|
+
import {
|
|
96
|
+
extractReasoningTokens
|
|
97
|
+
} from "./chunk-UM7N4G5A.js";
|
|
98
|
+
import "./chunk-HLW2MRKE.js";
|
|
99
|
+
import "./chunk-76ONBSIA.js";
|
|
46
100
|
import {
|
|
47
101
|
InvalidIdError
|
|
48
102
|
} from "./chunk-R346GLFC.js";
|
|
103
|
+
import "./chunk-IHXBNWMM.js";
|
|
104
|
+
import "./chunk-VFA6GDY5.js";
|
|
105
|
+
import "./chunk-BBVJUZHB.js";
|
|
106
|
+
import "./chunk-FQ4SKYE4.js";
|
|
107
|
+
import "./chunk-6EJMN2Y3.js";
|
|
49
108
|
import "./chunk-IWHMRKLL.js";
|
|
50
|
-
import "./chunk-
|
|
109
|
+
import "./chunk-GXNLGKAB.js";
|
|
110
|
+
import "./chunk-LL4KHSZI.js";
|
|
111
|
+
import "./chunk-4ZG3XFUR.js";
|
|
112
|
+
import "./chunk-BBVYXMFO.js";
|
|
51
113
|
import "./chunk-SST6Z5JA.js";
|
|
52
114
|
import "./chunk-IKCO5N3L.js";
|
|
53
115
|
import "./chunk-3I7MS7N2.js";
|
|
@@ -57,7 +119,7 @@ import {
|
|
|
57
119
|
import {
|
|
58
120
|
clioDataDir,
|
|
59
121
|
clioStateDir
|
|
60
|
-
} from "./chunk-
|
|
122
|
+
} from "./chunk-BYP5D4HI.js";
|
|
61
123
|
import "./chunk-WEPFGWHJ.js";
|
|
62
124
|
import "./chunk-YXLYO42X.js";
|
|
63
125
|
import {
|
|
@@ -67,31 +129,574 @@ import {
|
|
|
67
129
|
|
|
68
130
|
// src/cli/eval.ts
|
|
69
131
|
init_esm_shims();
|
|
70
|
-
import { resolve as
|
|
132
|
+
import { resolve as resolve8 } from "node:path";
|
|
71
133
|
|
|
72
134
|
// src/domains/eval/compare/compare.ts
|
|
73
135
|
init_esm_shims();
|
|
74
|
-
|
|
136
|
+
|
|
137
|
+
// src/domains/eval/metrics/aggregate.ts
|
|
138
|
+
init_esm_shims();
|
|
139
|
+
function aggregateEvalVerdicts(verdicts) {
|
|
140
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
141
|
+
for (const verdict of verdicts) {
|
|
142
|
+
const group = byScenario.get(verdict.scenarioId) ?? [];
|
|
143
|
+
group.push(verdict);
|
|
144
|
+
byScenario.set(verdict.scenarioId, group);
|
|
145
|
+
}
|
|
146
|
+
return [...byScenario.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([scenarioId, group]) => aggregateScenario(scenarioId, group));
|
|
147
|
+
}
|
|
148
|
+
function aggregateScenario(scenarioId, verdicts) {
|
|
149
|
+
const ordered = [...verdicts].sort((left, right) => left.trialIndex - right.trialIndex);
|
|
150
|
+
const passed = ordered.filter((verdict) => verdict.outcome === "pass").length;
|
|
151
|
+
const failed = ordered.filter((verdict) => verdict.outcome === "fail").length;
|
|
152
|
+
const unmeasured = ordered.filter((verdict) => verdict.outcome === "unmeasured").length;
|
|
153
|
+
const machineryFailures = ordered.filter((verdict) => verdict.machinery === "infrastructure_failure").length;
|
|
154
|
+
const fixed = Object.fromEntries(
|
|
155
|
+
EVAL_TRACKED_METRIC_NAMES.map((name) => [name, distribution(ordered.map((verdict) => verdict.trackedMetrics[name]))])
|
|
156
|
+
);
|
|
157
|
+
const reasons = new Set(ordered.flatMap((verdict) => Object.keys(verdict.trackedMetrics.expectedColdReasons)));
|
|
158
|
+
const expectedColdReasons = Object.fromEntries(
|
|
159
|
+
[...reasons].sort((left, right) => left.localeCompare(right)).map((reason) => [
|
|
160
|
+
reason,
|
|
161
|
+
distribution(
|
|
162
|
+
ordered.map(
|
|
163
|
+
(verdict) => verdict.trackedMetrics.expectedColdReasons[reason] ?? { value: 0, source: "ledger" }
|
|
164
|
+
)
|
|
165
|
+
)
|
|
166
|
+
])
|
|
167
|
+
);
|
|
168
|
+
const k = ordered.length;
|
|
169
|
+
return {
|
|
170
|
+
scenarioId,
|
|
171
|
+
trials: k,
|
|
172
|
+
k,
|
|
173
|
+
passed,
|
|
174
|
+
failed,
|
|
175
|
+
unmeasured,
|
|
176
|
+
machineryFailures,
|
|
177
|
+
passAtK: k > 0 && passed > 0 ? 1 : 0,
|
|
178
|
+
passPowK: k > 0 && passed === k ? 1 : 0,
|
|
179
|
+
trackedMetrics: { ...fixed, expectedColdReasons }
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
function distribution(metrics) {
|
|
183
|
+
const values = metrics.flatMap((metric) => metric.value === null ? [] : [metric.value]);
|
|
184
|
+
const sources = [...new Set(metrics.map((metric) => metric.source))].sort(compareSources);
|
|
185
|
+
if (values.length === 0) {
|
|
186
|
+
return {
|
|
187
|
+
observations: metrics.length,
|
|
188
|
+
measured: 0,
|
|
189
|
+
unmeasured: metrics.length,
|
|
190
|
+
mean: null,
|
|
191
|
+
min: null,
|
|
192
|
+
max: null,
|
|
193
|
+
p90: null,
|
|
194
|
+
variance: null,
|
|
195
|
+
standardDeviation: null,
|
|
196
|
+
sources
|
|
197
|
+
};
|
|
198
|
+
}
|
|
199
|
+
const ordered = [...values].sort((left, right) => left - right);
|
|
200
|
+
const p90Index = Math.max(0, Math.ceil(ordered.length * 0.9) - 1);
|
|
201
|
+
const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
|
|
202
|
+
const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
|
|
203
|
+
return {
|
|
204
|
+
observations: metrics.length,
|
|
205
|
+
measured: values.length,
|
|
206
|
+
unmeasured: metrics.length - values.length,
|
|
207
|
+
mean,
|
|
208
|
+
min: ordered[0] ?? null,
|
|
209
|
+
max: ordered.at(-1) ?? null,
|
|
210
|
+
p90: ordered[p90Index] ?? null,
|
|
211
|
+
variance,
|
|
212
|
+
standardDeviation: Math.sqrt(variance),
|
|
213
|
+
sources
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
function compareSources(left, right) {
|
|
217
|
+
return sourceOrder(left) - sourceOrder(right);
|
|
218
|
+
}
|
|
219
|
+
function sourceOrder(source) {
|
|
220
|
+
if (source === "ledger") return 0;
|
|
221
|
+
if (source === "receipt") return 1;
|
|
222
|
+
return 2;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// src/domains/eval/compare/behavioral.ts
|
|
226
|
+
init_esm_shims();
|
|
227
|
+
|
|
228
|
+
// src/domains/eval/compare/envelope.ts
|
|
229
|
+
init_esm_shims();
|
|
230
|
+
function compareEvalExecutionEnvelopesV1(identity, baseline, candidate, baselineDimensions, candidateDimensions) {
|
|
231
|
+
const leftDimensions = [...baselineDimensions].sort();
|
|
232
|
+
const rightDimensions = [...candidateDimensions].sort();
|
|
233
|
+
if (stableJson(leftDimensions) !== stableJson(rightDimensions)) {
|
|
234
|
+
return { ...identity, fields: ["matrix.dimensions"] };
|
|
235
|
+
}
|
|
236
|
+
if (baseline.some((result) => result.executionEnvelope !== void 0) && baseline.some((result) => result.executionEnvelope === void 0) || candidate.some((result) => result.executionEnvelope !== void 0) && candidate.some((result) => result.executionEnvelope === void 0)) {
|
|
237
|
+
return { ...identity, fields: ["executionEnvelope.missingTrial"] };
|
|
238
|
+
}
|
|
239
|
+
const ignored = new Set(leftDimensions);
|
|
240
|
+
const baselineEnvelopes = uniqueEnvelopes(baseline, ignored);
|
|
241
|
+
const candidateEnvelopes = uniqueEnvelopes(candidate, ignored);
|
|
242
|
+
if (baselineEnvelopes.length === 0 && candidateEnvelopes.length === 0) return null;
|
|
243
|
+
if (baselineEnvelopes.length === 0 || candidateEnvelopes.length === 0) {
|
|
244
|
+
return { ...identity, fields: ["executionEnvelope"] };
|
|
245
|
+
}
|
|
246
|
+
if (baselineEnvelopes.length > 1 || candidateEnvelopes.length > 1) {
|
|
247
|
+
return { ...identity, fields: ["executionEnvelope.withinRunVariance"] };
|
|
248
|
+
}
|
|
249
|
+
const left = baselineEnvelopes[0];
|
|
250
|
+
const right = candidateEnvelopes[0];
|
|
251
|
+
if (left === void 0 || right === void 0 || stableJson(left) === stableJson(right)) return null;
|
|
252
|
+
return { ...identity, fields: differingFields(left, right, ignored) };
|
|
253
|
+
}
|
|
254
|
+
function uniqueEnvelopes(results, ignored) {
|
|
255
|
+
const byIdentity = /* @__PURE__ */ new Map();
|
|
256
|
+
for (const result of results) {
|
|
257
|
+
if (result.executionEnvelope === void 0) continue;
|
|
258
|
+
const normalized = normalizedEnvelope(result.executionEnvelope, ignored);
|
|
259
|
+
byIdentity.set(stableJson(normalized), normalized);
|
|
260
|
+
}
|
|
261
|
+
return [...byIdentity.values()];
|
|
262
|
+
}
|
|
263
|
+
function normalizedEnvelope(envelope, ignored) {
|
|
264
|
+
return {
|
|
265
|
+
...envelope,
|
|
266
|
+
prompt: ignored.has("prompt") ? { fragments: [], compositionHash: null } : envelope.prompt,
|
|
267
|
+
recipe: ignored.has("recipe") ? null : envelope.recipe,
|
|
268
|
+
target: ignored.has("target") ? "<matrix>" : envelope.target,
|
|
269
|
+
wireModel: ignored.has("wireModel") ? null : envelope.wireModel,
|
|
270
|
+
runtime: ignored.has("runtime") ? null : envelope.runtime,
|
|
271
|
+
thinkingLevel: ignored.has("thinkingLevel") ? null : envelope.thinkingLevel,
|
|
272
|
+
toolSignature: ignored.has("toolSignature") ? null : envelope.toolSignature,
|
|
273
|
+
autonomy: ignored.has("autonomy") ? null : envelope.autonomy,
|
|
274
|
+
policyHashes: ignored.has("policy") ? { rulePack: null, project: null } : envelope.policyHashes,
|
|
275
|
+
projectContext: ignored.has("projectContext") ? {
|
|
276
|
+
kind: "none",
|
|
277
|
+
tier: null,
|
|
278
|
+
contentHash: null,
|
|
279
|
+
chars: null,
|
|
280
|
+
sections: [],
|
|
281
|
+
rulesApplied: [],
|
|
282
|
+
operatorProfileApplied: null
|
|
283
|
+
} : envelope.projectContext,
|
|
284
|
+
corpus: ignored.has("corpus") ? { id: "<matrix>", version: "<matrix>" } : envelope.corpus
|
|
285
|
+
};
|
|
286
|
+
}
|
|
287
|
+
function differingFields(left, right, ignored) {
|
|
288
|
+
const fields = [
|
|
289
|
+
["prompt", "prompt", left.prompt, right.prompt],
|
|
290
|
+
["recipe", "recipe", left.recipe, right.recipe],
|
|
291
|
+
["target", "target", left.target, right.target],
|
|
292
|
+
["wireModel", "wireModel", left.wireModel, right.wireModel],
|
|
293
|
+
["runtime", "runtime", left.runtime, right.runtime],
|
|
294
|
+
["thinkingLevel", "thinkingLevel", left.thinkingLevel, right.thinkingLevel],
|
|
295
|
+
["toolSignature", "toolSignature", left.toolSignature, right.toolSignature],
|
|
296
|
+
["autonomy", "autonomy", left.autonomy, right.autonomy],
|
|
297
|
+
["policy", "policyHashes", left.policyHashes, right.policyHashes],
|
|
298
|
+
["projectContext", "projectContext", left.projectContext, right.projectContext],
|
|
299
|
+
["corpus", "corpus", left.corpus, right.corpus]
|
|
300
|
+
];
|
|
301
|
+
return fields.flatMap(
|
|
302
|
+
([dimension, field, baseline, candidate]) => ignored.has(dimension) || stableJson(baseline) === stableJson(candidate) ? [] : [field]
|
|
303
|
+
);
|
|
304
|
+
}
|
|
305
|
+
function stableJson(value) {
|
|
306
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
|
|
307
|
+
if (typeof value === "object" && value !== null) {
|
|
308
|
+
return `{${Object.entries(value).filter(([, entry]) => entry !== void 0).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
|
|
309
|
+
}
|
|
310
|
+
return JSON.stringify(value);
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
// src/domains/eval/compare/behavioral.ts
|
|
314
|
+
function compareEvalBehaviorMetricsV1(baseline, candidate) {
|
|
315
|
+
const baselineGroups = behaviorGroups(baseline);
|
|
316
|
+
const candidateGroups = behaviorGroups(candidate);
|
|
317
|
+
const keys = /* @__PURE__ */ new Set([...baselineGroups.keys(), ...candidateGroups.keys()]);
|
|
318
|
+
const comparisons = [];
|
|
319
|
+
const envelopeMismatches = [];
|
|
320
|
+
const baselineDimensions = baseline.matrix.dimensions ?? [];
|
|
321
|
+
const candidateDimensions = candidate.matrix.dimensions ?? [];
|
|
322
|
+
for (const key of [...keys].sort((left, right) => left.localeCompare(right))) {
|
|
323
|
+
const baselineGroup = baselineGroups.get(key);
|
|
324
|
+
const candidateGroup = candidateGroups.get(key);
|
|
325
|
+
const identity = baselineGroup ?? candidateGroup;
|
|
326
|
+
if (identity === void 0) continue;
|
|
327
|
+
const envelopeMismatch = compareEvalExecutionEnvelopesV1(
|
|
328
|
+
identity,
|
|
329
|
+
baselineGroup?.results ?? [],
|
|
330
|
+
candidateGroup?.results ?? [],
|
|
331
|
+
baselineDimensions,
|
|
332
|
+
candidateDimensions
|
|
333
|
+
);
|
|
334
|
+
if (envelopeMismatch !== null) envelopeMismatches.push(envelopeMismatch);
|
|
335
|
+
const comparability = {
|
|
336
|
+
comparable: envelopeMismatch === null,
|
|
337
|
+
mismatchedFields: envelopeMismatch?.fields ?? []
|
|
338
|
+
};
|
|
339
|
+
for (const definition of EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1) {
|
|
340
|
+
const baselineDistribution = behaviorDistribution(baselineGroup?.results ?? [], definition);
|
|
341
|
+
const candidateDistribution = behaviorDistribution(candidateGroup?.results ?? [], definition);
|
|
342
|
+
comparisons.push({
|
|
343
|
+
scenarioId: identity.scenarioId,
|
|
344
|
+
role: identity.role,
|
|
345
|
+
target: identity.target,
|
|
346
|
+
metric: definition.name,
|
|
347
|
+
family: definition.family,
|
|
348
|
+
direction: definition.direction,
|
|
349
|
+
hardGate: definition.hardGate,
|
|
350
|
+
baseline: baselineDistribution,
|
|
351
|
+
candidate: candidateDistribution,
|
|
352
|
+
change: envelopeMismatch === null ? classifyChange(baselineDistribution.mean, candidateDistribution.mean, definition.direction) : "incomparable",
|
|
353
|
+
meanDelta: subtractNullable(candidateDistribution.mean, baselineDistribution.mean),
|
|
354
|
+
varianceChange: envelopeMismatch === null ? classifyChange(baselineDistribution.variance, candidateDistribution.variance, "lower") : "incomparable",
|
|
355
|
+
varianceDelta: subtractNullable(candidateDistribution.variance, baselineDistribution.variance),
|
|
356
|
+
comparability
|
|
357
|
+
});
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
const failures = comparisons.flatMap(
|
|
361
|
+
(comparison) => comparison.hardGate && (comparison.change === "regressed" || comparison.change === "incomparable" && comparison.baseline.mean !== null && comparison.candidate.mean === null) ? [
|
|
362
|
+
{
|
|
363
|
+
scenarioId: comparison.scenarioId,
|
|
364
|
+
role: comparison.role,
|
|
365
|
+
target: comparison.target,
|
|
366
|
+
metric: comparison.metric,
|
|
367
|
+
change: comparison.change
|
|
368
|
+
}
|
|
369
|
+
] : []
|
|
370
|
+
);
|
|
371
|
+
return {
|
|
372
|
+
comparisons,
|
|
373
|
+
hardGate: {
|
|
374
|
+
pass: failures.length === 0 && envelopeMismatches.length === 0,
|
|
375
|
+
failures,
|
|
376
|
+
envelopeFailures: envelopeMismatches
|
|
377
|
+
},
|
|
378
|
+
envelopeMismatches
|
|
379
|
+
};
|
|
380
|
+
}
|
|
381
|
+
function classifyChange(baseline, candidate, direction) {
|
|
382
|
+
if (baseline === null || candidate === null) return "incomparable";
|
|
383
|
+
if (baseline === candidate) return "unchanged";
|
|
384
|
+
if (direction === "higher") return candidate > baseline ? "improved" : "regressed";
|
|
385
|
+
return candidate < baseline ? "improved" : "regressed";
|
|
386
|
+
}
|
|
387
|
+
function behaviorGroups(artifact) {
|
|
388
|
+
const groups = /* @__PURE__ */ new Map();
|
|
389
|
+
for (const result of artifact.results) {
|
|
390
|
+
const behavioral = result.behavioralMetrics;
|
|
391
|
+
if (behavioral === void 0) continue;
|
|
392
|
+
const key = groupKey(behavioral.scenarioId, behavioral.role, behavioral.target);
|
|
393
|
+
const group = groups.get(key) ?? {
|
|
394
|
+
scenarioId: behavioral.scenarioId,
|
|
395
|
+
role: behavioral.role,
|
|
396
|
+
target: behavioral.target,
|
|
397
|
+
results: []
|
|
398
|
+
};
|
|
399
|
+
group.results.push(result);
|
|
400
|
+
groups.set(key, group);
|
|
401
|
+
}
|
|
402
|
+
return groups;
|
|
403
|
+
}
|
|
404
|
+
function behaviorDistribution(results, definition) {
|
|
405
|
+
const observations = results.map((result) => result.behavioralMetrics?.metrics[definition.name].value ?? null);
|
|
406
|
+
const values = observations.flatMap((value) => value === null ? [] : [value]);
|
|
407
|
+
if (values.length === 0) {
|
|
408
|
+
return {
|
|
409
|
+
observations: observations.length,
|
|
410
|
+
measured: 0,
|
|
411
|
+
unmeasured: observations.length,
|
|
412
|
+
mean: null,
|
|
413
|
+
min: null,
|
|
414
|
+
max: null,
|
|
415
|
+
p90: null,
|
|
416
|
+
variance: null,
|
|
417
|
+
standardDeviation: null,
|
|
418
|
+
source: definition.source
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
const ordered = [...values].sort((left, right) => left - right);
|
|
422
|
+
const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
|
|
423
|
+
const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
|
|
424
|
+
return {
|
|
425
|
+
observations: observations.length,
|
|
426
|
+
measured: values.length,
|
|
427
|
+
unmeasured: observations.length - values.length,
|
|
428
|
+
mean,
|
|
429
|
+
min: ordered[0] ?? null,
|
|
430
|
+
max: ordered.at(-1) ?? null,
|
|
431
|
+
p90: ordered[Math.max(0, Math.ceil(ordered.length * 0.9) - 1)] ?? null,
|
|
432
|
+
variance,
|
|
433
|
+
standardDeviation: Math.sqrt(variance),
|
|
434
|
+
source: definition.source
|
|
435
|
+
};
|
|
436
|
+
}
|
|
437
|
+
function groupKey(scenarioId, role, target) {
|
|
438
|
+
return JSON.stringify([scenarioId, role, target.id, target.model]);
|
|
439
|
+
}
|
|
440
|
+
function subtractNullable(left, right) {
|
|
441
|
+
return left === null || right === null ? null : left - right;
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
// src/domains/eval/compare/compare.ts
|
|
445
|
+
var EvalServingConfigurationDriftError = class extends Error {
|
|
446
|
+
baseline;
|
|
447
|
+
candidate;
|
|
448
|
+
constructor(baseline, candidate) {
|
|
449
|
+
super(
|
|
450
|
+
[
|
|
451
|
+
"serving configuration drift; pass --allow-config-drift to compare these runs",
|
|
452
|
+
`baseline serving: ${renderEvalServingConfiguration(baseline)}`,
|
|
453
|
+
`candidate serving: ${renderEvalServingConfiguration(candidate)}`
|
|
454
|
+
].join("\n")
|
|
455
|
+
);
|
|
456
|
+
this.name = "EvalServingConfigurationDriftError";
|
|
457
|
+
this.baseline = baseline;
|
|
458
|
+
this.candidate = candidate;
|
|
459
|
+
}
|
|
460
|
+
};
|
|
461
|
+
function compareEvalArtifactsV4(baseline, candidate, options = {}) {
|
|
75
462
|
const baselineTokens = baseline.summary.tokens;
|
|
76
463
|
const candidateTokens = candidate.summary.tokens;
|
|
464
|
+
const baselineServing = evalServingConfigurationOf(baseline);
|
|
465
|
+
const candidateServing = evalServingConfigurationOf(candidate);
|
|
466
|
+
const configDrift = !sameEvalServingConfiguration(baselineServing, candidateServing);
|
|
467
|
+
if (configDrift && options.allowConfigDrift !== true) {
|
|
468
|
+
throw new EvalServingConfigurationDriftError(baselineServing, candidateServing);
|
|
469
|
+
}
|
|
470
|
+
const behavioral = compareEvalBehaviorMetricsV1(baseline, candidate);
|
|
471
|
+
const trackedMetrics = compareTrackedMetrics(baseline, candidate, options.metric);
|
|
472
|
+
const normalizedFilter = normalizeMetricFilter(options.metric);
|
|
473
|
+
const behavioralMetrics = normalizedFilter === void 0 ? behavioral.comparisons : behavioral.comparisons.filter((row) => row.metric === normalizedFilter || row.family === normalizedFilter);
|
|
474
|
+
if (options.metric !== void 0 && trackedMetrics.length === 0 && behavioralMetrics.length === 0) {
|
|
475
|
+
throw new Error(`eval metric not found: ${options.metric}`);
|
|
476
|
+
}
|
|
77
477
|
return {
|
|
78
478
|
baselineEvalId: baseline.evalId,
|
|
79
479
|
candidateEvalId: candidate.evalId,
|
|
480
|
+
baselineServingConfiguration: baselineServing,
|
|
481
|
+
candidateServingConfiguration: candidateServing,
|
|
482
|
+
configDrift,
|
|
80
483
|
passRateDelta: candidate.summary.passRate - baseline.summary.passRate,
|
|
81
484
|
tokenDelta: baselineTokens.measured && candidateTokens.measured ? candidateTokens.total - baselineTokens.total : null,
|
|
82
|
-
wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs
|
|
485
|
+
wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs,
|
|
486
|
+
trackedMetrics,
|
|
487
|
+
behavioralMetrics,
|
|
488
|
+
hardGate: behavioral.hardGate,
|
|
489
|
+
envelopeMismatches: behavioral.envelopeMismatches,
|
|
490
|
+
scenarioReports: behaviorRollups(behavioralMetrics, (row) => row.scenarioId),
|
|
491
|
+
roleReports: behaviorRollups(behavioralMetrics, (row) => row.role),
|
|
492
|
+
affectedCorpusResults: behavioral.envelopeMismatches.flatMap((mismatch) => {
|
|
493
|
+
const changedFields = mismatch.fields.filter((field) => field === "prompt" || field === "recipe");
|
|
494
|
+
return changedFields.length === 0 ? [] : [{ scenarioId: mismatch.scenarioId, role: mismatch.role, changedFields }];
|
|
495
|
+
})
|
|
83
496
|
};
|
|
84
497
|
}
|
|
85
498
|
function renderEvalComparisonV4(summary) {
|
|
499
|
+
const envelopeFailures = summary.envelopeMismatches.map(
|
|
500
|
+
(mismatch) => ` incomparable envelope: ${mismatch.scenarioId} ${mismatch.role} ${mismatch.target.id}/${mismatch.target.model ?? "none"} fields=${mismatch.fields.join(",")}`
|
|
501
|
+
);
|
|
502
|
+
const affected = summary.affectedCorpusResults.map(
|
|
503
|
+
(result) => ` affected corpus result: ${result.scenarioId} role=${result.role} changed=${result.changedFields.join(",")}`
|
|
504
|
+
);
|
|
505
|
+
const scenarioReports = renderRollups("per-scenario baseline/candidate report", summary.scenarioReports);
|
|
506
|
+
const roleReports = renderRollups("per-role baseline/candidate report", summary.roleReports);
|
|
507
|
+
const hardFailures = summary.hardGate.failures.map(
|
|
508
|
+
(failure) => ` hard failure: ${failure.scenarioId} ${failure.role} ${failure.target.id}/${failure.target.model ?? "none"} ${failure.metric} ${failure.change}`
|
|
509
|
+
);
|
|
510
|
+
const tracked = summary.trackedMetrics.flatMap((row, index) => [
|
|
511
|
+
...index === 0 ? [
|
|
512
|
+
"tracked metrics:",
|
|
513
|
+
"scenario metric baseline_mean baseline_p90 baseline_variance candidate_mean candidate_p90 candidate_variance mean_delta p90_delta variance_delta change variance_change sources"
|
|
514
|
+
] : [],
|
|
515
|
+
[
|
|
516
|
+
row.scenarioId,
|
|
517
|
+
row.metric,
|
|
518
|
+
formatMetric(row.baseline.mean),
|
|
519
|
+
formatMetric(row.baseline.p90),
|
|
520
|
+
formatMetric(row.baseline.variance ?? null),
|
|
521
|
+
formatMetric(row.candidate.mean),
|
|
522
|
+
formatMetric(row.candidate.p90),
|
|
523
|
+
formatMetric(row.candidate.variance ?? null),
|
|
524
|
+
formatSignedMetric(row.meanDelta),
|
|
525
|
+
formatSignedMetric(row.p90Delta),
|
|
526
|
+
formatSignedMetric(row.varianceDelta),
|
|
527
|
+
row.change,
|
|
528
|
+
row.varianceChange,
|
|
529
|
+
`${row.baseline.sources.join("+") || "none"}->${row.candidate.sources.join("+") || "none"}`
|
|
530
|
+
].join(" ")
|
|
531
|
+
]);
|
|
532
|
+
const behavioral = summary.behavioralMetrics.flatMap((row, index) => [
|
|
533
|
+
...index === 0 ? [
|
|
534
|
+
"behavioral metrics:",
|
|
535
|
+
"scenario role target model family metric baseline_mean baseline_variance baseline_coverage candidate_mean candidate_variance candidate_coverage mean_delta variance_delta change variance_change comparability gate source"
|
|
536
|
+
] : [],
|
|
537
|
+
[
|
|
538
|
+
row.scenarioId,
|
|
539
|
+
row.role,
|
|
540
|
+
row.target.id,
|
|
541
|
+
row.target.model ?? "none",
|
|
542
|
+
row.family,
|
|
543
|
+
row.metric,
|
|
544
|
+
formatMetric(row.baseline.mean),
|
|
545
|
+
formatMetric(row.baseline.variance),
|
|
546
|
+
`${row.baseline.measured}/${row.baseline.observations}`,
|
|
547
|
+
formatMetric(row.candidate.mean),
|
|
548
|
+
formatMetric(row.candidate.variance),
|
|
549
|
+
`${row.candidate.measured}/${row.candidate.observations}`,
|
|
550
|
+
formatSignedMetric(row.meanDelta),
|
|
551
|
+
formatSignedMetric(row.varianceDelta),
|
|
552
|
+
row.change,
|
|
553
|
+
row.varianceChange,
|
|
554
|
+
row.comparability.comparable ? "comparable" : `incomparable:${row.comparability.mismatchedFields.join(",")}`,
|
|
555
|
+
row.hardGate ? "hard" : "informational",
|
|
556
|
+
row.baseline.source
|
|
557
|
+
].join(" ")
|
|
558
|
+
]);
|
|
86
559
|
return [
|
|
87
560
|
`baseline eval: ${summary.baselineEvalId}`,
|
|
88
561
|
`candidate eval: ${summary.candidateEvalId}`,
|
|
562
|
+
`baseline serving: ${renderEvalServingConfiguration(summary.baselineServingConfiguration)}`,
|
|
563
|
+
`candidate serving: ${renderEvalServingConfiguration(summary.candidateServingConfiguration)}`,
|
|
564
|
+
`config drift: ${summary.configDrift ? "allowed" : "none"}`,
|
|
89
565
|
`pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
|
|
90
566
|
`token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
|
|
91
567
|
`wall-time delta ms: ${summary.wallTimeDelta}`,
|
|
568
|
+
`behavioral hard gate: ${summary.hardGate.pass ? "pass" : `fail (${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length})`}`,
|
|
569
|
+
...hardFailures,
|
|
570
|
+
...envelopeFailures,
|
|
571
|
+
...affected,
|
|
572
|
+
...scenarioReports,
|
|
573
|
+
...roleReports,
|
|
574
|
+
...tracked,
|
|
575
|
+
...behavioral,
|
|
92
576
|
""
|
|
93
577
|
].join("\n");
|
|
94
578
|
}
|
|
579
|
+
function behaviorRollups(rows, keyOf) {
|
|
580
|
+
const groups = /* @__PURE__ */ new Map();
|
|
581
|
+
for (const row of rows) groups.set(keyOf(row), [...groups.get(keyOf(row)) ?? [], row]);
|
|
582
|
+
return [...groups.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([id, grouped]) => ({
|
|
583
|
+
id,
|
|
584
|
+
metrics: changeCounts(grouped.map((row) => row.change)),
|
|
585
|
+
variance: changeCounts(grouped.map((row) => row.varianceChange))
|
|
586
|
+
}));
|
|
587
|
+
}
|
|
588
|
+
function changeCounts(changes) {
|
|
589
|
+
return {
|
|
590
|
+
improved: changes.filter((change) => change === "improved").length,
|
|
591
|
+
regressed: changes.filter((change) => change === "regressed").length,
|
|
592
|
+
unchanged: changes.filter((change) => change === "unchanged").length,
|
|
593
|
+
incomparable: changes.filter((change) => change === "incomparable").length
|
|
594
|
+
};
|
|
595
|
+
}
|
|
596
|
+
function renderRollups(title, reports) {
|
|
597
|
+
if (reports.length === 0) return [];
|
|
598
|
+
return [
|
|
599
|
+
`${title}:`,
|
|
600
|
+
...reports.map(
|
|
601
|
+
(report) => ` ${report.id}: metrics ${renderChangeCounts(report.metrics)}; variance ${renderChangeCounts(report.variance)}`
|
|
602
|
+
)
|
|
603
|
+
];
|
|
604
|
+
}
|
|
605
|
+
function renderChangeCounts(counts) {
|
|
606
|
+
return `improved=${counts.improved} regressed=${counts.regressed} unchanged=${counts.unchanged} incomparable=${counts.incomparable}`;
|
|
607
|
+
}
|
|
608
|
+
function compareTrackedMetrics(baseline, candidate, metricFilter) {
|
|
609
|
+
const baselineAggregates = baseline.aggregates ?? aggregateEvalVerdicts(baseline.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
|
|
610
|
+
const candidateAggregates = candidate.aggregates ?? aggregateEvalVerdicts(candidate.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
|
|
611
|
+
const baselineByScenario = new Map(baselineAggregates.map((entry) => [entry.scenarioId, entry]));
|
|
612
|
+
const candidateByScenario = new Map(candidateAggregates.map((entry) => [entry.scenarioId, entry]));
|
|
613
|
+
const scenarioIds = [...baselineByScenario.keys()].filter((scenarioId) => candidateByScenario.has(scenarioId)).sort((left, right) => left.localeCompare(right));
|
|
614
|
+
const filter = normalizeMetricFilter(metricFilter);
|
|
615
|
+
const rows = [];
|
|
616
|
+
for (const scenarioId of scenarioIds) {
|
|
617
|
+
const baselineAggregate = baselineByScenario.get(scenarioId);
|
|
618
|
+
const candidateAggregate = candidateByScenario.get(scenarioId);
|
|
619
|
+
if (baselineAggregate === void 0 || candidateAggregate === void 0) continue;
|
|
620
|
+
for (const metric of EVAL_TRACKED_METRIC_NAMES) {
|
|
621
|
+
if (filter !== void 0 && filter !== metric) continue;
|
|
622
|
+
rows.push(
|
|
623
|
+
metricComparison(
|
|
624
|
+
scenarioId,
|
|
625
|
+
metric,
|
|
626
|
+
baselineAggregate.trackedMetrics[metric],
|
|
627
|
+
candidateAggregate.trackedMetrics[metric]
|
|
628
|
+
)
|
|
629
|
+
);
|
|
630
|
+
}
|
|
631
|
+
const reasons = /* @__PURE__ */ new Set([
|
|
632
|
+
...Object.keys(baselineAggregate.trackedMetrics.expectedColdReasons),
|
|
633
|
+
...Object.keys(candidateAggregate.trackedMetrics.expectedColdReasons)
|
|
634
|
+
]);
|
|
635
|
+
for (const reason of [...reasons].sort((left, right) => left.localeCompare(right))) {
|
|
636
|
+
const metric = `expectedColdReasons.${reason}`;
|
|
637
|
+
if (filter !== void 0 && filter !== metric && filter !== "expectedColdReasons") continue;
|
|
638
|
+
rows.push(
|
|
639
|
+
metricComparison(
|
|
640
|
+
scenarioId,
|
|
641
|
+
metric,
|
|
642
|
+
baselineAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(baselineAggregate.k),
|
|
643
|
+
candidateAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(candidateAggregate.k)
|
|
644
|
+
)
|
|
645
|
+
);
|
|
646
|
+
}
|
|
647
|
+
}
|
|
648
|
+
return rows;
|
|
649
|
+
}
|
|
650
|
+
function metricComparison(scenarioId, metric, baseline, candidate) {
|
|
651
|
+
assertComparableTrackedMetricSources(`${scenarioId}.${metric}`, baseline.sources, candidate.sources);
|
|
652
|
+
const direction = trackedMetricDirection(metric);
|
|
653
|
+
return {
|
|
654
|
+
scenarioId,
|
|
655
|
+
metric,
|
|
656
|
+
baseline,
|
|
657
|
+
candidate,
|
|
658
|
+
meanDelta: subtractNullable2(candidate.mean, baseline.mean),
|
|
659
|
+
p90Delta: subtractNullable2(candidate.p90, baseline.p90),
|
|
660
|
+
varianceDelta: subtractNullable2(candidate.variance ?? null, baseline.variance ?? null),
|
|
661
|
+
change: classifyChange(baseline.mean, candidate.mean, direction),
|
|
662
|
+
varianceChange: classifyChange(baseline.variance ?? null, candidate.variance ?? null, "lower")
|
|
663
|
+
};
|
|
664
|
+
}
|
|
665
|
+
function normalizeMetricFilter(metric) {
|
|
666
|
+
if (metric === void 0) return void 0;
|
|
667
|
+
const trimmed = metric.trim();
|
|
668
|
+
if (trimmed.startsWith("trackedMetrics.")) return trimmed.slice("trackedMetrics.".length);
|
|
669
|
+
if (trimmed.startsWith("behavioralMetrics.")) return trimmed.slice("behavioralMetrics.".length);
|
|
670
|
+
return trimmed;
|
|
671
|
+
}
|
|
672
|
+
function trackedMetricDirection(metric) {
|
|
673
|
+
return metric === "cacheReadTokens" ? "higher" : "lower";
|
|
674
|
+
}
|
|
675
|
+
function zeroDistribution(observations) {
|
|
676
|
+
return {
|
|
677
|
+
observations,
|
|
678
|
+
measured: observations,
|
|
679
|
+
unmeasured: 0,
|
|
680
|
+
mean: 0,
|
|
681
|
+
min: 0,
|
|
682
|
+
max: 0,
|
|
683
|
+
p90: 0,
|
|
684
|
+
variance: 0,
|
|
685
|
+
standardDeviation: 0,
|
|
686
|
+
sources: ["ledger"]
|
|
687
|
+
};
|
|
688
|
+
}
|
|
689
|
+
function subtractNullable2(left, right) {
|
|
690
|
+
return left === null || right === null ? null : left - right;
|
|
691
|
+
}
|
|
692
|
+
function formatMetric(value) {
|
|
693
|
+
return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(2);
|
|
694
|
+
}
|
|
695
|
+
function formatSignedMetric(value) {
|
|
696
|
+
if (value === null) return "null";
|
|
697
|
+
const formatted = formatMetric(value);
|
|
698
|
+
return value > 0 ? `+${formatted}` : formatted;
|
|
699
|
+
}
|
|
95
700
|
|
|
96
701
|
// src/domains/eval/compare/gates.ts
|
|
97
702
|
init_esm_shims();
|
|
@@ -102,12 +707,34 @@ var import_yaml = __toESM(require_dist(), 1);
|
|
|
102
707
|
import { readFileSync } from "node:fs";
|
|
103
708
|
function loadThresholds(path) {
|
|
104
709
|
const parsed = (0, import_yaml.parse)(readFileSync(path, "utf8"));
|
|
105
|
-
|
|
106
|
-
if (isRecord(
|
|
107
|
-
return {
|
|
710
|
+
const root = isRecord(parsed) && isRecord(parsed.thresholds) ? parsed.thresholds : parsed;
|
|
711
|
+
if (isRecord(root) && (Array.isArray(root.fail) || Array.isArray(root.informational))) {
|
|
712
|
+
return {
|
|
713
|
+
fail: parseAssertions(root.fail, `${path}.fail`),
|
|
714
|
+
informational: parseAssertions(root.informational, `${path}.informational`)
|
|
715
|
+
};
|
|
108
716
|
}
|
|
109
717
|
throw new Error(`invalid thresholds file: ${path}`);
|
|
110
718
|
}
|
|
719
|
+
function parseAssertions(value, source) {
|
|
720
|
+
if (value === void 0) return [];
|
|
721
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
722
|
+
return value.map((entry, index) => {
|
|
723
|
+
if (!isRecord(entry)) throw new Error(`${source}[${index}]: expected object`);
|
|
724
|
+
if (typeof entry.metric !== "string" || entry.metric.length === 0) {
|
|
725
|
+
throw new Error(`${source}[${index}].metric: expected non-empty string`);
|
|
726
|
+
}
|
|
727
|
+
if (!isOp(entry.op)) throw new Error(`${source}[${index}].op: expected lt, lte, gt, gte, eq, or neq`);
|
|
728
|
+
if (!isScalar(entry.value)) throw new Error(`${source}[${index}].value: expected scalar`);
|
|
729
|
+
return { metric: entry.metric, op: entry.op, value: entry.value };
|
|
730
|
+
});
|
|
731
|
+
}
|
|
732
|
+
function isOp(value) {
|
|
733
|
+
return value === "lt" || value === "lte" || value === "gt" || value === "gte" || value === "eq" || value === "neq";
|
|
734
|
+
}
|
|
735
|
+
function isScalar(value) {
|
|
736
|
+
return typeof value === "number" && Number.isFinite(value) || typeof value === "string" || typeof value === "boolean";
|
|
737
|
+
}
|
|
111
738
|
function resolveMetricAssertion(assertion, metrics, artifact) {
|
|
112
739
|
const actual = metricValue(assertion.metric, metrics, artifact);
|
|
113
740
|
return { actual, unresolved: actual === null, holds: comparisonHolds(assertion, actual) };
|
|
@@ -152,21 +779,26 @@ function isRecord(value) {
|
|
|
152
779
|
|
|
153
780
|
// src/domains/eval/compare/gates.ts
|
|
154
781
|
function evaluateGate(artifact, thresholds) {
|
|
155
|
-
const failures =
|
|
156
|
-
|
|
782
|
+
const failures = evaluateAssertions(artifact, thresholds.fail);
|
|
783
|
+
const informational = evaluateAssertions(artifact, thresholds.informational ?? []);
|
|
784
|
+
return { pass: failures.length === 0, failures, informational };
|
|
785
|
+
}
|
|
786
|
+
function evaluateAssertions(artifact, assertions) {
|
|
787
|
+
const findings = [];
|
|
788
|
+
for (const assertion of assertions) {
|
|
157
789
|
const whole = resolveMetricAssertion(assertion, {}, artifact);
|
|
158
790
|
if (!whole.unresolved) {
|
|
159
|
-
if (whole.holds)
|
|
791
|
+
if (whole.holds) findings.push({ assertion, actual: whole.actual, unresolved: false });
|
|
160
792
|
continue;
|
|
161
793
|
}
|
|
162
794
|
if (artifact.results.length === 0) {
|
|
163
|
-
|
|
795
|
+
findings.push({ assertion, actual: null, unresolved: true });
|
|
164
796
|
continue;
|
|
165
797
|
}
|
|
166
798
|
for (const result of artifact.results) {
|
|
167
799
|
const perRun = resolveMetricAssertion(assertion, result.metrics);
|
|
168
800
|
if (!perRun.unresolved && !perRun.holds) continue;
|
|
169
|
-
|
|
801
|
+
findings.push({
|
|
170
802
|
assertion,
|
|
171
803
|
actual: perRun.actual,
|
|
172
804
|
unresolved: perRun.unresolved,
|
|
@@ -175,7 +807,7 @@ function evaluateGate(artifact, thresholds) {
|
|
|
175
807
|
});
|
|
176
808
|
}
|
|
177
809
|
}
|
|
178
|
-
return
|
|
810
|
+
return findings;
|
|
179
811
|
}
|
|
180
812
|
function renderGateFailure(failure) {
|
|
181
813
|
const run = failure.taskId === void 0 ? "" : ` [${failure.taskId}#${failure.repeatIndex ?? 0}]`;
|
|
@@ -184,6 +816,121 @@ function renderGateFailure(failure) {
|
|
|
184
816
|
` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(failure.actual)}
|
|
185
817
|
`;
|
|
186
818
|
}
|
|
819
|
+
function renderInformationalBudget(finding) {
|
|
820
|
+
const run = finding.taskId === void 0 ? "" : ` [${finding.taskId}#${finding.repeatIndex ?? 0}]`;
|
|
821
|
+
const { metric, op, value } = finding.assertion;
|
|
822
|
+
return finding.unresolved ? ` ${metric}${run}: unmeasured informational budget
|
|
823
|
+
` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(finding.actual)}
|
|
824
|
+
`;
|
|
825
|
+
}
|
|
826
|
+
|
|
827
|
+
// src/domains/eval/reports/comparison.ts
|
|
828
|
+
init_esm_shims();
|
|
829
|
+
function renderEvalComparisonReportV1(summary, format2) {
|
|
830
|
+
if (format2 === "json") return `${JSON.stringify(summary, null, 2)}
|
|
831
|
+
`;
|
|
832
|
+
if (format2 === "md") return renderMarkdown(summary);
|
|
833
|
+
if (format2 === "junit") return renderJunit(summary);
|
|
834
|
+
return renderEvalComparisonV4(summary);
|
|
835
|
+
}
|
|
836
|
+
function renderMarkdown(summary) {
|
|
837
|
+
const rows = summary.behavioralMetrics.map(
|
|
838
|
+
(row) => `| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`
|
|
839
|
+
);
|
|
840
|
+
const scenarioRows = summary.scenarioReports.map(
|
|
841
|
+
(report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
|
|
842
|
+
);
|
|
843
|
+
const roleRows = summary.roleReports.map(
|
|
844
|
+
(report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
|
|
845
|
+
);
|
|
846
|
+
return [
|
|
847
|
+
`# Eval comparison ${summary.baselineEvalId} \u2192 ${summary.candidateEvalId}`,
|
|
848
|
+
"",
|
|
849
|
+
`Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
|
|
850
|
+
...summary.hardGate.failures.map(
|
|
851
|
+
(failure) => `- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`
|
|
852
|
+
),
|
|
853
|
+
...summary.envelopeMismatches.map(
|
|
854
|
+
(mismatch) => `- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`
|
|
855
|
+
),
|
|
856
|
+
...summary.affectedCorpusResults.map(
|
|
857
|
+
(result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`
|
|
858
|
+
),
|
|
859
|
+
`Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
|
|
860
|
+
`Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
|
|
861
|
+
`Wall-time delta ms: ${summary.wallTimeDelta}`,
|
|
862
|
+
"",
|
|
863
|
+
"| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
|
|
864
|
+
"|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
|
|
865
|
+
...rows,
|
|
866
|
+
"",
|
|
867
|
+
"## Per-scenario baseline/candidate report",
|
|
868
|
+
"",
|
|
869
|
+
"| Scenario | Metric changes | Variance changes |",
|
|
870
|
+
"|---|---|---|",
|
|
871
|
+
...scenarioRows,
|
|
872
|
+
"",
|
|
873
|
+
"## Per-role baseline/candidate report",
|
|
874
|
+
"",
|
|
875
|
+
"| Role | Metric changes | Variance changes |",
|
|
876
|
+
"|---|---|---|",
|
|
877
|
+
...roleRows,
|
|
878
|
+
""
|
|
879
|
+
].join("\n");
|
|
880
|
+
}
|
|
881
|
+
function renderJunit(summary) {
|
|
882
|
+
const failures = new Set(
|
|
883
|
+
summary.hardGate.failures.map(
|
|
884
|
+
(failure) => JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric])
|
|
885
|
+
)
|
|
886
|
+
);
|
|
887
|
+
const represented = /* @__PURE__ */ new Set();
|
|
888
|
+
const cases = summary.behavioralMetrics.map((row) => {
|
|
889
|
+
const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
|
|
890
|
+
const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
|
|
891
|
+
represented.add(key);
|
|
892
|
+
const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
|
|
893
|
+
return failures.has(key) ? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>` : ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
|
|
894
|
+
});
|
|
895
|
+
for (const failure of summary.hardGate.failures) {
|
|
896
|
+
const key = JSON.stringify([
|
|
897
|
+
failure.scenarioId,
|
|
898
|
+
failure.role,
|
|
899
|
+
failure.target.id,
|
|
900
|
+
failure.target.model,
|
|
901
|
+
failure.metric
|
|
902
|
+
]);
|
|
903
|
+
if (represented.has(key)) continue;
|
|
904
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
|
|
905
|
+
cases.push(
|
|
906
|
+
` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`
|
|
907
|
+
);
|
|
908
|
+
}
|
|
909
|
+
for (const failure of summary.hardGate.envelopeFailures) {
|
|
910
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
|
|
911
|
+
cases.push(
|
|
912
|
+
` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`
|
|
913
|
+
);
|
|
914
|
+
}
|
|
915
|
+
return [
|
|
916
|
+
`<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
|
|
917
|
+
...cases,
|
|
918
|
+
"</testsuite>",
|
|
919
|
+
""
|
|
920
|
+
].join("\n");
|
|
921
|
+
}
|
|
922
|
+
function changeCounts2(counts) {
|
|
923
|
+
return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
|
|
924
|
+
}
|
|
925
|
+
function format(value) {
|
|
926
|
+
return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
|
|
927
|
+
}
|
|
928
|
+
function cell(value) {
|
|
929
|
+
return value.replaceAll("|", "\\|");
|
|
930
|
+
}
|
|
931
|
+
function escapeXml(value) {
|
|
932
|
+
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll('"', """);
|
|
933
|
+
}
|
|
187
934
|
|
|
188
935
|
// src/domains/eval/reports/json.ts
|
|
189
936
|
init_esm_shims();
|
|
@@ -195,21 +942,35 @@ function renderEvalJsonReportV4(artifact) {
|
|
|
195
942
|
// src/domains/eval/reports/junit.ts
|
|
196
943
|
init_esm_shims();
|
|
197
944
|
function renderEvalJunitReportV4(artifact) {
|
|
945
|
+
let failures = 0;
|
|
946
|
+
let skipped = 0;
|
|
198
947
|
const cases = artifact.results.map((result) => {
|
|
199
|
-
const name =
|
|
948
|
+
const name = escapeXml2(
|
|
200
949
|
`${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`
|
|
201
950
|
);
|
|
202
|
-
if (result.pass)
|
|
203
|
-
|
|
951
|
+
if (!result.pass) {
|
|
952
|
+
failures += 1;
|
|
953
|
+
return ` <testcase name="${name}"><failure message="${escapeXml2(result.failureClass ?? "failed")}" /></testcase>`;
|
|
954
|
+
}
|
|
955
|
+
const outcome = result.behavioral?.outcome;
|
|
956
|
+
if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
|
|
957
|
+
failures += 1;
|
|
958
|
+
return ` <testcase name="${name}"><failure message="${escapeXml2(outcome)}" /></testcase>`;
|
|
959
|
+
}
|
|
960
|
+
if (outcome === "unknown" || outcome === "unmeasured") {
|
|
961
|
+
skipped += 1;
|
|
962
|
+
return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml2(outcome)}" /></testcase>`;
|
|
963
|
+
}
|
|
964
|
+
return ` <testcase name="${name}" />`;
|
|
204
965
|
}).join("\n");
|
|
205
966
|
return [
|
|
206
|
-
`<testsuite name="${
|
|
967
|
+
`<testsuite name="${escapeXml2(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
|
|
207
968
|
cases,
|
|
208
969
|
"</testsuite>",
|
|
209
970
|
""
|
|
210
971
|
].join("\n");
|
|
211
972
|
}
|
|
212
|
-
function
|
|
973
|
+
function escapeXml2(value) {
|
|
213
974
|
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll('"', """);
|
|
214
975
|
}
|
|
215
976
|
|
|
@@ -223,10 +984,10 @@ function renderEvalMarkdownReportV4(artifact) {
|
|
|
223
984
|
`Target: ${artifact.matrix.target}`,
|
|
224
985
|
`Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
|
|
225
986
|
"",
|
|
226
|
-
"| Task | Target | Model | Repeat |
|
|
227
|
-
"
|
|
987
|
+
"| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
|
|
988
|
+
"|---|---|---|---|---:|---|---|---|",
|
|
228
989
|
...artifact.results.map(
|
|
229
|
-
(result) => `| ${result.taskId} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.failureClass ?? ""} |`
|
|
990
|
+
(result) => `| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`
|
|
230
991
|
),
|
|
231
992
|
""
|
|
232
993
|
];
|
|
@@ -251,6 +1012,12 @@ function renderEvalSweJsonlReportV4(artifact) {
|
|
|
251
1012
|
init_esm_shims();
|
|
252
1013
|
function renderEvalTextReportV4(artifact) {
|
|
253
1014
|
const tokens = artifact.summary.tokens;
|
|
1015
|
+
const behavioral = artifact.results.flatMap(
|
|
1016
|
+
(result) => result.behavioral === void 0 ? [] : [result.behavioral.outcome]
|
|
1017
|
+
);
|
|
1018
|
+
const behavioralSummary = behavioral.length === 0 ? [] : [
|
|
1019
|
+
`behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`
|
|
1020
|
+
];
|
|
254
1021
|
return [
|
|
255
1022
|
`eval: ${artifact.evalId}`,
|
|
256
1023
|
`suite: ${artifact.suite.id}`,
|
|
@@ -265,9 +1032,13 @@ function renderEvalTextReportV4(artifact) {
|
|
|
265
1032
|
// reported next to how many runs it actually covers.
|
|
266
1033
|
!tokens.measured ? `tokens total: unmeasured (0 of ${tokens.runs} runs reported usage)` : tokens.measuredRuns === tokens.runs ? `tokens total: ${tokens.total}` : `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
|
|
267
1034
|
`wall time ms: ${artifact.summary.wallTimeMs}`,
|
|
1035
|
+
...behavioralSummary,
|
|
268
1036
|
""
|
|
269
1037
|
].join("\n");
|
|
270
1038
|
}
|
|
1039
|
+
function count(values, wanted) {
|
|
1040
|
+
return values.filter((value) => value === wanted).length;
|
|
1041
|
+
}
|
|
271
1042
|
|
|
272
1043
|
// src/domains/eval/suites/load.ts
|
|
273
1044
|
init_esm_shims();
|
|
@@ -278,12 +1049,6 @@ import { dirname, resolve } from "node:path";
|
|
|
278
1049
|
|
|
279
1050
|
// src/domains/eval/schema/validate.ts
|
|
280
1051
|
init_esm_shims();
|
|
281
|
-
|
|
282
|
-
// src/domains/eval/schema/suite.ts
|
|
283
|
-
init_esm_shims();
|
|
284
|
-
var EVAL_SUITE_V2_VERSION = 2;
|
|
285
|
-
|
|
286
|
-
// src/domains/eval/schema/validate.ts
|
|
287
1052
|
var RUNNER_KINDS = /* @__PURE__ */ new Set(["clio-run", "context-index", "context-init", "external-command"]);
|
|
288
1053
|
var WORKSPACE_KINDS = /* @__PURE__ */ new Set(["local", "git", "temp-copy"]);
|
|
289
1054
|
var OPS = /* @__PURE__ */ new Set(["lt", "lte", "gt", "gte", "eq", "neq"]);
|
|
@@ -351,12 +1116,26 @@ function readMatrix(value, path, issues) {
|
|
|
351
1116
|
];
|
|
352
1117
|
});
|
|
353
1118
|
if (repeats === null || targets.length === 0) return null;
|
|
1119
|
+
let dimensions;
|
|
1120
|
+
if (value.dimensions !== void 0) {
|
|
1121
|
+
try {
|
|
1122
|
+
dimensions = parseEvalExecutionMatrixDimensionsV1(value.dimensions, `${path}.dimensions`);
|
|
1123
|
+
} catch (error) {
|
|
1124
|
+
issues.push({ path: `${path}.dimensions`, message: error instanceof Error ? error.message : String(error) });
|
|
1125
|
+
return null;
|
|
1126
|
+
}
|
|
1127
|
+
}
|
|
354
1128
|
const maxCostUsd = value.maxCostUsd;
|
|
355
1129
|
if (maxCostUsd !== void 0 && (typeof maxCostUsd !== "number" || !Number.isFinite(maxCostUsd) || maxCostUsd < 0)) {
|
|
356
1130
|
issues.push({ path: `${path}.maxCostUsd`, message: "expected non-negative number" });
|
|
357
1131
|
return null;
|
|
358
1132
|
}
|
|
359
|
-
return {
|
|
1133
|
+
return {
|
|
1134
|
+
targets,
|
|
1135
|
+
repeats,
|
|
1136
|
+
...dimensions === void 0 ? {} : { dimensions },
|
|
1137
|
+
...maxCostUsd === void 0 ? {} : { maxCostUsd }
|
|
1138
|
+
};
|
|
360
1139
|
}
|
|
361
1140
|
function readTasks(value, path, issues) {
|
|
362
1141
|
if (!Array.isArray(value) || value.length === 0) {
|
|
@@ -383,10 +1162,12 @@ function readTask(value, path, issues) {
|
|
|
383
1162
|
const workspace = readWorkspace(value.workspace, `${path}.workspace`, issues);
|
|
384
1163
|
const runner = readRunner(value.runner, `${path}.runner`, issues);
|
|
385
1164
|
const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
|
|
1165
|
+
const behavioral = readBehavioral(value.behavioral, `${path}.behavioral`, issues);
|
|
386
1166
|
if (id === null || workspace === null || runner === null || timeoutMs === null) return null;
|
|
387
1167
|
return {
|
|
388
1168
|
id,
|
|
389
1169
|
tags: readOptionalStringArray(value, "tags", `${path}.tags`, issues),
|
|
1170
|
+
...behavioral === void 0 ? {} : { behavioral },
|
|
390
1171
|
workspace,
|
|
391
1172
|
runner,
|
|
392
1173
|
verify: readVerify(value.verify, `${path}.verify`, issues),
|
|
@@ -394,6 +1175,15 @@ function readTask(value, path, issues) {
|
|
|
394
1175
|
timeoutMs
|
|
395
1176
|
};
|
|
396
1177
|
}
|
|
1178
|
+
function readBehavioral(value, path, issues) {
|
|
1179
|
+
if (value === void 0) return void 0;
|
|
1180
|
+
try {
|
|
1181
|
+
return parseEvalBehaviorScenarioV1(value, path);
|
|
1182
|
+
} catch (error) {
|
|
1183
|
+
issues.push({ path, message: error instanceof Error ? error.message : String(error) });
|
|
1184
|
+
return void 0;
|
|
1185
|
+
}
|
|
1186
|
+
}
|
|
397
1187
|
function readWorkspace(value, path, issues) {
|
|
398
1188
|
if (!isRecord2(value)) {
|
|
399
1189
|
issues.push({ path, message: "expected object" });
|
|
@@ -430,6 +1220,10 @@ function readRunner(value, path, issues) {
|
|
|
430
1220
|
}
|
|
431
1221
|
const prompt = optionalString(value, "prompt");
|
|
432
1222
|
const agent = optionalString(value, "agent");
|
|
1223
|
+
const autonomy = optionalString(value, "autonomy");
|
|
1224
|
+
if (autonomy !== void 0 && !["read-only", "suggest", "auto-edit", "full-auto"].includes(autonomy)) {
|
|
1225
|
+
issues.push({ path: `${path}.autonomy`, message: "expected read-only, suggest, auto-edit, or full-auto" });
|
|
1226
|
+
}
|
|
433
1227
|
if (agent !== void 0 && kind !== "clio-run") {
|
|
434
1228
|
issues.push({ path: `${path}.agent`, message: "agent is only valid on the clio-run runner" });
|
|
435
1229
|
}
|
|
@@ -437,6 +1231,7 @@ function readRunner(value, path, issues) {
|
|
|
437
1231
|
return {
|
|
438
1232
|
kind,
|
|
439
1233
|
...prompt === void 0 ? {} : { prompt },
|
|
1234
|
+
...autonomy === void 0 ? {} : { autonomy },
|
|
440
1235
|
...agent === void 0 ? {} : { agent },
|
|
441
1236
|
...command === void 0 ? {} : { command },
|
|
442
1237
|
commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
|
|
@@ -463,7 +1258,22 @@ function readMetrics(value, path, issues) {
|
|
|
463
1258
|
issues.push({ path, message: "expected object" });
|
|
464
1259
|
return { collect: [] };
|
|
465
1260
|
}
|
|
466
|
-
|
|
1261
|
+
const observation = value.readObservation;
|
|
1262
|
+
let readObservation;
|
|
1263
|
+
if (observation !== void 0) {
|
|
1264
|
+
if (!isRecord2(observation)) {
|
|
1265
|
+
issues.push({ path: `${path}.readObservation`, message: "expected object" });
|
|
1266
|
+
} else {
|
|
1267
|
+
readObservation = {
|
|
1268
|
+
allowedPaths: readOptionalStringArray(observation, "allowedPaths", `${path}.readObservation.allowedPaths`, issues),
|
|
1269
|
+
decoyPaths: readOptionalStringArray(observation, "decoyPaths", `${path}.readObservation.decoyPaths`, issues)
|
|
1270
|
+
};
|
|
1271
|
+
}
|
|
1272
|
+
}
|
|
1273
|
+
return {
|
|
1274
|
+
collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues),
|
|
1275
|
+
...readObservation === void 0 ? {} : { readObservation }
|
|
1276
|
+
};
|
|
467
1277
|
}
|
|
468
1278
|
function readThresholds(value, path, issues) {
|
|
469
1279
|
if (value === void 0) return void 0;
|
|
@@ -471,7 +1281,10 @@ function readThresholds(value, path, issues) {
|
|
|
471
1281
|
issues.push({ path, message: "expected object" });
|
|
472
1282
|
return void 0;
|
|
473
1283
|
}
|
|
474
|
-
return {
|
|
1284
|
+
return {
|
|
1285
|
+
fail: readAssertions(value.fail, `${path}.fail`, issues),
|
|
1286
|
+
informational: readAssertions(value.informational, `${path}.informational`, issues)
|
|
1287
|
+
};
|
|
475
1288
|
}
|
|
476
1289
|
function readAssertions(value, path, issues) {
|
|
477
1290
|
if (value === void 0) return [];
|
|
@@ -620,7 +1433,8 @@ function resolveSuiteForRun(suite, options) {
|
|
|
620
1433
|
...suite,
|
|
621
1434
|
matrix: {
|
|
622
1435
|
...suite.matrix,
|
|
623
|
-
targets
|
|
1436
|
+
targets,
|
|
1437
|
+
...options.trials === void 0 ? {} : { repeats: options.trials }
|
|
624
1438
|
}
|
|
625
1439
|
};
|
|
626
1440
|
}
|
|
@@ -646,9 +1460,166 @@ function resolveTargets(targets, options) {
|
|
|
646
1460
|
|
|
647
1461
|
// src/domains/eval/suites/run.ts
|
|
648
1462
|
init_esm_shims();
|
|
649
|
-
import { mkdtemp as mkdtemp3, rm as rm3 } from "node:fs/promises";
|
|
1463
|
+
import { mkdtemp as mkdtemp3, rm as rm3, writeFile } from "node:fs/promises";
|
|
650
1464
|
import { tmpdir as tmpdir3 } from "node:os";
|
|
651
|
-
import { resolve as
|
|
1465
|
+
import { resolve as resolve7 } from "node:path";
|
|
1466
|
+
|
|
1467
|
+
// src/domains/eval/execution-provenance.ts
|
|
1468
|
+
init_esm_shims();
|
|
1469
|
+
import { createHash as createHash2 } from "node:crypto";
|
|
1470
|
+
function buildEvalExecutionEnvelopeV1(input) {
|
|
1471
|
+
const scenario = input.task.behavioral;
|
|
1472
|
+
if (scenario === void 0) throw new Error(`behavioral task ${input.task.id} has no behavioral scenario`);
|
|
1473
|
+
const manifest = input.ledger.promptManifests.at(-1) ?? null;
|
|
1474
|
+
const contextSnapshot = input.ledger.contextSnapshots.at(-1) ?? null;
|
|
1475
|
+
const recipe = recipeIdentity(
|
|
1476
|
+
input,
|
|
1477
|
+
scenario.execution.subject.kind === "worker" ? scenario.execution.subject.role : null
|
|
1478
|
+
);
|
|
1479
|
+
const policy = policyIdentity(input.cwd, input.receipt, input.observation);
|
|
1480
|
+
const autonomy = input.receipt?.autonomyEnforcement?.autonomy ?? input.observation?.autonomy ?? input.task.runner.autonomy ?? null;
|
|
1481
|
+
const promptFragments = promptFragmentIdentities(manifest, recipe, autonomy);
|
|
1482
|
+
const compositionHash = input.receipt?.staticCompositionHash ?? input.observation?.compositionHash ?? manifest?.systemPromptHash ?? contextSnapshot?.promptHash ?? null;
|
|
1483
|
+
const projectContext = projectContextIdentity(input, manifest, promptFragments);
|
|
1484
|
+
return {
|
|
1485
|
+
schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
1486
|
+
prompt: { fragments: promptFragments, compositionHash },
|
|
1487
|
+
recipe: recipe === null ? null : { id: recipe.id, version: recipe.version, contentHash: recipe.contentHash },
|
|
1488
|
+
target: input.receipt?.targetId ?? input.observation?.target ?? input.target.id,
|
|
1489
|
+
wireModel: input.receipt?.wireModelId ?? input.observation?.wireModel ?? contextSnapshot?.modelId ?? input.target.model ?? null,
|
|
1490
|
+
runtime: input.receipt?.runtimeId ?? input.observation?.runtime ?? contextSnapshot?.runtimeId ?? null,
|
|
1491
|
+
thinkingLevel: input.receipt?.runtimeResolution?.effectiveThinkingLevel ?? input.observation?.thinkingLevel ?? manifest?.thinkingLevel ?? input.target.thinking ?? null,
|
|
1492
|
+
toolSignature: input.receipt?.toolSignature ?? input.observation?.toolSignature ?? contextSnapshot?.toolSignature ?? null,
|
|
1493
|
+
autonomy,
|
|
1494
|
+
policyHashes: policy,
|
|
1495
|
+
projectContext,
|
|
1496
|
+
corpus: { ...scenario.corpus }
|
|
1497
|
+
};
|
|
1498
|
+
}
|
|
1499
|
+
function recipeIdentity(input, role) {
|
|
1500
|
+
const id = input.receipt?.agentId ?? input.task.runner.agent ?? role;
|
|
1501
|
+
if (id === null || input.cwd === null) return null;
|
|
1502
|
+
try {
|
|
1503
|
+
const recipe = discoverAgentRecipes(input.cwd).find((entry) => entry.id === id);
|
|
1504
|
+
if (recipe === void 0) return null;
|
|
1505
|
+
return {
|
|
1506
|
+
id: recipe.id,
|
|
1507
|
+
version: recipe.version,
|
|
1508
|
+
contentHash: agentSpecFingerprint(normalizeAgentSpec(recipe)),
|
|
1509
|
+
personaHash: sha256(recipe.body)
|
|
1510
|
+
};
|
|
1511
|
+
} catch {
|
|
1512
|
+
return null;
|
|
1513
|
+
}
|
|
1514
|
+
}
|
|
1515
|
+
function promptFragmentIdentities(manifest, recipe, autonomy) {
|
|
1516
|
+
let versions = /* @__PURE__ */ new Map();
|
|
1517
|
+
try {
|
|
1518
|
+
versions = new Map([...loadFragments().byId.values()].map((fragment) => [fragment.id, fragment.version]));
|
|
1519
|
+
} catch {
|
|
1520
|
+
}
|
|
1521
|
+
if (manifest !== null) {
|
|
1522
|
+
return manifest.fragments.map((fragment) => ({
|
|
1523
|
+
id: fragment.id,
|
|
1524
|
+
version: versions.get(fragment.id) ?? "unversioned",
|
|
1525
|
+
contentHash: fragment.contentHash
|
|
1526
|
+
})).sort((left, right) => left.id.localeCompare(right.id));
|
|
1527
|
+
}
|
|
1528
|
+
if (recipe === null) return [];
|
|
1529
|
+
const selected = ["identity.clio-worker", "operating.contract", "operating.worker"];
|
|
1530
|
+
if (autonomy !== null) selected.push(`safety.${autonomy}`);
|
|
1531
|
+
const fragments = [];
|
|
1532
|
+
try {
|
|
1533
|
+
const table = loadFragments();
|
|
1534
|
+
for (const id of selected) {
|
|
1535
|
+
const fragment = table.byId.get(id);
|
|
1536
|
+
if (fragment !== void 0) {
|
|
1537
|
+
fragments.push({ id, version: fragment.version, contentHash: fragment.contentHash });
|
|
1538
|
+
}
|
|
1539
|
+
}
|
|
1540
|
+
} catch {
|
|
1541
|
+
}
|
|
1542
|
+
fragments.push({ id: `persona.${recipe.id}`, version: recipe.version, contentHash: recipe.personaHash });
|
|
1543
|
+
return fragments.sort((left, right) => left.id.localeCompare(right.id));
|
|
1544
|
+
}
|
|
1545
|
+
function policyIdentity(cwd, receipt, observation) {
|
|
1546
|
+
const sealed = receipt?.reproducibility?.safetyPolicy;
|
|
1547
|
+
if (sealed !== void 0) return { rulePack: sealed.rulePackHash, project: sealed.projectPolicyHash };
|
|
1548
|
+
if (observation !== void 0) return { ...observation.policyHashes };
|
|
1549
|
+
if (cwd === null) return { rulePack: null, project: null };
|
|
1550
|
+
try {
|
|
1551
|
+
const metadata = createSafetyPolicyEngine({ cwd }).metadata();
|
|
1552
|
+
return { rulePack: metadata.rulePackHash, project: metadata.projectPolicyHash };
|
|
1553
|
+
} catch {
|
|
1554
|
+
return { rulePack: null, project: null };
|
|
1555
|
+
}
|
|
1556
|
+
}
|
|
1557
|
+
function projectContextIdentity(input, manifest, fragments) {
|
|
1558
|
+
const receipt = input.receipt;
|
|
1559
|
+
if (receipt?.projectContext !== void 0) {
|
|
1560
|
+
const sections = [...receipt.projectContext.sections ?? []].sort();
|
|
1561
|
+
const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
|
|
1562
|
+
return {
|
|
1563
|
+
kind: "worker",
|
|
1564
|
+
tier: receipt.projectContext.tier,
|
|
1565
|
+
contentHash: hasContentBearingContext ? receipt.projectContext.contentHash ?? null : null,
|
|
1566
|
+
chars: hasContentBearingContext ? receipt.projectContext.chars ?? null : null,
|
|
1567
|
+
sections,
|
|
1568
|
+
rulesApplied: [...receipt.rulesApplied ?? []].sort(),
|
|
1569
|
+
operatorProfileApplied: receipt.operatorProfileApplied ?? null
|
|
1570
|
+
};
|
|
1571
|
+
}
|
|
1572
|
+
const observed = input.observation?.projectContext;
|
|
1573
|
+
if (observed !== void 0 && observed !== null) {
|
|
1574
|
+
const sections = [...observed.sections].sort();
|
|
1575
|
+
const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
|
|
1576
|
+
return {
|
|
1577
|
+
kind: "worker",
|
|
1578
|
+
tier: observed.tier,
|
|
1579
|
+
contentHash: hasContentBearingContext ? observed.contentHash : null,
|
|
1580
|
+
chars: hasContentBearingContext ? observed.chars : null,
|
|
1581
|
+
sections,
|
|
1582
|
+
rulesApplied: [...observed.rulesApplied].sort(),
|
|
1583
|
+
operatorProfileApplied: observed.operatorProfileApplied
|
|
1584
|
+
};
|
|
1585
|
+
}
|
|
1586
|
+
if (manifest !== null) {
|
|
1587
|
+
const contextFragments = fragments.filter((fragment) => fragment.id.startsWith("context."));
|
|
1588
|
+
const preload = manifest.projectPreload;
|
|
1589
|
+
const identity = {
|
|
1590
|
+
preload,
|
|
1591
|
+
fragments: contextFragments.map((fragment) => [fragment.id, fragment.contentHash])
|
|
1592
|
+
};
|
|
1593
|
+
return {
|
|
1594
|
+
kind: "session",
|
|
1595
|
+
tier: preload?.mode ?? null,
|
|
1596
|
+
contentHash: sha256(stableJson2(identity)),
|
|
1597
|
+
chars: preload?.chars ?? null,
|
|
1598
|
+
sections: contextFragments.map((fragment) => fragment.id).sort(),
|
|
1599
|
+
rulesApplied: [],
|
|
1600
|
+
operatorProfileApplied: contextFragments.some((fragment) => fragment.id === "context.operator-profile")
|
|
1601
|
+
};
|
|
1602
|
+
}
|
|
1603
|
+
return {
|
|
1604
|
+
kind: "none",
|
|
1605
|
+
tier: null,
|
|
1606
|
+
contentHash: null,
|
|
1607
|
+
chars: null,
|
|
1608
|
+
sections: [],
|
|
1609
|
+
rulesApplied: [],
|
|
1610
|
+
operatorProfileApplied: null
|
|
1611
|
+
};
|
|
1612
|
+
}
|
|
1613
|
+
function stableJson2(value) {
|
|
1614
|
+
if (Array.isArray(value)) return `[${value.map(stableJson2).join(",")}]`;
|
|
1615
|
+
if (typeof value === "object" && value !== null) {
|
|
1616
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson2(entry)}`).join(",")}}`;
|
|
1617
|
+
}
|
|
1618
|
+
return JSON.stringify(value);
|
|
1619
|
+
}
|
|
1620
|
+
function sha256(value) {
|
|
1621
|
+
return createHash2("sha256").update(value, "utf8").digest("hex");
|
|
1622
|
+
}
|
|
652
1623
|
|
|
653
1624
|
// src/domains/eval/metrics/context.ts
|
|
654
1625
|
init_esm_shims();
|
|
@@ -1295,19 +2266,317 @@ function zeroToolCallMetrics() {
|
|
|
1295
2266
|
};
|
|
1296
2267
|
}
|
|
1297
2268
|
|
|
2269
|
+
// src/domains/eval/metrics/tracked.ts
|
|
2270
|
+
init_esm_shims();
|
|
2271
|
+
import { readFile as readFile2 } from "node:fs/promises";
|
|
2272
|
+
import { dirname as dirname2, join as join2 } from "node:path";
|
|
2273
|
+
async function readEvalLedgerSnapshot(stateDir) {
|
|
2274
|
+
const refs = await listSessionLedgerRefs(stateDir);
|
|
2275
|
+
const entries = [];
|
|
2276
|
+
const compiledPromptHashes = [];
|
|
2277
|
+
const promptManifests = [];
|
|
2278
|
+
const contextSnapshots = [];
|
|
2279
|
+
for (const ref of refs) {
|
|
2280
|
+
try {
|
|
2281
|
+
const raw = await readFile2(ref.path, "utf8");
|
|
2282
|
+
entries.push(...parseSessionEntries(raw, ref.path).entries);
|
|
2283
|
+
} catch {
|
|
2284
|
+
}
|
|
2285
|
+
try {
|
|
2286
|
+
const manifest = await readFile2(join2(dirname2(ref.path), "prompt-manifest.jsonl"), "utf8");
|
|
2287
|
+
for (const line of manifest.split(/\r?\n/u)) {
|
|
2288
|
+
const record = parseJsonRecord2(line);
|
|
2289
|
+
if (record === null) continue;
|
|
2290
|
+
const hash = record?.systemPromptHash;
|
|
2291
|
+
if (typeof hash !== "string" || !/^[a-f0-9]{64}$/u.test(hash)) continue;
|
|
2292
|
+
compiledPromptHashes.push(hash);
|
|
2293
|
+
const observation = promptManifestObservation(record);
|
|
2294
|
+
if (observation !== null) promptManifests.push(observation);
|
|
2295
|
+
}
|
|
2296
|
+
} catch {
|
|
2297
|
+
}
|
|
2298
|
+
try {
|
|
2299
|
+
const snapshots = await readFile2(join2(dirname2(ref.path), "context-snapshots.jsonl"), "utf8");
|
|
2300
|
+
for (const line of snapshots.split(/\r?\n/u)) {
|
|
2301
|
+
const record = parseJsonRecord2(line);
|
|
2302
|
+
if (record === null) continue;
|
|
2303
|
+
contextSnapshots.push({
|
|
2304
|
+
runtimeId: nullableString(record.runtimeId),
|
|
2305
|
+
modelId: nullableString(record.modelId),
|
|
2306
|
+
promptHash: nullableDigest(record.promptHash),
|
|
2307
|
+
toolSignature: nullableDigest(record.toolSignature)
|
|
2308
|
+
});
|
|
2309
|
+
}
|
|
2310
|
+
} catch {
|
|
2311
|
+
}
|
|
2312
|
+
}
|
|
2313
|
+
return { entries, compiledPromptHashes: [...new Set(compiledPromptHashes)], promptManifests, contextSnapshots };
|
|
2314
|
+
}
|
|
2315
|
+
function buildEvalTrackedMetrics(input) {
|
|
2316
|
+
const calls = assistantCalls(input.ledgerEntries);
|
|
2317
|
+
const compactionEntries = input.ledgerEntries.filter((entry) => entry.kind === "compactionSummary");
|
|
2318
|
+
const compactionUsage = compactionEntries.flatMap((entry) => {
|
|
2319
|
+
if (entry.kind !== "compactionSummary" || !isRecord5(entry.usage)) return [];
|
|
2320
|
+
return [entry.usage];
|
|
2321
|
+
});
|
|
2322
|
+
const modelCallReadings = calls.map(() => ledgerReading(1));
|
|
2323
|
+
for (const entry of compactionEntries) {
|
|
2324
|
+
if (entry.kind !== "compactionSummary") continue;
|
|
2325
|
+
const apiCalls = isRecord5(entry.usage) ? nonNegativeNumber(entry.usage.apiCalls) : null;
|
|
2326
|
+
modelCallReadings.push(apiCalls === null ? estimatedReading(1) : ledgerReading(apiCalls));
|
|
2327
|
+
}
|
|
2328
|
+
const uncachedReadings = calls.map(uncachedPrefillForCall);
|
|
2329
|
+
const cacheReadings = calls.map(cacheReadForCall);
|
|
2330
|
+
const generatedReadings = calls.map(generatedForCall);
|
|
2331
|
+
for (const usage of compactionUsage) {
|
|
2332
|
+
uncachedReadings.push(readingFromUsage(usage, "input"));
|
|
2333
|
+
cacheReadings.push(readingFromUsage(usage, "cacheRead"));
|
|
2334
|
+
generatedReadings.push(readingFromUsage(usage, "output"));
|
|
2335
|
+
}
|
|
2336
|
+
const reasoning = reasoningMetric(input.receipt, calls, compactionUsage);
|
|
2337
|
+
const receiptToolMetrics = input.receipt === null ? null : evalHarnessMetricsFromReceipt(input.receipt);
|
|
2338
|
+
const ledgerToolCalls = input.ledgerEntries.filter(
|
|
2339
|
+
(entry) => entry.kind === "message" && entry.role === "tool_call"
|
|
2340
|
+
).length;
|
|
2341
|
+
const ledgerToolErrors = input.ledgerEntries.filter((entry) => {
|
|
2342
|
+
if (entry.kind !== "message" || entry.role !== "tool_result" || !isRecord5(entry.payload)) return false;
|
|
2343
|
+
return entry.payload.isError === true || entry.payload.outcome === "error";
|
|
2344
|
+
}).length;
|
|
2345
|
+
const receiptToolErrors = input.receipt?.toolStats.reduce((sum2, stat) => sum2 + finiteNonNegative(stat.errors), 0);
|
|
2346
|
+
const expectedColdReasons = expectedColdReasonMetrics(calls);
|
|
2347
|
+
return {
|
|
2348
|
+
modelCalls: sumReadings(modelCallReadings, "ledger"),
|
|
2349
|
+
uncachedPrefillTokens: sumReadings(uncachedReadings, "estimated"),
|
|
2350
|
+
cacheReadTokens: sumReadings(cacheReadings, "estimated"),
|
|
2351
|
+
generatedTokens: sumReadings(generatedReadings, "estimated"),
|
|
2352
|
+
reasoningTokens: reasoning,
|
|
2353
|
+
toolCalls: receiptToolMetrics === null ? { value: ledgerToolCalls, source: "ledger" } : { value: receiptToolMetrics.toolCalls, source: "receipt" },
|
|
2354
|
+
toolErrors: receiptToolErrors === void 0 ? { value: ledgerToolErrors, source: "ledger" } : { value: receiptToolErrors, source: "receipt" },
|
|
2355
|
+
ttftMsFirstCall: firstCallTtft(calls),
|
|
2356
|
+
wallClockMs: wallClockMetric(input.receipt, input.fallbackWallClockMs),
|
|
2357
|
+
contextTokensAtEnd: contextTokensAtEnd(calls, compactionEntries),
|
|
2358
|
+
compactions: { value: compactionEntries.length, source: "ledger" },
|
|
2359
|
+
expectedColdReasons
|
|
2360
|
+
};
|
|
2361
|
+
}
|
|
2362
|
+
function emptyEvalTrackedMetrics(source = "estimated") {
|
|
2363
|
+
const zero = () => ({ value: 0, source });
|
|
2364
|
+
return {
|
|
2365
|
+
modelCalls: zero(),
|
|
2366
|
+
uncachedPrefillTokens: zero(),
|
|
2367
|
+
cacheReadTokens: zero(),
|
|
2368
|
+
generatedTokens: zero(),
|
|
2369
|
+
reasoningTokens: { value: null, source },
|
|
2370
|
+
toolCalls: zero(),
|
|
2371
|
+
toolErrors: zero(),
|
|
2372
|
+
ttftMsFirstCall: zero(),
|
|
2373
|
+
wallClockMs: zero(),
|
|
2374
|
+
contextTokensAtEnd: zero(),
|
|
2375
|
+
compactions: zero(),
|
|
2376
|
+
expectedColdReasons: {}
|
|
2377
|
+
};
|
|
2378
|
+
}
|
|
2379
|
+
function assistantCalls(entries) {
|
|
2380
|
+
return entries.flatMap((entry) => {
|
|
2381
|
+
if (entry.kind !== "message" || entry.role !== "assistant" || !isRecord5(entry.payload)) return [];
|
|
2382
|
+
const promptCache = recordField(entry.payload, "promptCache");
|
|
2383
|
+
const timing = recordField(entry.payload, "timing");
|
|
2384
|
+
const usage = recordField(entry.payload, "usage");
|
|
2385
|
+
if (promptCache === null && timing === null && usage === null) return [];
|
|
2386
|
+
return [
|
|
2387
|
+
{
|
|
2388
|
+
payload: entry.payload,
|
|
2389
|
+
promptCache,
|
|
2390
|
+
backend: promptCache === null ? null : recordField(promptCache, "backend"),
|
|
2391
|
+
timing,
|
|
2392
|
+
usage
|
|
2393
|
+
}
|
|
2394
|
+
];
|
|
2395
|
+
});
|
|
2396
|
+
}
|
|
2397
|
+
function uncachedPrefillForCall(call) {
|
|
2398
|
+
const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
|
|
2399
|
+
const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
|
|
2400
|
+
if (promptTokens !== null && cachedTokens !== null && cachedTokens <= promptTokens) {
|
|
2401
|
+
return ledgerReading(promptTokens - cachedTokens);
|
|
2402
|
+
}
|
|
2403
|
+
const piInput = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
|
|
2404
|
+
if (piInput !== null) return ledgerReading(piInput);
|
|
2405
|
+
const legacyInput = call.usage === null ? null : nonNegativeNumber(call.usage.input);
|
|
2406
|
+
return legacyInput === null ? estimatedReading(0) : estimatedReading(legacyInput);
|
|
2407
|
+
}
|
|
2408
|
+
function cacheReadForCall(call) {
|
|
2409
|
+
const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
|
|
2410
|
+
if (cachedTokens !== null) return ledgerReading(cachedTokens);
|
|
2411
|
+
const piCacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
|
|
2412
|
+
if (piCacheRead !== null) return ledgerReading(piCacheRead);
|
|
2413
|
+
const legacyCacheRead = call.usage === null ? null : nonNegativeNumber(call.usage.cacheRead);
|
|
2414
|
+
return legacyCacheRead === null ? estimatedReading(0) : estimatedReading(legacyCacheRead);
|
|
2415
|
+
}
|
|
2416
|
+
function generatedForCall(call) {
|
|
2417
|
+
const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
|
|
2418
|
+
if (predictedTokens !== null) return ledgerReading(predictedTokens);
|
|
2419
|
+
const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
|
|
2420
|
+
return output === null ? estimatedReading(0) : ledgerReading(output);
|
|
2421
|
+
}
|
|
2422
|
+
function readingFromUsage(usage, field) {
|
|
2423
|
+
const value = nonNegativeNumber(usage[field]);
|
|
2424
|
+
return value === null ? estimatedReading(0) : ledgerReading(value);
|
|
2425
|
+
}
|
|
2426
|
+
function reasoningMetric(receipt, calls, compactionUsage) {
|
|
2427
|
+
if (receipt !== null && typeof receipt.reasoningTokenCount === "number") {
|
|
2428
|
+
return { value: finiteNonNegative(receipt.reasoningTokenCount), source: "receipt" };
|
|
2429
|
+
}
|
|
2430
|
+
let total = 0;
|
|
2431
|
+
let measured = false;
|
|
2432
|
+
for (const call of calls) {
|
|
2433
|
+
const value = extractReasoningTokens(call.usage);
|
|
2434
|
+
if (value === null) continue;
|
|
2435
|
+
measured = true;
|
|
2436
|
+
total += finiteNonNegative(value);
|
|
2437
|
+
}
|
|
2438
|
+
for (const usage of compactionUsage) {
|
|
2439
|
+
const value = nonNegativeNumber(usage.reasoning);
|
|
2440
|
+
if (value === null) continue;
|
|
2441
|
+
measured = true;
|
|
2442
|
+
total += value;
|
|
2443
|
+
}
|
|
2444
|
+
return measured ? { value: total, source: "ledger" } : { value: null, source: "estimated" };
|
|
2445
|
+
}
|
|
2446
|
+
function firstCallTtft(calls) {
|
|
2447
|
+
const first = calls[0];
|
|
2448
|
+
const value = first?.timing === null || first?.timing === void 0 ? null : nonNegativeNumber(first.timing.ttftMs);
|
|
2449
|
+
return value === null ? { value: 0, source: "estimated" } : { value, source: "ledger" };
|
|
2450
|
+
}
|
|
2451
|
+
function wallClockMetric(receipt, fallback) {
|
|
2452
|
+
if (receipt !== null) {
|
|
2453
|
+
const started = Date.parse(receipt.startedAt);
|
|
2454
|
+
const ended = Date.parse(receipt.endedAt);
|
|
2455
|
+
if (Number.isFinite(started) && Number.isFinite(ended) && ended >= started) {
|
|
2456
|
+
return { value: ended - started, source: "receipt" };
|
|
2457
|
+
}
|
|
2458
|
+
}
|
|
2459
|
+
return { value: finiteNonNegative(fallback), source: "estimated" };
|
|
2460
|
+
}
|
|
2461
|
+
function contextTokensAtEnd(calls, compactions) {
|
|
2462
|
+
const lastCall = calls.at(-1);
|
|
2463
|
+
if (lastCall !== void 0) {
|
|
2464
|
+
const lastReading = contextForCall(lastCall);
|
|
2465
|
+
if (lastReading !== null && lastReading > 0) return { value: lastReading, source: "ledger" };
|
|
2466
|
+
if (lastReading === 0) {
|
|
2467
|
+
for (const call of [...calls.slice(0, -1)].reverse()) {
|
|
2468
|
+
const reading = contextForCall(call);
|
|
2469
|
+
if (reading !== null && reading > 0) return { value: reading, source: "ledger" };
|
|
2470
|
+
}
|
|
2471
|
+
}
|
|
2472
|
+
}
|
|
2473
|
+
const lastCompaction = compactions.at(-1);
|
|
2474
|
+
if (lastCompaction?.kind === "compactionSummary") {
|
|
2475
|
+
const tokensAfter = nonNegativeNumber(lastCompaction.tokensAfter);
|
|
2476
|
+
if (tokensAfter !== null) return { value: tokensAfter, source: "ledger" };
|
|
2477
|
+
}
|
|
2478
|
+
return { value: 0, source: "estimated" };
|
|
2479
|
+
}
|
|
2480
|
+
function contextForCall(call) {
|
|
2481
|
+
const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
|
|
2482
|
+
const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
|
|
2483
|
+
if (promptTokens !== null && predictedTokens !== null) return promptTokens + predictedTokens;
|
|
2484
|
+
const input = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
|
|
2485
|
+
const cacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
|
|
2486
|
+
const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
|
|
2487
|
+
return input === null || cacheRead === null || output === null ? null : input + cacheRead + output;
|
|
2488
|
+
}
|
|
2489
|
+
function expectedColdReasonMetrics(calls) {
|
|
2490
|
+
const counts = /* @__PURE__ */ new Map();
|
|
2491
|
+
for (const call of calls) {
|
|
2492
|
+
const reasons = call.promptCache?.expectedColdReasons;
|
|
2493
|
+
if (!Array.isArray(reasons)) continue;
|
|
2494
|
+
const unique = new Set(reasons.filter((reason) => typeof reason === "string" && reason.length > 0));
|
|
2495
|
+
for (const reason of unique) counts.set(reason, (counts.get(reason) ?? 0) + 1);
|
|
2496
|
+
}
|
|
2497
|
+
return Object.fromEntries(
|
|
2498
|
+
[...counts.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([reason, value]) => [reason, { value, source: "ledger" }])
|
|
2499
|
+
);
|
|
2500
|
+
}
|
|
2501
|
+
function sumReadings(readings, emptySource) {
|
|
2502
|
+
if (readings.length === 0) return { value: 0, source: emptySource };
|
|
2503
|
+
return {
|
|
2504
|
+
value: readings.reduce((sum2, reading) => sum2 + reading.value, 0),
|
|
2505
|
+
source: readings.some((reading) => reading.source === "estimated") ? "estimated" : "ledger"
|
|
2506
|
+
};
|
|
2507
|
+
}
|
|
2508
|
+
function ledgerReading(value) {
|
|
2509
|
+
return { value: finiteNonNegative(value), source: "ledger" };
|
|
2510
|
+
}
|
|
2511
|
+
function estimatedReading(value) {
|
|
2512
|
+
return { value: finiteNonNegative(value), source: "estimated" };
|
|
2513
|
+
}
|
|
2514
|
+
function finiteNonNegative(value) {
|
|
2515
|
+
return Number.isFinite(value) && value >= 0 ? value : 0;
|
|
2516
|
+
}
|
|
2517
|
+
function nonNegativeNumber(value) {
|
|
2518
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
|
|
2519
|
+
}
|
|
2520
|
+
function recordField(record, field) {
|
|
2521
|
+
return isRecord5(record[field]) ? record[field] : null;
|
|
2522
|
+
}
|
|
2523
|
+
function parseJsonRecord2(line) {
|
|
2524
|
+
if (line.trim().length === 0) return null;
|
|
2525
|
+
try {
|
|
2526
|
+
const parsed = JSON.parse(line);
|
|
2527
|
+
return isRecord5(parsed) ? parsed : null;
|
|
2528
|
+
} catch {
|
|
2529
|
+
return null;
|
|
2530
|
+
}
|
|
2531
|
+
}
|
|
2532
|
+
function promptManifestObservation(record) {
|
|
2533
|
+
const systemPromptHash = nullableDigest(record.systemPromptHash);
|
|
2534
|
+
if (systemPromptHash === null || !Array.isArray(record.fragments)) return null;
|
|
2535
|
+
const fragments = record.fragments.flatMap((entry) => {
|
|
2536
|
+
if (!isRecord5(entry) || typeof entry.id !== "string") return [];
|
|
2537
|
+
const contentHash = nullableDigest(entry.contentHash);
|
|
2538
|
+
return contentHash === null ? [] : [{ id: entry.id, contentHash }];
|
|
2539
|
+
});
|
|
2540
|
+
const preload = record.projectPreload;
|
|
2541
|
+
const projectPreload = preload === null ? null : isRecord5(preload) && (preload.mode === "full" || preload.mode === "synopsis" || preload.mode === "none") && typeof preload.chars === "number" && Number.isInteger(preload.chars) && typeof preload.lines === "number" && Number.isInteger(preload.lines) && typeof preload.nearLimit === "boolean" && typeof preload.label === "string" ? {
|
|
2542
|
+
mode: preload.mode,
|
|
2543
|
+
chars: preload.chars,
|
|
2544
|
+
lines: preload.lines,
|
|
2545
|
+
reason: nullableString(preload.reason),
|
|
2546
|
+
nearLimit: preload.nearLimit,
|
|
2547
|
+
label: preload.label
|
|
2548
|
+
} : null;
|
|
2549
|
+
return {
|
|
2550
|
+
systemPromptHash,
|
|
2551
|
+
thinkingLevel: nullableString(record.thinkingLevel),
|
|
2552
|
+
projectPreload,
|
|
2553
|
+
fragments
|
|
2554
|
+
};
|
|
2555
|
+
}
|
|
2556
|
+
function nullableString(value) {
|
|
2557
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
2558
|
+
}
|
|
2559
|
+
function nullableDigest(value) {
|
|
2560
|
+
return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
|
|
2561
|
+
}
|
|
2562
|
+
function isRecord5(value) {
|
|
2563
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2564
|
+
}
|
|
2565
|
+
|
|
1298
2566
|
// src/domains/eval/runners/clio-run.ts
|
|
1299
2567
|
init_esm_shims();
|
|
2568
|
+
import { isAbsolute, relative, resolve as resolve2, sep } from "node:path";
|
|
1300
2569
|
|
|
1301
2570
|
// src/domains/eval/metrics/evidence.ts
|
|
1302
2571
|
init_esm_shims();
|
|
1303
2572
|
import { readFileSync as readFileSync3 } from "node:fs";
|
|
1304
|
-
import { join as
|
|
2573
|
+
import { join as join3 } from "node:path";
|
|
1305
2574
|
function dispatchScopeMetrics(receipt) {
|
|
1306
2575
|
const scope = receipt.pathScope;
|
|
1307
2576
|
if (scope === void 0) return {};
|
|
1308
2577
|
const entries = [...scope.workingContextPaths, ...scope.writeBoundaries];
|
|
1309
2578
|
const evidence = entries.flatMap((entry) => entry.evidence);
|
|
1310
|
-
const
|
|
2579
|
+
const count2 = (source) => evidence.filter((entry) => entry.source === source).length;
|
|
1311
2580
|
return {
|
|
1312
2581
|
"dispatch.scope.mode": scope.mode,
|
|
1313
2582
|
"dispatch.scope.inferredPathCount": entries.filter(
|
|
@@ -1316,9 +2585,9 @@ function dispatchScopeMetrics(receipt) {
|
|
|
1316
2585
|
"dispatch.scope.derivedPathCount": entries.filter(
|
|
1317
2586
|
(entry) => entry.evidence.some((item) => item.provenance === "derived")
|
|
1318
2587
|
).length,
|
|
1319
|
-
"dispatch.scope.source.task":
|
|
1320
|
-
"dispatch.scope.source.briefing":
|
|
1321
|
-
"dispatch.scope.source.writeRoots":
|
|
2588
|
+
"dispatch.scope.source.task": count2("task"),
|
|
2589
|
+
"dispatch.scope.source.briefing": count2("briefing"),
|
|
2590
|
+
"dispatch.scope.source.writeRoots": count2("writeRoots")
|
|
1322
2591
|
};
|
|
1323
2592
|
}
|
|
1324
2593
|
function receiptFromRunJsonStdout(stdout) {
|
|
@@ -1365,7 +2634,7 @@ function evidenceTrustMetrics(receipt, envelope) {
|
|
|
1365
2634
|
}
|
|
1366
2635
|
function readRunEnvelopeForReceipt(receipt, stateDir) {
|
|
1367
2636
|
try {
|
|
1368
|
-
const parsed = JSON.parse(readFileSync3(
|
|
2637
|
+
const parsed = JSON.parse(readFileSync3(join3(stateDir, "runs.json"), "utf8"));
|
|
1369
2638
|
if (!Array.isArray(parsed)) return null;
|
|
1370
2639
|
const row = parsed.find(
|
|
1371
2640
|
(entry) => typeof entry === "object" && entry !== null && entry.id === receipt.runId
|
|
@@ -1396,10 +2665,10 @@ function createTokenUsageFold() {
|
|
|
1396
2665
|
} catch {
|
|
1397
2666
|
return;
|
|
1398
2667
|
}
|
|
1399
|
-
if (!
|
|
1400
|
-
const message =
|
|
2668
|
+
if (!isRecord6(event) || event.type !== "message_end") return;
|
|
2669
|
+
const message = isRecord6(event.message) ? event.message : void 0;
|
|
1401
2670
|
if (message === void 0 || message.role !== "assistant") return;
|
|
1402
|
-
const usage =
|
|
2671
|
+
const usage = isRecord6(message.usage) ? message.usage : void 0;
|
|
1403
2672
|
if (usage === void 0) return;
|
|
1404
2673
|
measured = true;
|
|
1405
2674
|
const input = numberField2(usage, "input");
|
|
@@ -1412,7 +2681,7 @@ function createTokenUsageFold() {
|
|
|
1412
2681
|
tokens.cacheRead += cacheRead;
|
|
1413
2682
|
tokens.cacheWrite += cacheWrite;
|
|
1414
2683
|
tokens.total += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
|
|
1415
|
-
if (
|
|
2684
|
+
if (isRecord6(usage.cost)) costUsd += numberField2(usage.cost, "total");
|
|
1416
2685
|
};
|
|
1417
2686
|
return {
|
|
1418
2687
|
push(chunk) {
|
|
@@ -1462,14 +2731,115 @@ function numberField2(record, field) {
|
|
|
1462
2731
|
const value = record[field];
|
|
1463
2732
|
return typeof value === "number" && Number.isFinite(value) ? value : 0;
|
|
1464
2733
|
}
|
|
1465
|
-
function
|
|
2734
|
+
function isRecord6(value) {
|
|
1466
2735
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1467
2736
|
}
|
|
1468
2737
|
|
|
1469
2738
|
// src/domains/eval/runners/external-command.ts
|
|
1470
2739
|
init_esm_shims();
|
|
1471
2740
|
import { spawn } from "node:child_process";
|
|
2741
|
+
import { performance as performance2 } from "node:perf_hooks";
|
|
2742
|
+
|
|
2743
|
+
// src/domains/eval/metrics/call-ledger-stream.ts
|
|
2744
|
+
init_esm_shims();
|
|
1472
2745
|
import { performance } from "node:perf_hooks";
|
|
2746
|
+
function createEvalCallLedgerFold(now = () => performance.now()) {
|
|
2747
|
+
const entries = [];
|
|
2748
|
+
let pending = "";
|
|
2749
|
+
let activeStartedAt = null;
|
|
2750
|
+
let activeFirstOutputAt = null;
|
|
2751
|
+
const consume = (line) => {
|
|
2752
|
+
const event = parseRecord(line);
|
|
2753
|
+
if (event === null) return;
|
|
2754
|
+
if (event.type === "message_start" && isAssistantMessage(event.message)) {
|
|
2755
|
+
activeStartedAt = now();
|
|
2756
|
+
activeFirstOutputAt = null;
|
|
2757
|
+
return;
|
|
2758
|
+
}
|
|
2759
|
+
if (event.type === "message_update" && activeStartedAt !== null && activeFirstOutputAt === null) {
|
|
2760
|
+
activeFirstOutputAt = now();
|
|
2761
|
+
return;
|
|
2762
|
+
}
|
|
2763
|
+
if (event.type !== "message_end" || !isAssistantMessage(event.message)) return;
|
|
2764
|
+
const message = event.message;
|
|
2765
|
+
const usage = isRecord7(message.usage) ? message.usage : null;
|
|
2766
|
+
if (usage === null) {
|
|
2767
|
+
activeStartedAt = null;
|
|
2768
|
+
activeFirstOutputAt = null;
|
|
2769
|
+
return;
|
|
2770
|
+
}
|
|
2771
|
+
const endedAt = now();
|
|
2772
|
+
const promptCache = {
|
|
2773
|
+
input: nonNegativeNumber2(usage.input) ?? 0,
|
|
2774
|
+
cacheRead: nonNegativeNumber2(usage.cacheRead) ?? 0,
|
|
2775
|
+
cacheWrite: nonNegativeNumber2(usage.cacheWrite) ?? 0,
|
|
2776
|
+
backendVerdict: "unknown"
|
|
2777
|
+
};
|
|
2778
|
+
if (isRecord7(message.backendTimings)) promptCache.backend = structuredClone(message.backendTimings);
|
|
2779
|
+
const previous = entries.at(-1);
|
|
2780
|
+
entries.push({
|
|
2781
|
+
kind: "message",
|
|
2782
|
+
role: "assistant",
|
|
2783
|
+
turnId: `eval-call-${entries.length + 1}`,
|
|
2784
|
+
parentTurnId: previous?.turnId ?? null,
|
|
2785
|
+
timestamp: messageTimestamp(message.timestamp),
|
|
2786
|
+
payload: {
|
|
2787
|
+
promptCache,
|
|
2788
|
+
timing: {
|
|
2789
|
+
ttftMs: activeStartedAt === null ? null : Math.round(Math.max(0, (activeFirstOutputAt ?? endedAt) - activeStartedAt)),
|
|
2790
|
+
apiMs: activeStartedAt === null ? 0 : Math.round(Math.max(0, endedAt - activeStartedAt))
|
|
2791
|
+
},
|
|
2792
|
+
usage: structuredClone(usage)
|
|
2793
|
+
}
|
|
2794
|
+
});
|
|
2795
|
+
activeStartedAt = null;
|
|
2796
|
+
activeFirstOutputAt = null;
|
|
2797
|
+
};
|
|
2798
|
+
return {
|
|
2799
|
+
push(chunk) {
|
|
2800
|
+
pending += chunk;
|
|
2801
|
+
for (; ; ) {
|
|
2802
|
+
const newline = pending.indexOf("\n");
|
|
2803
|
+
if (newline === -1) break;
|
|
2804
|
+
consume(pending.slice(0, newline).replace(/\r$/u, ""));
|
|
2805
|
+
pending = pending.slice(newline + 1);
|
|
2806
|
+
}
|
|
2807
|
+
},
|
|
2808
|
+
entries() {
|
|
2809
|
+
if (pending.length > 0) {
|
|
2810
|
+
consume(pending.replace(/\r$/u, ""));
|
|
2811
|
+
pending = "";
|
|
2812
|
+
}
|
|
2813
|
+
return structuredClone(entries);
|
|
2814
|
+
}
|
|
2815
|
+
};
|
|
2816
|
+
}
|
|
2817
|
+
function isAssistantMessage(value) {
|
|
2818
|
+
return isRecord7(value) && value.role === "assistant";
|
|
2819
|
+
}
|
|
2820
|
+
function messageTimestamp(value) {
|
|
2821
|
+
const milliseconds = nonNegativeNumber2(value);
|
|
2822
|
+
if (milliseconds === null) return (/* @__PURE__ */ new Date(0)).toISOString();
|
|
2823
|
+
const date = new Date(milliseconds);
|
|
2824
|
+
return Number.isFinite(date.getTime()) ? date.toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
|
|
2825
|
+
}
|
|
2826
|
+
function nonNegativeNumber2(value) {
|
|
2827
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
|
|
2828
|
+
}
|
|
2829
|
+
function parseRecord(line) {
|
|
2830
|
+
if (line.trim().length === 0) return null;
|
|
2831
|
+
try {
|
|
2832
|
+
const value = JSON.parse(line);
|
|
2833
|
+
return isRecord7(value) ? value : null;
|
|
2834
|
+
} catch {
|
|
2835
|
+
return null;
|
|
2836
|
+
}
|
|
2837
|
+
}
|
|
2838
|
+
function isRecord7(value) {
|
|
2839
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2840
|
+
}
|
|
2841
|
+
|
|
2842
|
+
// src/domains/eval/runners/external-command.ts
|
|
1473
2843
|
var OUTPUT_LIMIT = 2e5;
|
|
1474
2844
|
var OUTPUT_HEAD_LIMIT = 2e4;
|
|
1475
2845
|
var OUTPUT_TRUNCATION_MARKER = "\n[output middle truncated; tail preserved]\n";
|
|
@@ -1484,6 +2854,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1484
2854
|
let usage = UNMEASURED_TOKEN_USAGE;
|
|
1485
2855
|
let streamInvariants = EMPTY_STREAM_INVARIANTS;
|
|
1486
2856
|
let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
|
|
2857
|
+
const ledgerEntries = [];
|
|
1487
2858
|
for (const command of commands) {
|
|
1488
2859
|
const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1489
2860
|
stdout = appendLimited(stdout, result.stdout);
|
|
@@ -1492,6 +2863,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1492
2863
|
usage = addTokenStreamUsage(usage, result.usage);
|
|
1493
2864
|
streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
|
|
1494
2865
|
fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
|
|
2866
|
+
ledgerEntries.push(...result.ledgerEntries);
|
|
1495
2867
|
if (result.exitCode !== 0) {
|
|
1496
2868
|
return {
|
|
1497
2869
|
assignmentId: null,
|
|
@@ -1507,7 +2879,8 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1507
2879
|
...fleetLoopMetricEntries(fleetLoops),
|
|
1508
2880
|
"verifier.exitCode": result.exitCode
|
|
1509
2881
|
},
|
|
1510
|
-
artifacts: {}
|
|
2882
|
+
artifacts: {},
|
|
2883
|
+
ledgerEntries
|
|
1511
2884
|
};
|
|
1512
2885
|
}
|
|
1513
2886
|
}
|
|
@@ -1525,18 +2898,20 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1525
2898
|
...fleetLoopMetricEntries(fleetLoops),
|
|
1526
2899
|
"verifier.exitCode": 0
|
|
1527
2900
|
},
|
|
1528
|
-
artifacts: {}
|
|
2901
|
+
artifacts: {},
|
|
2902
|
+
ledgerEntries
|
|
1529
2903
|
};
|
|
1530
2904
|
}
|
|
1531
2905
|
function runShellCommand(command, cwd, timeoutMs, env) {
|
|
1532
|
-
const started =
|
|
1533
|
-
return new Promise((
|
|
2906
|
+
const started = performance2.now();
|
|
2907
|
+
return new Promise((resolve9) => {
|
|
1534
2908
|
let stdout = "";
|
|
1535
2909
|
let stderr = "";
|
|
1536
2910
|
const metricCapture = createJsonlMetricCapture();
|
|
1537
2911
|
const usageFold = createTokenUsageFold();
|
|
1538
2912
|
const streamFold = createStreamInvariantFold();
|
|
1539
2913
|
const fleetLoopFold = createFleetLoopFold();
|
|
2914
|
+
const callLedgerFold = createEvalCallLedgerFold();
|
|
1540
2915
|
let timedOut = false;
|
|
1541
2916
|
let settled = false;
|
|
1542
2917
|
const child = spawn(command, {
|
|
@@ -1555,6 +2930,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1555
2930
|
usageFold.push(chunk);
|
|
1556
2931
|
streamFold.push(chunk);
|
|
1557
2932
|
fleetLoopFold.push(chunk);
|
|
2933
|
+
callLedgerFold.push(chunk);
|
|
1558
2934
|
});
|
|
1559
2935
|
child.stderr.on("data", (chunk) => {
|
|
1560
2936
|
stderr = appendLimited(stderr, chunk);
|
|
@@ -1568,7 +2944,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1568
2944
|
if (settled) return;
|
|
1569
2945
|
settled = true;
|
|
1570
2946
|
clearTimeout(timer);
|
|
1571
|
-
|
|
2947
|
+
resolve9({
|
|
1572
2948
|
command,
|
|
1573
2949
|
exitCode,
|
|
1574
2950
|
stdout,
|
|
@@ -1576,8 +2952,9 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1576
2952
|
usage: usageFold.usage(),
|
|
1577
2953
|
streamInvariants: streamFold.invariants(),
|
|
1578
2954
|
fleetLoops: fleetLoopFold.observation(),
|
|
2955
|
+
ledgerEntries: callLedgerFold.entries(),
|
|
1579
2956
|
stderr,
|
|
1580
|
-
wallTimeMs: Math.round(
|
|
2957
|
+
wallTimeMs: Math.round(performance2.now() - started),
|
|
1581
2958
|
timedOut
|
|
1582
2959
|
});
|
|
1583
2960
|
};
|
|
@@ -1609,7 +2986,7 @@ function createJsonlMetricCapture() {
|
|
|
1609
2986
|
} catch {
|
|
1610
2987
|
return;
|
|
1611
2988
|
}
|
|
1612
|
-
if (!
|
|
2989
|
+
if (!isRecord8(parsed)) return;
|
|
1613
2990
|
const compact = compactMetricEvent(parsed);
|
|
1614
2991
|
if (compact === null) return;
|
|
1615
2992
|
const encoded = JSON.stringify(compact);
|
|
@@ -1663,7 +3040,7 @@ function compactMetricEvent(event) {
|
|
|
1663
3040
|
type,
|
|
1664
3041
|
...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
|
|
1665
3042
|
toolName,
|
|
1666
|
-
...toolName === "dispatch" &&
|
|
3043
|
+
...toolName === "dispatch" && isRecord8(event.args) ? { args: event.args } : toolName === "read" && isRecord8(event.args) ? { args: boundedReadArgs(event.args) } : toolName === "code_nav" && isRecord8(event.args) ? { args: { mode: event.args.mode } } : {}
|
|
1667
3044
|
};
|
|
1668
3045
|
}
|
|
1669
3046
|
if (type === "tool_execution_end") {
|
|
@@ -1675,7 +3052,7 @@ function compactMetricEvent(event) {
|
|
|
1675
3052
|
...stringField(event, "outcome") !== void 0 ? { outcome: stringField(event, "outcome") } : {}
|
|
1676
3053
|
};
|
|
1677
3054
|
}
|
|
1678
|
-
if (type !== "clio_tool_finish" || !
|
|
3055
|
+
if (type !== "clio_tool_finish" || !isRecord8(event.payload)) return null;
|
|
1679
3056
|
return {
|
|
1680
3057
|
type,
|
|
1681
3058
|
payload: {
|
|
@@ -1685,7 +3062,14 @@ function compactMetricEvent(event) {
|
|
|
1685
3062
|
}
|
|
1686
3063
|
};
|
|
1687
3064
|
}
|
|
1688
|
-
function
|
|
3065
|
+
function boundedReadArgs(args) {
|
|
3066
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
3067
|
+
const value = args[field];
|
|
3068
|
+
if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4096) };
|
|
3069
|
+
}
|
|
3070
|
+
return {};
|
|
3071
|
+
}
|
|
3072
|
+
function isRecord8(value) {
|
|
1689
3073
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1690
3074
|
}
|
|
1691
3075
|
function stringField(record, field) {
|
|
@@ -1694,7 +3078,7 @@ function stringField(record, field) {
|
|
|
1694
3078
|
}
|
|
1695
3079
|
|
|
1696
3080
|
// src/domains/eval/runners/clio-run.ts
|
|
1697
|
-
async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
|
|
3081
|
+
async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env, readObservation) {
|
|
1698
3082
|
const prompt = runner.prompt ?? "";
|
|
1699
3083
|
const args = [
|
|
1700
3084
|
shellQuote(clioEntry),
|
|
@@ -1705,12 +3089,14 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1705
3089
|
shellQuote(target.id),
|
|
1706
3090
|
...target.model === void 0 ? [] : ["--model", shellQuote(target.model)],
|
|
1707
3091
|
...target.thinking === void 0 ? [] : ["--thinking", shellQuote(target.thinking)],
|
|
3092
|
+
...runner.autonomy === void 0 ? [] : ["--autonomy", runner.autonomy],
|
|
1708
3093
|
shellQuote(prompt)
|
|
1709
3094
|
];
|
|
1710
3095
|
const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1711
3096
|
const tokens = result.usage;
|
|
1712
3097
|
const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
|
|
1713
3098
|
const tools = toolCallMetricsFromJsonl(toolMetricStream);
|
|
3099
|
+
const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
|
|
1714
3100
|
const receipt = receiptFromRunJsonStdout(result.stdout);
|
|
1715
3101
|
const envelope = receipt === null ? null : readRunEnvelopeForReceipt(receipt, env?.CLIO_CODER_STATE_DIR ?? clioStateDir());
|
|
1716
3102
|
return {
|
|
@@ -1729,6 +3115,7 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1729
3115
|
"tools.totalCalls": tools.totalCalls,
|
|
1730
3116
|
"tools.failed": tools.failed,
|
|
1731
3117
|
"tools.blocked": tools.blocked,
|
|
3118
|
+
...behavioralTools,
|
|
1732
3119
|
"verifier.exitCode": result.exitCode,
|
|
1733
3120
|
...receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope }),
|
|
1734
3121
|
...receipt === null ? {} : { "evidence.qualityLabel": receipt.quality.typedValidations.length > 0 ? "measured" : "unmeasured" }
|
|
@@ -1736,8 +3123,11 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1736
3123
|
artifacts: {
|
|
1737
3124
|
stdout: result.stdout,
|
|
1738
3125
|
stderr: result.stderr,
|
|
3126
|
+
callLedger: JSON.stringify(result.ledgerEntries),
|
|
1739
3127
|
...receipt === null ? {} : { receipt: JSON.stringify(receipt) }
|
|
1740
|
-
}
|
|
3128
|
+
},
|
|
3129
|
+
receipt,
|
|
3130
|
+
ledgerEntries: result.ledgerEntries
|
|
1741
3131
|
};
|
|
1742
3132
|
}
|
|
1743
3133
|
function toolCallMetricsFromJsonl(stdout) {
|
|
@@ -1750,7 +3140,7 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1750
3140
|
let event;
|
|
1751
3141
|
try {
|
|
1752
3142
|
const parsed = JSON.parse(line);
|
|
1753
|
-
if (!
|
|
3143
|
+
if (!isRecord9(parsed)) continue;
|
|
1754
3144
|
event = parsed;
|
|
1755
3145
|
} catch {
|
|
1756
3146
|
continue;
|
|
@@ -1764,7 +3154,7 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1764
3154
|
recordToolOutcome(executionEnds, toolOutcome(event) ?? (event.isError === true ? "error" : "ok"));
|
|
1765
3155
|
continue;
|
|
1766
3156
|
}
|
|
1767
|
-
if (event.type !== "clio_tool_finish" || !
|
|
3157
|
+
if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
|
|
1768
3158
|
const outcome = toolOutcome(event.payload);
|
|
1769
3159
|
if (outcome === void 0) continue;
|
|
1770
3160
|
const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId");
|
|
@@ -1776,6 +3166,95 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1776
3166
|
}
|
|
1777
3167
|
return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
|
|
1778
3168
|
}
|
|
3169
|
+
function toolBehaviorMetricEntriesFromJsonl(stdout, cwd, readObservation) {
|
|
3170
|
+
const starts = /* @__PURE__ */ new Map();
|
|
3171
|
+
const readPaths = /* @__PURE__ */ new Set();
|
|
3172
|
+
const executionEnds = [];
|
|
3173
|
+
const canonicalFinishes = [];
|
|
3174
|
+
const seenExecution = /* @__PURE__ */ new Set();
|
|
3175
|
+
const seenCanonical = /* @__PURE__ */ new Set();
|
|
3176
|
+
for (const line of stdout.split(/\r?\n/)) {
|
|
3177
|
+
if (line.trim().length === 0) continue;
|
|
3178
|
+
let event;
|
|
3179
|
+
try {
|
|
3180
|
+
const parsed = JSON.parse(line);
|
|
3181
|
+
if (!isRecord9(parsed)) continue;
|
|
3182
|
+
event = parsed;
|
|
3183
|
+
} catch {
|
|
3184
|
+
continue;
|
|
3185
|
+
}
|
|
3186
|
+
if (event.type === "tool_execution_start") {
|
|
3187
|
+
const callId2 = stringField2(event, "toolCallId");
|
|
3188
|
+
const tool2 = stringField2(event, "toolName");
|
|
3189
|
+
if (callId2 === void 0 || tool2 === void 0) continue;
|
|
3190
|
+
const path = tool2 === "read" && isRecord9(event.args) ? toolPath(event.args) : null;
|
|
3191
|
+
starts.set(callId2, { tool: tool2, path });
|
|
3192
|
+
if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
|
|
3193
|
+
continue;
|
|
3194
|
+
}
|
|
3195
|
+
if (event.type === "tool_execution_end") {
|
|
3196
|
+
const callId2 = stringField2(event, "toolCallId") ?? null;
|
|
3197
|
+
if (callId2 !== null && seenExecution.has(callId2)) continue;
|
|
3198
|
+
if (callId2 !== null) seenExecution.add(callId2);
|
|
3199
|
+
const tool2 = stringField2(event, "toolName") ?? (callId2 === null ? void 0 : starts.get(callId2)?.tool);
|
|
3200
|
+
if (tool2 === void 0) continue;
|
|
3201
|
+
executionEnds.push({ callId: callId2, tool: tool2, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
|
|
3202
|
+
continue;
|
|
3203
|
+
}
|
|
3204
|
+
if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
|
|
3205
|
+
const outcome = toolOutcome(event.payload);
|
|
3206
|
+
const tool = stringField2(event.payload, "tool");
|
|
3207
|
+
if (outcome === void 0 || tool === void 0) continue;
|
|
3208
|
+
const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId") ?? null;
|
|
3209
|
+
if (callId !== null && seenCanonical.has(callId)) continue;
|
|
3210
|
+
if (callId !== null) seenCanonical.add(callId);
|
|
3211
|
+
canonicalFinishes.push({ callId, tool, outcome });
|
|
3212
|
+
}
|
|
3213
|
+
const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
|
|
3214
|
+
const calls = /* @__PURE__ */ new Map();
|
|
3215
|
+
const blocked = /* @__PURE__ */ new Map();
|
|
3216
|
+
for (const terminal of terminals) {
|
|
3217
|
+
const tool = metricToolName(terminal.tool);
|
|
3218
|
+
calls.set(tool, (calls.get(tool) ?? 0) + 1);
|
|
3219
|
+
if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
|
|
3220
|
+
}
|
|
3221
|
+
const namedTools = /* @__PURE__ */ new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
|
|
3222
|
+
const entries = { "tools.read.distinctPaths": readPaths.size };
|
|
3223
|
+
for (const tool of [...namedTools].sort()) {
|
|
3224
|
+
entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
|
|
3225
|
+
entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
|
|
3226
|
+
}
|
|
3227
|
+
if (readObservation !== void 0) {
|
|
3228
|
+
const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
3229
|
+
const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
3230
|
+
entries["tools.read.outsideAllowed"] = [...readPaths].filter(
|
|
3231
|
+
(path) => !allowed.some((root) => pathWithin(path, root))
|
|
3232
|
+
).length;
|
|
3233
|
+
entries["tools.read.decoyHits"] = [...readPaths].filter(
|
|
3234
|
+
(path) => decoys.some((root) => pathWithin(path, root))
|
|
3235
|
+
).length;
|
|
3236
|
+
}
|
|
3237
|
+
return entries;
|
|
3238
|
+
}
|
|
3239
|
+
function toolPath(args) {
|
|
3240
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
3241
|
+
const value = args[field];
|
|
3242
|
+
if (typeof value === "string" && value.length > 0 && value.length <= 4096) return value;
|
|
3243
|
+
}
|
|
3244
|
+
return null;
|
|
3245
|
+
}
|
|
3246
|
+
function normalizeObservedPath(cwd, path) {
|
|
3247
|
+
const absolute = resolve2(cwd, path);
|
|
3248
|
+
const local = relative(cwd, absolute);
|
|
3249
|
+
return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".").split(sep).join("/");
|
|
3250
|
+
}
|
|
3251
|
+
function pathWithin(path, root) {
|
|
3252
|
+
if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
|
|
3253
|
+
return path === root || path.startsWith(`${root}/`);
|
|
3254
|
+
}
|
|
3255
|
+
function metricToolName(tool) {
|
|
3256
|
+
return tool.toLowerCase().replaceAll(/[^a-z0-9_-]/gu, "_").slice(0, 64) || "unknown";
|
|
3257
|
+
}
|
|
1779
3258
|
function recordToolOutcome(metrics, outcome) {
|
|
1780
3259
|
metrics.totalCalls += 1;
|
|
1781
3260
|
if (outcome === "error") metrics.failed += 1;
|
|
@@ -1789,7 +3268,7 @@ function stringField2(record, field) {
|
|
|
1789
3268
|
const value = record[field];
|
|
1790
3269
|
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
1791
3270
|
}
|
|
1792
|
-
function
|
|
3271
|
+
function isRecord9(value) {
|
|
1793
3272
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1794
3273
|
}
|
|
1795
3274
|
|
|
@@ -1840,7 +3319,7 @@ function parseContextIndexOutput(stdout) {
|
|
|
1840
3319
|
// src/domains/eval/runners/context-init.ts
|
|
1841
3320
|
init_esm_shims();
|
|
1842
3321
|
import { existsSync as existsSync2, statSync } from "node:fs";
|
|
1843
|
-
import { join as
|
|
3322
|
+
import { join as join4 } from "node:path";
|
|
1844
3323
|
async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
|
|
1845
3324
|
const extraArgs = runner.args ?? [];
|
|
1846
3325
|
const command = [
|
|
@@ -1858,22 +3337,22 @@ async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, e
|
|
|
1858
3337
|
].map(shellQuote).join(" ");
|
|
1859
3338
|
const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1860
3339
|
const payload = parseInitPayload(result.stdout);
|
|
1861
|
-
const candidateGeneration =
|
|
3340
|
+
const candidateGeneration = recordField2(payload, "generation");
|
|
1862
3341
|
const generation = isValidGenerationPayload(payload, candidateGeneration) ? candidateGeneration : null;
|
|
1863
3342
|
const routeError = generation ? generationRouteError(generation, target) : null;
|
|
1864
3343
|
const payloadError = generation ? routeError : "context-init runner did not receive a valid JSON generation result";
|
|
1865
3344
|
const exitCode = result.exitCode === 0 && payloadError ? 1 : result.exitCode;
|
|
1866
3345
|
const stderr = payloadError ? `${result.stderr}${result.stderr.endsWith("\n") || result.stderr.length === 0 ? "" : "\n"}${payloadError}
|
|
1867
3346
|
` : result.stderr;
|
|
1868
|
-
const run =
|
|
1869
|
-
const tokens =
|
|
3347
|
+
const run = recordField2(generation, "run");
|
|
3348
|
+
const tokens = recordField2(run, "tokens");
|
|
1870
3349
|
const effectiveTarget = stringField3(run, "targetId");
|
|
1871
3350
|
const effectiveModel = stringField3(run, "wireModelId");
|
|
1872
3351
|
const effectiveRuntime = stringField3(run, "runtimeId");
|
|
1873
3352
|
const effectiveRuntimeKind = stringField3(run, "runtimeKind");
|
|
1874
3353
|
const effectiveThinking = stringField3(run, "thinkingLevel");
|
|
1875
3354
|
const structuredOutputMode = stringField3(run, "structuredOutputMode");
|
|
1876
|
-
const clioMdPath =
|
|
3355
|
+
const clioMdPath = join4(cwd, "CLIO-CODER.md");
|
|
1877
3356
|
const clioMdBytes = existsSync2(clioMdPath) ? statSync(clioMdPath).size : 0;
|
|
1878
3357
|
return {
|
|
1879
3358
|
assignmentId: null,
|
|
@@ -1927,7 +3406,7 @@ function isNonnegativeFiniteNumber(value) {
|
|
|
1927
3406
|
return typeof value === "number" && Number.isFinite(value) && value >= 0;
|
|
1928
3407
|
}
|
|
1929
3408
|
function isValidRunPayload(value) {
|
|
1930
|
-
const run =
|
|
3409
|
+
const run = recordField2(value);
|
|
1931
3410
|
if (!run) return false;
|
|
1932
3411
|
for (const key of ["durationMs", "promptBytes", "outputBytes"]) {
|
|
1933
3412
|
if (!isNonnegativeFiniteNumber(run[key])) return false;
|
|
@@ -1936,7 +3415,7 @@ function isValidRunPayload(value) {
|
|
|
1936
3415
|
for (const key of ["toolCalls", "toolFailures", "toolBlocked"]) {
|
|
1937
3416
|
if (run[key] !== void 0 && !isNonnegativeFiniteNumber(run[key])) return false;
|
|
1938
3417
|
}
|
|
1939
|
-
const tokens =
|
|
3418
|
+
const tokens = recordField2(run, "tokens");
|
|
1940
3419
|
if (run.tokens !== void 0 && !tokens) return false;
|
|
1941
3420
|
if (tokens) {
|
|
1942
3421
|
for (const key of ["total", "input", "output", "cacheRead", "cacheWrite", "reasoning"]) {
|
|
@@ -1962,14 +3441,14 @@ function isValidGenerationPayload(payload, generation) {
|
|
|
1962
3441
|
}
|
|
1963
3442
|
const runPresent = generation.run !== void 0;
|
|
1964
3443
|
if (runPresent && !isValidRunPayload(generation.run)) return false;
|
|
1965
|
-
const run =
|
|
3444
|
+
const run = recordField2(generation, "run");
|
|
1966
3445
|
if (mode === "model" && (parserOutcome !== "parsed" || !hasReceiptIdentity(run))) return false;
|
|
1967
3446
|
if ((parserOutcome === "parsed" || parserOutcome === "rejected") && !runPresent) return false;
|
|
1968
3447
|
if (parserOutcome === "rejected" && !hasReceiptIdentity(run)) return false;
|
|
1969
3448
|
return true;
|
|
1970
3449
|
}
|
|
1971
3450
|
function generationRouteError(generation, target) {
|
|
1972
|
-
const run =
|
|
3451
|
+
const run = recordField2(generation, "run");
|
|
1973
3452
|
if (!run) return null;
|
|
1974
3453
|
const actualTarget = stringField3(run, "targetId");
|
|
1975
3454
|
const actualModel = stringField3(run, "wireModelId");
|
|
@@ -1988,34 +3467,101 @@ function generationRouteError(generation, target) {
|
|
|
1988
3467
|
function parseInitPayload(stdout) {
|
|
1989
3468
|
try {
|
|
1990
3469
|
const parsed = JSON.parse(stdout);
|
|
1991
|
-
return
|
|
3470
|
+
return recordField2(parsed);
|
|
1992
3471
|
} catch {
|
|
1993
3472
|
return null;
|
|
1994
3473
|
}
|
|
1995
3474
|
}
|
|
1996
|
-
function
|
|
3475
|
+
function recordField2(value, field) {
|
|
1997
3476
|
const selected = field && typeof value === "object" && value !== null && !Array.isArray(value) ? value[field] : value;
|
|
1998
3477
|
return typeof selected === "object" && selected !== null && !Array.isArray(selected) ? selected : null;
|
|
1999
3478
|
}
|
|
2000
3479
|
function numberField3(value, field) {
|
|
2001
|
-
const record =
|
|
3480
|
+
const record = recordField2(value);
|
|
2002
3481
|
const selected = record?.[field];
|
|
2003
3482
|
return typeof selected === "number" && Number.isFinite(selected) ? selected : null;
|
|
2004
3483
|
}
|
|
2005
3484
|
function stringField3(value, field) {
|
|
2006
|
-
const record =
|
|
3485
|
+
const record = recordField2(value);
|
|
2007
3486
|
const selected = record?.[field];
|
|
2008
3487
|
return typeof selected === "string" && selected.length > 0 ? selected : null;
|
|
2009
3488
|
}
|
|
2010
3489
|
|
|
3490
|
+
// src/domains/eval/schema/adapter.ts
|
|
3491
|
+
init_esm_shims();
|
|
3492
|
+
import { createHash as createHash3 } from "node:crypto";
|
|
3493
|
+
function adaptSuiteV2ResultToVerdictV1(result, trackedMetrics) {
|
|
3494
|
+
const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
|
|
3495
|
+
const outcome = result.pass ? "pass" : "fail";
|
|
3496
|
+
const graderExitCode = result.metrics["task.exitCode"];
|
|
3497
|
+
return parseEvalVerdictEnvelopeV1({
|
|
3498
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
3499
|
+
scenarioId: result.taskId,
|
|
3500
|
+
trialIndex: result.repeatIndex,
|
|
3501
|
+
outcome,
|
|
3502
|
+
machinery,
|
|
3503
|
+
reason: result.pass ? null : result.failureClass ?? "result_failed",
|
|
3504
|
+
trackedMetrics,
|
|
3505
|
+
behavioral: null,
|
|
3506
|
+
evidence: {
|
|
3507
|
+
assignmentId: result.assignmentId,
|
|
3508
|
+
terminalReceiptDigest: result.terminalReceiptDigest,
|
|
3509
|
+
graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null
|
|
3510
|
+
}
|
|
3511
|
+
});
|
|
3512
|
+
}
|
|
3513
|
+
function adaptSuiteV2ResultToBehaviorV1(result, verdict, scenario) {
|
|
3514
|
+
const requestedFacts = new Set(
|
|
3515
|
+
[...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
|
|
3516
|
+
(rule) => `${rule.fact.source}\0${rule.fact.key}`
|
|
3517
|
+
)
|
|
3518
|
+
);
|
|
3519
|
+
const observedSources = /* @__PURE__ */ new Set();
|
|
3520
|
+
const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
|
|
3521
|
+
if (value === null) return [];
|
|
3522
|
+
const source = metricFactSource(key);
|
|
3523
|
+
observedSources.add(source);
|
|
3524
|
+
if (!requestedFacts.has(`${source}\0${key}`)) return [];
|
|
3525
|
+
const serialized = JSON.stringify({ source, key, value });
|
|
3526
|
+
const digest = createHash3("sha256").update(serialized, "utf8").digest("hex");
|
|
3527
|
+
const fact = {
|
|
3528
|
+
id: `metric-${digest.slice(0, 16)}`,
|
|
3529
|
+
source,
|
|
3530
|
+
key,
|
|
3531
|
+
value,
|
|
3532
|
+
evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1e3) }
|
|
3533
|
+
};
|
|
3534
|
+
return [fact];
|
|
3535
|
+
});
|
|
3536
|
+
const allSources = ["transcript", "tool", "receipt", "grader"];
|
|
3537
|
+
const unavailableSources = allSources.filter(
|
|
3538
|
+
(source) => !observedSources.has(source) || source === "tool" && scenario.execution.toolTarget === "none"
|
|
3539
|
+
);
|
|
3540
|
+
const behavior = judgeEvalBehaviorV1(scenario, verdict, {
|
|
3541
|
+
facts,
|
|
3542
|
+
unavailableSources,
|
|
3543
|
+
infrastructureFailure: verdict.machinery === "infrastructure_failure"
|
|
3544
|
+
});
|
|
3545
|
+
assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
|
|
3546
|
+
return behavior;
|
|
3547
|
+
}
|
|
3548
|
+
function metricFactSource(key) {
|
|
3549
|
+
if (key.startsWith("tools.")) return "tool";
|
|
3550
|
+
if (key.startsWith("task.") || key.startsWith("claims.") || key.startsWith("completion.") || key === "result.pass" || key === "verifier.exitCode")
|
|
3551
|
+
return "grader";
|
|
3552
|
+
if (key.startsWith("receipt.") || key.startsWith("evidence.") || key.startsWith("boundary.") || key.startsWith("loop.") || key.startsWith("cost."))
|
|
3553
|
+
return "receipt";
|
|
3554
|
+
return "transcript";
|
|
3555
|
+
}
|
|
3556
|
+
|
|
2011
3557
|
// src/domains/eval/verifiers/command.ts
|
|
2012
3558
|
init_esm_shims();
|
|
2013
|
-
async function runCommandVerifiers(commands, cwd, timeoutMs) {
|
|
3559
|
+
async function runCommandVerifiers(commands, cwd, timeoutMs, env) {
|
|
2014
3560
|
let stdout = "";
|
|
2015
3561
|
let stderr = "";
|
|
2016
3562
|
let wallTimeMs = 0;
|
|
2017
3563
|
for (const command of commands) {
|
|
2018
|
-
const result = await runShellCommand(command, cwd, timeoutMs);
|
|
3564
|
+
const result = await runShellCommand(command, cwd, timeoutMs, env);
|
|
2019
3565
|
stdout += result.stdout;
|
|
2020
3566
|
stderr += result.stderr;
|
|
2021
3567
|
wallTimeMs += result.wallTimeMs;
|
|
@@ -2027,9 +3573,9 @@ async function runCommandVerifiers(commands, cwd, timeoutMs) {
|
|
|
2027
3573
|
// src/domains/eval/verifiers/file-exists.ts
|
|
2028
3574
|
init_esm_shims();
|
|
2029
3575
|
import { existsSync as existsSync3 } from "node:fs";
|
|
2030
|
-
import { resolve as
|
|
3576
|
+
import { resolve as resolve3 } from "node:path";
|
|
2031
3577
|
function forbiddenPathHits(cwd, paths) {
|
|
2032
|
-
return paths.filter((path) => existsSync3(
|
|
3578
|
+
return paths.filter((path) => existsSync3(resolve3(cwd, path)));
|
|
2033
3579
|
}
|
|
2034
3580
|
|
|
2035
3581
|
// src/domains/eval/verifiers/patch.ts
|
|
@@ -2051,10 +3597,10 @@ init_esm_shims();
|
|
|
2051
3597
|
import { spawn as spawn2 } from "node:child_process";
|
|
2052
3598
|
import { mkdtemp, rm } from "node:fs/promises";
|
|
2053
3599
|
import { tmpdir } from "node:os";
|
|
2054
|
-
import { resolve as
|
|
3600
|
+
import { resolve as resolve4 } from "node:path";
|
|
2055
3601
|
async function prepareGitWorkspace(workspace) {
|
|
2056
3602
|
if (workspace.url === void 0) throw new Error("git workspace requires url");
|
|
2057
|
-
const dest = await mkdtemp(
|
|
3603
|
+
const dest = await mkdtemp(resolve4(tmpdir(), "clio-eval-git-"));
|
|
2058
3604
|
try {
|
|
2059
3605
|
await runGit(["clone", "--quiet", workspace.url, dest], process.cwd());
|
|
2060
3606
|
const ref = workspace.checkout ?? workspace.commit;
|
|
@@ -2089,9 +3635,9 @@ function runGit(args, cwd) {
|
|
|
2089
3635
|
// src/domains/eval/workspaces/local.ts
|
|
2090
3636
|
init_esm_shims();
|
|
2091
3637
|
import { access } from "node:fs/promises";
|
|
2092
|
-
import { resolve as
|
|
3638
|
+
import { resolve as resolve5 } from "node:path";
|
|
2093
3639
|
async function prepareLocalWorkspace(baseDir, workspace) {
|
|
2094
|
-
const dir =
|
|
3640
|
+
const dir = resolve5(baseDir, workspace.path ?? ".");
|
|
2095
3641
|
await access(dir);
|
|
2096
3642
|
return { dir, cleanup: async () => {
|
|
2097
3643
|
} };
|
|
@@ -2099,28 +3645,113 @@ async function prepareLocalWorkspace(baseDir, workspace) {
|
|
|
2099
3645
|
|
|
2100
3646
|
// src/domains/eval/workspaces/temp-copy.ts
|
|
2101
3647
|
init_esm_shims();
|
|
2102
|
-
import {
|
|
3648
|
+
import { execFile } from "node:child_process";
|
|
3649
|
+
import { cp, lstat, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
|
|
2103
3650
|
import { tmpdir as tmpdir2 } from "node:os";
|
|
2104
|
-
import { relative, resolve as
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2110
|
-
|
|
2111
|
-
|
|
2112
|
-
|
|
2113
|
-
|
|
2114
|
-
|
|
2115
|
-
|
|
2116
|
-
|
|
2117
|
-
|
|
2118
|
-
|
|
3651
|
+
import { relative as relative2, resolve as resolve6 } from "node:path";
|
|
3652
|
+
import { promisify } from "node:util";
|
|
3653
|
+
var execFileAsync = promisify(execFile);
|
|
3654
|
+
var GIT_FILE_LIST_LIMIT_BYTES = 128 * 1024 * 1024;
|
|
3655
|
+
async function prepareTempCopyWorkspace(baseDir, workspace, options = {}) {
|
|
3656
|
+
const source = resolve6(baseDir, workspace.path ?? ".");
|
|
3657
|
+
const dest = await mkdtemp2(resolve6(options.tempRoot ?? tmpdir2(), "clio-eval-workspace-"));
|
|
3658
|
+
try {
|
|
3659
|
+
const selection = await gitCopySelection(source);
|
|
3660
|
+
const excludes = workspace.excludes ?? [];
|
|
3661
|
+
const copyWorkspace = options.copy ?? defaultCopy;
|
|
3662
|
+
await copyWorkspace(source, dest, {
|
|
3663
|
+
recursive: true,
|
|
3664
|
+
filter: (path) => shouldCopy(relative2(source, path), excludes, selection)
|
|
3665
|
+
});
|
|
3666
|
+
return {
|
|
3667
|
+
dir: dest,
|
|
3668
|
+
cleanup: async () => {
|
|
3669
|
+
await rm2(dest, { recursive: true, force: true });
|
|
3670
|
+
}
|
|
3671
|
+
};
|
|
3672
|
+
} catch (error) {
|
|
3673
|
+
await rm2(dest, { recursive: true, force: true });
|
|
3674
|
+
throw error;
|
|
3675
|
+
}
|
|
2119
3676
|
}
|
|
2120
3677
|
function isExcluded(rel, excludes) {
|
|
2121
3678
|
const normalized = rel.replaceAll("\\", "/");
|
|
2122
3679
|
return excludes.some((entry) => normalized === entry || normalized.startsWith(`${entry.replaceAll("\\", "/")}/`));
|
|
2123
3680
|
}
|
|
3681
|
+
async function defaultCopy(source, destination, options) {
|
|
3682
|
+
await cp(source, destination, options);
|
|
3683
|
+
}
|
|
3684
|
+
function shouldCopy(relativePath, excludes, selection) {
|
|
3685
|
+
const normalized = relativePath.replaceAll("\\", "/");
|
|
3686
|
+
if (normalized.length === 0) return true;
|
|
3687
|
+
if (isExcluded(normalized, excludes)) return false;
|
|
3688
|
+
if (selection === null) return true;
|
|
3689
|
+
return selection.files.has(normalized) || selection.directories.has(normalized);
|
|
3690
|
+
}
|
|
3691
|
+
async function gitCopySelection(source) {
|
|
3692
|
+
let inside;
|
|
3693
|
+
try {
|
|
3694
|
+
inside = await gitOutput(source, ["rev-parse", "--is-inside-work-tree"]);
|
|
3695
|
+
} catch (error) {
|
|
3696
|
+
if (await hasGitMarker(source)) throw error;
|
|
3697
|
+
return null;
|
|
3698
|
+
}
|
|
3699
|
+
if (inside.trim() !== "true") return null;
|
|
3700
|
+
const output = await gitOutput(source, [
|
|
3701
|
+
"--literal-pathspecs",
|
|
3702
|
+
"ls-files",
|
|
3703
|
+
"-z",
|
|
3704
|
+
"--cached",
|
|
3705
|
+
"--others",
|
|
3706
|
+
"--exclude-standard",
|
|
3707
|
+
"--",
|
|
3708
|
+
"."
|
|
3709
|
+
]);
|
|
3710
|
+
const files = /* @__PURE__ */ new Set();
|
|
3711
|
+
const directories = /* @__PURE__ */ new Set();
|
|
3712
|
+
for (const path of output.split("\0")) {
|
|
3713
|
+
if (path.length === 0) continue;
|
|
3714
|
+
const normalized = normalizeGitPath(path);
|
|
3715
|
+
if (normalized === null) throw new Error("git ls-files returned a path outside the eval workspace");
|
|
3716
|
+
files.add(normalized);
|
|
3717
|
+
let separator = normalized.lastIndexOf("/");
|
|
3718
|
+
while (separator >= 0) {
|
|
3719
|
+
directories.add(normalized.slice(0, separator));
|
|
3720
|
+
separator = normalized.lastIndexOf("/", separator - 1);
|
|
3721
|
+
}
|
|
3722
|
+
}
|
|
3723
|
+
return { files, directories };
|
|
3724
|
+
}
|
|
3725
|
+
async function gitOutput(cwd, args) {
|
|
3726
|
+
const { stdout } = await execFileAsync("git", [...args], {
|
|
3727
|
+
cwd,
|
|
3728
|
+
encoding: "utf8",
|
|
3729
|
+
maxBuffer: GIT_FILE_LIST_LIMIT_BYTES
|
|
3730
|
+
});
|
|
3731
|
+
return stdout;
|
|
3732
|
+
}
|
|
3733
|
+
function normalizeGitPath(path) {
|
|
3734
|
+
const normalized = path.replaceAll("\\", "/").replace(/^\.\//u, "");
|
|
3735
|
+
if (normalized.length === 0 || normalized.startsWith("/") || /^[A-Za-z]:\//u.test(normalized)) return null;
|
|
3736
|
+
const segments = normalized.split("/");
|
|
3737
|
+
if (segments.some((segment) => segment.length === 0 || segment === "." || segment === "..")) return null;
|
|
3738
|
+
return normalized;
|
|
3739
|
+
}
|
|
3740
|
+
async function hasGitMarker(source) {
|
|
3741
|
+
let current = resolve6(source);
|
|
3742
|
+
while (true) {
|
|
3743
|
+
try {
|
|
3744
|
+
await lstat(resolve6(current, ".git"));
|
|
3745
|
+
return true;
|
|
3746
|
+
} catch (error) {
|
|
3747
|
+
const code = typeof error === "object" && error !== null && "code" in error ? error.code : void 0;
|
|
3748
|
+
if (code !== "ENOENT" && code !== "ENOTDIR") throw error;
|
|
3749
|
+
}
|
|
3750
|
+
const parent = resolve6(current, "..");
|
|
3751
|
+
if (parent === current) return false;
|
|
3752
|
+
current = parent;
|
|
3753
|
+
}
|
|
3754
|
+
}
|
|
2124
3755
|
|
|
2125
3756
|
// src/domains/eval/suites/matrix.ts
|
|
2126
3757
|
init_esm_shims();
|
|
@@ -2148,28 +3779,39 @@ async function runEvalSuiteV2(loaded, options) {
|
|
|
2148
3779
|
const started = now();
|
|
2149
3780
|
const evalId = createEvalId(started, loaded.hash);
|
|
2150
3781
|
const results = [];
|
|
3782
|
+
const servingObservations = [];
|
|
2151
3783
|
const maxCostUsd = loaded.suite.matrix.maxCostUsd;
|
|
2152
3784
|
let spentUsd = 0;
|
|
2153
3785
|
for (const item of expandEvalMatrix(loaded.suite)) {
|
|
2154
3786
|
if (maxCostUsd !== void 0 && spentUsd > maxCostUsd) {
|
|
2155
|
-
results.push(budgetExhaustedResult(item.task
|
|
3787
|
+
results.push(budgetExhaustedResult(loaded, item.task, item.target, item.repeatIndex, spentUsd, maxCostUsd));
|
|
2156
3788
|
continue;
|
|
2157
3789
|
}
|
|
2158
|
-
const
|
|
2159
|
-
|
|
2160
|
-
|
|
3790
|
+
const completed = await runMatrixItem(
|
|
3791
|
+
loaded,
|
|
3792
|
+
item.task,
|
|
3793
|
+
item.target,
|
|
3794
|
+
item.repeatIndex,
|
|
3795
|
+
options.clioEntry,
|
|
3796
|
+
options.freshWorkspaces === true,
|
|
3797
|
+
options.tempCopy
|
|
3798
|
+
);
|
|
3799
|
+
spentUsd += resultCostUsd(completed.result);
|
|
3800
|
+
results.push(completed.result);
|
|
3801
|
+
servingObservations.push(completed.serving);
|
|
2161
3802
|
}
|
|
2162
|
-
|
|
3803
|
+
const serving = await evalServingConfiguration(loaded.suite.matrix.targets, servingObservations);
|
|
3804
|
+
return buildArtifact(loaded, evalId, results, options.clioEntry, serving);
|
|
2163
3805
|
}
|
|
2164
3806
|
function resultCostUsd(result) {
|
|
2165
3807
|
const value = result.metrics["cost.usd"];
|
|
2166
3808
|
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0;
|
|
2167
3809
|
}
|
|
2168
|
-
function budgetExhaustedResult(
|
|
2169
|
-
|
|
3810
|
+
function budgetExhaustedResult(loaded, task, target, repeatIndex, spentUsd, maxCostUsd) {
|
|
3811
|
+
const result = {
|
|
2170
3812
|
assignmentId: null,
|
|
2171
3813
|
terminalReceiptDigest: null,
|
|
2172
|
-
taskId,
|
|
3814
|
+
taskId: task.id,
|
|
2173
3815
|
repeatIndex,
|
|
2174
3816
|
target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
|
|
2175
3817
|
pass: false,
|
|
@@ -2184,21 +3826,36 @@ function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd
|
|
|
2184
3826
|
error: `matrix cost budget exhausted: spent $${spentUsd.toFixed(4)} of max $${maxCostUsd.toFixed(4)} before this item`
|
|
2185
3827
|
}
|
|
2186
3828
|
};
|
|
3829
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(result, emptyEvalTrackedMetrics());
|
|
3830
|
+
attachBehavioralResult(result, task);
|
|
3831
|
+
attachExecutionEnvelope(result, task, target, loaded.baseDir, null, emptyLedgerSnapshot());
|
|
3832
|
+
return result;
|
|
2187
3833
|
}
|
|
2188
|
-
async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
3834
|
+
async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry, freshWorkspace, tempCopy) {
|
|
2189
3835
|
let workspace = null;
|
|
2190
|
-
|
|
3836
|
+
let receipt = null;
|
|
3837
|
+
let runnerWallTimeMs = 0;
|
|
3838
|
+
let executionObservation;
|
|
3839
|
+
const stateDir = await mkdtemp3(resolve7(tempCopy?.tempRoot ?? tmpdir3(), "clio-eval-state-"));
|
|
2191
3840
|
try {
|
|
2192
|
-
workspace = await prepareWorkspace(loaded.baseDir, task);
|
|
3841
|
+
workspace = await prepareWorkspace(loaded.baseDir, task, freshWorkspace, tempCopy);
|
|
2193
3842
|
const setup = await runCommandVerifiers(task.workspace.setup ?? [], workspace.dir, task.timeoutMs);
|
|
2194
3843
|
if (!setup.pass) throw new EvalWorkspaceSetupError(setup.exitCode, setup.stderr);
|
|
2195
3844
|
const runner = await runTaskRunner(task, target, workspace.dir, clioEntry, {
|
|
2196
3845
|
CLIO_CODER_STATE_DIR: stateDir,
|
|
2197
3846
|
CLIO_CODER_ENTRY: clioEntry
|
|
2198
3847
|
});
|
|
3848
|
+
const runnerStdoutFile = resolve7(stateDir, "eval-runner-output.jsonl");
|
|
3849
|
+
await writeFile(runnerStdoutFile, runner.stdout, "utf8");
|
|
3850
|
+
receipt = runner.receipt ?? null;
|
|
3851
|
+
runnerWallTimeMs = runner.wallTimeMs;
|
|
2199
3852
|
const patch = collectPatchMetrics(workspace.dir);
|
|
2200
3853
|
const receiptExitCode = runner.exitCode;
|
|
2201
3854
|
const journalMetrics = invariantMetrics(stateDir, receiptExitCode);
|
|
3855
|
+
const measurement = await measureTaskOutcome(task, workspace.dir, {
|
|
3856
|
+
CLIO_EVAL_RUNNER_STDOUT_FILE: runnerStdoutFile
|
|
3857
|
+
});
|
|
3858
|
+
executionObservation = measurement.executionObservation;
|
|
2202
3859
|
const metrics = {
|
|
2203
3860
|
...zeroToolCallMetrics(),
|
|
2204
3861
|
...collectContextMetrics(workspace.dir),
|
|
@@ -2214,15 +3871,16 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2214
3871
|
"patch.testFilesModified": patch.testFilesModified,
|
|
2215
3872
|
"result.pass": runner.exitCode === 0,
|
|
2216
3873
|
"result.failureClass": runner.exitCode === 0 ? null : "runner_failed",
|
|
2217
|
-
...
|
|
3874
|
+
...measurement.metrics
|
|
2218
3875
|
};
|
|
2219
3876
|
const verifier = await runVerifiers(task, workspace.dir, metrics);
|
|
2220
|
-
const
|
|
2221
|
-
const
|
|
3877
|
+
const graderFailed = metrics["task.solved"] === false;
|
|
3878
|
+
const pass = runner.exitCode === 0 && verifier.pass && !graderFailed;
|
|
3879
|
+
const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : !verifier.pass ? verifier.failureClass : "grader_failed";
|
|
2222
3880
|
metrics["verifier.exitCode"] = verifier.exitCode;
|
|
2223
3881
|
metrics["result.pass"] = pass;
|
|
2224
3882
|
metrics["result.failureClass"] = failureClass;
|
|
2225
|
-
|
|
3883
|
+
const result = {
|
|
2226
3884
|
assignmentId: runner.assignmentId,
|
|
2227
3885
|
terminalReceiptDigest: runner.terminalReceiptDigest,
|
|
2228
3886
|
taskId: task.id,
|
|
@@ -2233,13 +3891,30 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2233
3891
|
metrics,
|
|
2234
3892
|
artifacts: {
|
|
2235
3893
|
...runner.artifacts,
|
|
3894
|
+
workspace: workspace.dir,
|
|
2236
3895
|
...verifier.stdout.length > 0 ? { verifierStdout: verifier.stdout } : {},
|
|
2237
3896
|
...verifier.stderr.length > 0 ? { verifierStderr: verifier.stderr } : {}
|
|
2238
3897
|
}
|
|
2239
3898
|
};
|
|
3899
|
+
const snapshot = await readEvalLedgerSnapshot(stateDir);
|
|
3900
|
+
const ledgerEntries = [...snapshot.entries, ...runner.ledgerEntries ?? []];
|
|
3901
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(
|
|
3902
|
+
result,
|
|
3903
|
+
buildEvalTrackedMetrics({
|
|
3904
|
+
ledgerEntries,
|
|
3905
|
+
receipt: receipt ?? null,
|
|
3906
|
+
fallbackWallClockMs: runner.wallTimeMs
|
|
3907
|
+
})
|
|
3908
|
+
);
|
|
3909
|
+
attachBehavioralResult(result, task);
|
|
3910
|
+
attachExecutionEnvelope(result, task, target, workspace.dir, receipt ?? null, snapshot, executionObservation);
|
|
3911
|
+
return {
|
|
3912
|
+
result,
|
|
3913
|
+
serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
|
|
3914
|
+
};
|
|
2240
3915
|
} catch (error) {
|
|
2241
3916
|
const failureClass = error instanceof EvalWorkspaceSetupError ? "setup_failed" : "command_error";
|
|
2242
|
-
|
|
3917
|
+
const result = {
|
|
2243
3918
|
assignmentId: null,
|
|
2244
3919
|
terminalReceiptDigest: null,
|
|
2245
3920
|
taskId: task.id,
|
|
@@ -2253,13 +3928,61 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2253
3928
|
"verifier.exitCode": 1,
|
|
2254
3929
|
"latency.wallMs": 0
|
|
2255
3930
|
},
|
|
2256
|
-
artifacts: {
|
|
3931
|
+
artifacts: {
|
|
3932
|
+
error: error instanceof Error ? error.message : String(error),
|
|
3933
|
+
...workspace === null ? {} : { workspace: workspace.dir }
|
|
3934
|
+
}
|
|
3935
|
+
};
|
|
3936
|
+
const snapshot = await readEvalLedgerSnapshot(stateDir);
|
|
3937
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(
|
|
3938
|
+
result,
|
|
3939
|
+
buildEvalTrackedMetrics({
|
|
3940
|
+
ledgerEntries: snapshot.entries,
|
|
3941
|
+
receipt: receipt ?? null,
|
|
3942
|
+
fallbackWallClockMs: runnerWallTimeMs
|
|
3943
|
+
})
|
|
3944
|
+
);
|
|
3945
|
+
attachBehavioralResult(result, task);
|
|
3946
|
+
attachExecutionEnvelope(
|
|
3947
|
+
result,
|
|
3948
|
+
task,
|
|
3949
|
+
target,
|
|
3950
|
+
workspace?.dir ?? loaded.baseDir,
|
|
3951
|
+
receipt ?? null,
|
|
3952
|
+
snapshot,
|
|
3953
|
+
executionObservation
|
|
3954
|
+
);
|
|
3955
|
+
return {
|
|
3956
|
+
result,
|
|
3957
|
+
serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
|
|
2257
3958
|
};
|
|
2258
3959
|
} finally {
|
|
2259
|
-
|
|
2260
|
-
|
|
3960
|
+
try {
|
|
3961
|
+
await workspace?.cleanup();
|
|
3962
|
+
} finally {
|
|
3963
|
+
await rm3(stateDir, { recursive: true, force: true });
|
|
3964
|
+
}
|
|
2261
3965
|
}
|
|
2262
3966
|
}
|
|
3967
|
+
function attachBehavioralResult(result, task) {
|
|
3968
|
+
if (task.behavioral === void 0 || result.verdict === void 0) return;
|
|
3969
|
+
result.behavioral = adaptSuiteV2ResultToBehaviorV1(result, result.verdict, task.behavioral);
|
|
3970
|
+
result.behavioralMetrics = buildEvalBehaviorMetricsV1(result, task.behavioral.execution.subject.role);
|
|
3971
|
+
}
|
|
3972
|
+
function attachExecutionEnvelope(result, task, target, cwd, receipt, ledger, observation) {
|
|
3973
|
+
if (task.behavioral === void 0) return;
|
|
3974
|
+
result.executionEnvelope = buildEvalExecutionEnvelopeV1({
|
|
3975
|
+
task,
|
|
3976
|
+
target,
|
|
3977
|
+
cwd,
|
|
3978
|
+
receipt,
|
|
3979
|
+
ledger,
|
|
3980
|
+
...observation === void 0 ? {} : { observation }
|
|
3981
|
+
});
|
|
3982
|
+
}
|
|
3983
|
+
function emptyLedgerSnapshot() {
|
|
3984
|
+
return { entries: [], compiledPromptHashes: [], promptManifests: [], contextSnapshots: [] };
|
|
3985
|
+
}
|
|
2263
3986
|
function invariantMetrics(stateDir, runnerExitCode) {
|
|
2264
3987
|
const journal = readRunJournal(stateDir);
|
|
2265
3988
|
return {
|
|
@@ -2270,23 +3993,80 @@ function invariantMetrics(stateDir, runnerExitCode) {
|
|
|
2270
3993
|
...writeBoundaryInvariantMetrics(stateDir)
|
|
2271
3994
|
};
|
|
2272
3995
|
}
|
|
2273
|
-
async function prepareWorkspace(baseDir, task) {
|
|
3996
|
+
async function prepareWorkspace(baseDir, task, freshWorkspace, tempCopy) {
|
|
3997
|
+
if (task.workspace.kind === "local" && freshWorkspace) {
|
|
3998
|
+
return prepareTempCopyWorkspace(baseDir, { ...task.workspace, kind: "temp-copy" }, tempCopy);
|
|
3999
|
+
}
|
|
2274
4000
|
if (task.workspace.kind === "local") return prepareLocalWorkspace(baseDir, task.workspace);
|
|
2275
4001
|
if (task.workspace.kind === "git") return prepareGitWorkspace(task.workspace);
|
|
2276
|
-
return prepareTempCopyWorkspace(baseDir, task.workspace);
|
|
4002
|
+
return prepareTempCopyWorkspace(baseDir, task.workspace, tempCopy);
|
|
2277
4003
|
}
|
|
2278
4004
|
async function runTaskRunner(task, target, cwd, clioEntry, env) {
|
|
2279
4005
|
if (task.runner.kind === "external-command") return runExternalCommandRunner(task.runner, cwd, task.timeoutMs, env);
|
|
2280
4006
|
if (task.runner.kind === "context-index") return runContextIndexRunner(cwd, clioEntry, task.timeoutMs, target, env);
|
|
2281
4007
|
if (task.runner.kind === "context-init")
|
|
2282
4008
|
return runContextInitRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
|
|
2283
|
-
return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
|
|
4009
|
+
return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env, task.metrics.readObservation);
|
|
2284
4010
|
}
|
|
2285
|
-
async function measureTaskOutcome(task, cwd) {
|
|
4011
|
+
async function measureTaskOutcome(task, cwd, env) {
|
|
2286
4012
|
const commands = task.verify.measure ?? [];
|
|
2287
|
-
if (commands.length === 0) return {};
|
|
2288
|
-
const result = await runCommandVerifiers(commands, cwd, task.timeoutMs);
|
|
2289
|
-
|
|
4013
|
+
if (commands.length === 0) return { metrics: {} };
|
|
4014
|
+
const result = await runCommandVerifiers(commands, cwd, task.timeoutMs, env);
|
|
4015
|
+
const behavioral = graderBehaviorMeasurement(result.stdout);
|
|
4016
|
+
return {
|
|
4017
|
+
metrics: {
|
|
4018
|
+
"task.exitCode": result.exitCode,
|
|
4019
|
+
"task.solved": result.exitCode === 0,
|
|
4020
|
+
...behavioral.metrics
|
|
4021
|
+
},
|
|
4022
|
+
...behavioral.executionObservation === void 0 ? {} : { executionObservation: behavioral.executionObservation }
|
|
4023
|
+
};
|
|
4024
|
+
}
|
|
4025
|
+
function graderBehaviorMeasurement(stdout) {
|
|
4026
|
+
const metrics = {};
|
|
4027
|
+
let executionObservation;
|
|
4028
|
+
for (const line of stdout.split(/\r?\n/u)) {
|
|
4029
|
+
if (line.trim().length === 0) continue;
|
|
4030
|
+
let value;
|
|
4031
|
+
try {
|
|
4032
|
+
value = JSON.parse(line);
|
|
4033
|
+
} catch {
|
|
4034
|
+
continue;
|
|
4035
|
+
}
|
|
4036
|
+
if (!isRecord10(value)) continue;
|
|
4037
|
+
if (value.schema === "clio.eval.measure.v1" && isRecord10(value.metrics)) {
|
|
4038
|
+
for (const [key, metric] of Object.entries(value.metrics)) {
|
|
4039
|
+
if (key !== "claims.unsupported" && key !== "completion.reported") continue;
|
|
4040
|
+
if (typeof metric === "boolean" || typeof metric === "number" && Number.isFinite(metric)) metrics[key] = metric;
|
|
4041
|
+
}
|
|
4042
|
+
}
|
|
4043
|
+
if (value.schema === "clio.eval.execution-observation.v1") {
|
|
4044
|
+
executionObservation = parseExecutionObservation(value);
|
|
4045
|
+
}
|
|
4046
|
+
}
|
|
4047
|
+
return { metrics, ...executionObservation === void 0 ? {} : { executionObservation } };
|
|
4048
|
+
}
|
|
4049
|
+
function parseExecutionObservation(value) {
|
|
4050
|
+
const policies = isRecord10(value.policyHashes) ? value.policyHashes : {};
|
|
4051
|
+
const project = isRecord10(value.projectContext) ? value.projectContext : null;
|
|
4052
|
+
return {
|
|
4053
|
+
compositionHash: nullableDigest2(value.compositionHash),
|
|
4054
|
+
target: nullableString2(value.target),
|
|
4055
|
+
wireModel: nullableString2(value.wireModel),
|
|
4056
|
+
runtime: nullableString2(value.runtime),
|
|
4057
|
+
thinkingLevel: nullableString2(value.thinkingLevel),
|
|
4058
|
+
toolSignature: nullableDigest2(value.toolSignature),
|
|
4059
|
+
autonomy: nullableString2(value.autonomy),
|
|
4060
|
+
policyHashes: { rulePack: nullableDigest2(policies.rulePack), project: nullableDigest2(policies.project) },
|
|
4061
|
+
projectContext: project === null ? null : {
|
|
4062
|
+
tier: nullableString2(project.tier),
|
|
4063
|
+
contentHash: nullableDigest2(project.contentHash),
|
|
4064
|
+
chars: nullableNonNegativeInteger(project.chars),
|
|
4065
|
+
sections: stringArray(project.sections),
|
|
4066
|
+
rulesApplied: stringArray(project.rulesApplied),
|
|
4067
|
+
operatorProfileApplied: typeof project.operatorProfileApplied === "boolean" ? project.operatorProfileApplied : null
|
|
4068
|
+
}
|
|
4069
|
+
};
|
|
2290
4070
|
}
|
|
2291
4071
|
async function runVerifiers(task, cwd, metrics) {
|
|
2292
4072
|
const commandResult = await runCommandVerifiers(task.verify.commands ?? [], cwd, task.timeoutMs);
|
|
@@ -2322,7 +4102,7 @@ async function runVerifiers(task, cwd, metrics) {
|
|
|
2322
4102
|
}
|
|
2323
4103
|
return { pass: true, exitCode: 0, failureClass: null, stdout: commandResult.stdout, stderr: commandResult.stderr };
|
|
2324
4104
|
}
|
|
2325
|
-
function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
4105
|
+
function buildArtifact(loaded, evalId, results, clioEntry, servingConfiguration) {
|
|
2326
4106
|
const passed = results.filter((result) => result.pass).length;
|
|
2327
4107
|
return {
|
|
2328
4108
|
version: 4,
|
|
@@ -2330,7 +4110,11 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
|
2330
4110
|
suite: { id: loaded.suite.suite.id, hash: loaded.hash },
|
|
2331
4111
|
clio: evalClioProvenance({ entry: clioEntry }),
|
|
2332
4112
|
environment: evalEnvironmentProvenance(),
|
|
2333
|
-
matrix:
|
|
4113
|
+
matrix: {
|
|
4114
|
+
...artifactMatrixIdentity(loaded.suite.matrix.targets),
|
|
4115
|
+
...loaded.suite.matrix.dimensions === void 0 ? {} : { dimensions: loaded.suite.matrix.dimensions }
|
|
4116
|
+
},
|
|
4117
|
+
servingConfiguration,
|
|
2334
4118
|
summary: {
|
|
2335
4119
|
runs: results.length,
|
|
2336
4120
|
passed,
|
|
@@ -2339,26 +4123,57 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
|
2339
4123
|
tokens: tokenAccountingFrom(results),
|
|
2340
4124
|
wallTimeMs: results.reduce((sum2, result) => sum2 + wallTimeMetric(result.metrics), 0)
|
|
2341
4125
|
},
|
|
4126
|
+
aggregates: aggregateEvalVerdicts(
|
|
4127
|
+
results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict])
|
|
4128
|
+
),
|
|
2342
4129
|
results
|
|
2343
4130
|
};
|
|
2344
4131
|
}
|
|
2345
4132
|
function assertionMessage(assertion, actual) {
|
|
2346
4133
|
return `assertion failed: ${assertion.metric} ${assertion.op} ${String(assertion.value)} (actual ${JSON.stringify(actual)})`;
|
|
2347
4134
|
}
|
|
4135
|
+
function isRecord10(value) {
|
|
4136
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
4137
|
+
}
|
|
4138
|
+
function nullableString2(value) {
|
|
4139
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
4140
|
+
}
|
|
4141
|
+
function nullableDigest2(value) {
|
|
4142
|
+
return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
|
|
4143
|
+
}
|
|
4144
|
+
function nullableNonNegativeInteger(value) {
|
|
4145
|
+
return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : null;
|
|
4146
|
+
}
|
|
4147
|
+
function stringArray(value) {
|
|
4148
|
+
return Array.isArray(value) ? value.filter((entry) => typeof entry === "string") : [];
|
|
4149
|
+
}
|
|
2348
4150
|
|
|
2349
4151
|
// src/cli/eval.ts
|
|
2350
4152
|
var HELP = `clio-coder eval <command>
|
|
2351
4153
|
|
|
2352
4154
|
Commands:
|
|
2353
4155
|
clio-coder eval validate --suite <suite.yaml>
|
|
2354
|
-
|
|
2355
|
-
|
|
4156
|
+
clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
|
|
4157
|
+
clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
|
|
2356
4158
|
clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
|
|
2357
|
-
|
|
4159
|
+
clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
|
|
2358
4160
|
clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
|
|
4161
|
+
clio-coder eval inventory --json
|
|
4162
|
+
|
|
4163
|
+
inventory is the fixed machine-readable read a GUI host may run. Unlike report
|
|
4164
|
+
and compare it names no eval id, so the process it starts cannot be steered to a
|
|
4165
|
+
different report or a wider window. It carries each stored report's identity,
|
|
4166
|
+
provenance, serving facts, accounting, and per-scenario outcomes, and none of
|
|
4167
|
+
the runner attachments a report holds.
|
|
2359
4168
|
`;
|
|
2360
4169
|
function parseEvalArgs(args) {
|
|
2361
|
-
const parsed = {
|
|
4170
|
+
const parsed = {
|
|
4171
|
+
repeat: 1,
|
|
4172
|
+
compareIds: [],
|
|
4173
|
+
format: "text",
|
|
4174
|
+
allowConfigDrift: false,
|
|
4175
|
+
help: false
|
|
4176
|
+
};
|
|
2362
4177
|
for (let index = 0; index < args.length; index += 1) {
|
|
2363
4178
|
const arg = args[index];
|
|
2364
4179
|
if (arg === void 0) continue;
|
|
@@ -2417,6 +4232,11 @@ function parseEvalArgs(args) {
|
|
|
2417
4232
|
index += 1;
|
|
2418
4233
|
continue;
|
|
2419
4234
|
}
|
|
4235
|
+
if (arg === "--trials") {
|
|
4236
|
+
parsed.trials = positiveInteger(requiredValue(args, index, "--trials"), "--trials");
|
|
4237
|
+
index += 1;
|
|
4238
|
+
continue;
|
|
4239
|
+
}
|
|
2420
4240
|
throw new Error(`unknown eval run argument: ${arg}`);
|
|
2421
4241
|
}
|
|
2422
4242
|
if (parsed.command === "report") {
|
|
@@ -2432,6 +4252,20 @@ function parseEvalArgs(args) {
|
|
|
2432
4252
|
throw new Error(`unexpected eval report argument: ${arg}`);
|
|
2433
4253
|
}
|
|
2434
4254
|
if (parsed.command === "compare") {
|
|
4255
|
+
if (arg === "--format") {
|
|
4256
|
+
parsed.format = comparisonFormat(requiredValue(args, index, "--format"));
|
|
4257
|
+
index += 1;
|
|
4258
|
+
continue;
|
|
4259
|
+
}
|
|
4260
|
+
if (arg === "--metric") {
|
|
4261
|
+
parsed.metric = requiredValue(args, index, "--metric");
|
|
4262
|
+
index += 1;
|
|
4263
|
+
continue;
|
|
4264
|
+
}
|
|
4265
|
+
if (arg === "--allow-config-drift") {
|
|
4266
|
+
parsed.allowConfigDrift = true;
|
|
4267
|
+
continue;
|
|
4268
|
+
}
|
|
2435
4269
|
if (!arg.startsWith("-")) {
|
|
2436
4270
|
parsed.compareIds.push(arg);
|
|
2437
4271
|
continue;
|
|
@@ -2472,6 +4306,10 @@ function parseEvalArgs(args) {
|
|
|
2472
4306
|
return parsed;
|
|
2473
4307
|
}
|
|
2474
4308
|
async function runEvalCommand(args) {
|
|
4309
|
+
if (args[0] === "inventory") {
|
|
4310
|
+
const { runEvalInventory } = await import("./eval-inventory-SXH7PDKX.js");
|
|
4311
|
+
return runEvalInventory(args.slice(1));
|
|
4312
|
+
}
|
|
2475
4313
|
let parsed;
|
|
2476
4314
|
try {
|
|
2477
4315
|
parsed = parseEvalArgs(args);
|
|
@@ -2504,13 +4342,19 @@ async function runEvalValidate(parsed) {
|
|
|
2504
4342
|
}
|
|
2505
4343
|
async function runEvalRun(parsed) {
|
|
2506
4344
|
try {
|
|
2507
|
-
const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.repeat);
|
|
4345
|
+
const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.trials ?? parsed.repeat);
|
|
2508
4346
|
const resolveOptions = {};
|
|
2509
4347
|
if (parsed.target !== void 0) resolveOptions.target = parsed.target;
|
|
2510
4348
|
if (parsed.model !== void 0) resolveOptions.model = parsed.model;
|
|
2511
|
-
const suite = resolveSuiteForRun(loaded.suite,
|
|
2512
|
-
|
|
2513
|
-
|
|
4349
|
+
const suite = resolveSuiteForRun(loaded.suite, {
|
|
4350
|
+
...resolveOptions,
|
|
4351
|
+
...parsed.trials ? { trials: parsed.trials } : {}
|
|
4352
|
+
});
|
|
4353
|
+
const clioEntry = resolve8(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
|
|
4354
|
+
const artifact = await runEvalSuiteV2(
|
|
4355
|
+
{ ...loaded, suite },
|
|
4356
|
+
{ clioEntry, freshWorkspaces: parsed.trials !== void 0 }
|
|
4357
|
+
);
|
|
2514
4358
|
const artifactPath = await writeEvalArtifactV4(clioDataDir(), artifact, parsed.out);
|
|
2515
4359
|
process.stdout.write(`${renderEvalTextReportV4(artifact)}artifact: ${artifactPath}
|
|
2516
4360
|
`);
|
|
@@ -2520,6 +4364,11 @@ async function runEvalRun(parsed) {
|
|
|
2520
4364
|
`);
|
|
2521
4365
|
for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
|
|
2522
4366
|
}
|
|
4367
|
+
if (gate !== null && gate.informational.length > 0) {
|
|
4368
|
+
process.stdout.write(`informational budgets: ${gate.informational.length} notice
|
|
4369
|
+
`);
|
|
4370
|
+
for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
|
|
4371
|
+
}
|
|
2523
4372
|
return artifact.summary.failed === 0 && (gate === null || gate.pass) ? 0 : 1;
|
|
2524
4373
|
} catch (error) {
|
|
2525
4374
|
return handleEvalLoadError(error, 1);
|
|
@@ -2543,8 +4392,12 @@ async function runEvalCompareCommand(parsed) {
|
|
|
2543
4392
|
const dataDir = clioDataDir();
|
|
2544
4393
|
const baseline = await loadEvalArtifactV4(dataDir, baselineEvalId);
|
|
2545
4394
|
const candidate = await loadEvalArtifactV4(dataDir, candidateEvalId);
|
|
2546
|
-
|
|
2547
|
-
|
|
4395
|
+
const summary = compareEvalArtifactsV4(baseline, candidate, {
|
|
4396
|
+
allowConfigDrift: parsed.allowConfigDrift,
|
|
4397
|
+
...parsed.metric === void 0 ? {} : { metric: parsed.metric }
|
|
4398
|
+
});
|
|
4399
|
+
process.stdout.write(renderEvalComparisonReportV1(summary, parsed.format));
|
|
4400
|
+
return summary.hardGate.pass ? 0 : 1;
|
|
2548
4401
|
} catch (error) {
|
|
2549
4402
|
printError(error instanceof Error ? error.message : String(error));
|
|
2550
4403
|
return error instanceof InvalidIdError ? 2 : 1;
|
|
@@ -2554,27 +4407,46 @@ async function runEvalGateCommand(parsed) {
|
|
|
2554
4407
|
try {
|
|
2555
4408
|
const dataDir = clioDataDir();
|
|
2556
4409
|
const candidate = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
|
|
2557
|
-
await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
|
|
2558
|
-
const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }] } : loadThresholds(parsed.thresholds);
|
|
4410
|
+
const baseline = await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
|
|
4411
|
+
const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }], informational: [] } : loadThresholds(parsed.thresholds);
|
|
2559
4412
|
const gate = evaluateGate(candidate, thresholds);
|
|
2560
|
-
|
|
4413
|
+
const comparison = compareEvalArtifactsV4(baseline, candidate);
|
|
4414
|
+
if (gate.informational.length > 0) {
|
|
4415
|
+
process.stdout.write(`informational budgets: ${gate.informational.length} notice
|
|
4416
|
+
`);
|
|
4417
|
+
for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
|
|
4418
|
+
}
|
|
4419
|
+
if (gate.pass && comparison.hardGate.pass) {
|
|
2561
4420
|
process.stdout.write("gate: pass\n");
|
|
2562
4421
|
return 0;
|
|
2563
4422
|
}
|
|
2564
|
-
|
|
4423
|
+
const failureCount = gate.failures.length + comparison.hardGate.failures.length + comparison.hardGate.envelopeFailures.length;
|
|
4424
|
+
process.stdout.write(`gate: fail (${failureCount} hard failure)
|
|
2565
4425
|
`);
|
|
2566
4426
|
for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
|
|
4427
|
+
for (const failure of comparison.hardGate.failures) {
|
|
4428
|
+
process.stdout.write(
|
|
4429
|
+
` ${failure.metric} [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: ${failure.change} (hard behavioral gate)
|
|
4430
|
+
`
|
|
4431
|
+
);
|
|
4432
|
+
}
|
|
4433
|
+
for (const failure of comparison.hardGate.envelopeFailures) {
|
|
4434
|
+
process.stdout.write(
|
|
4435
|
+
` execution envelope [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: incomparable fields ${failure.fields.join(", ")}
|
|
4436
|
+
`
|
|
4437
|
+
);
|
|
4438
|
+
}
|
|
2567
4439
|
return 1;
|
|
2568
4440
|
} catch (error) {
|
|
2569
4441
|
printError(error instanceof Error ? error.message : String(error));
|
|
2570
4442
|
return error instanceof InvalidIdError ? 2 : 1;
|
|
2571
4443
|
}
|
|
2572
4444
|
}
|
|
2573
|
-
function renderArtifactReport(artifact,
|
|
2574
|
-
if (
|
|
2575
|
-
if (
|
|
2576
|
-
if (
|
|
2577
|
-
if (
|
|
4445
|
+
function renderArtifactReport(artifact, format2, _dataDir) {
|
|
4446
|
+
if (format2 === "json") return renderEvalJsonReportV4(artifact);
|
|
4447
|
+
if (format2 === "md") return renderEvalMarkdownReportV4(artifact);
|
|
4448
|
+
if (format2 === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
|
|
4449
|
+
if (format2 === "junit") return renderEvalJunitReportV4(artifact);
|
|
2578
4450
|
return renderEvalTextReportV4(artifact);
|
|
2579
4451
|
}
|
|
2580
4452
|
function handleEvalLoadError(error, fallback = 2) {
|
|
@@ -2603,7 +4475,11 @@ function reportFormat(value) {
|
|
|
2603
4475
|
if (value === "text" || value === "json" || value === "md" || value === "swe-jsonl" || value === "junit") return value;
|
|
2604
4476
|
throw new Error("--format must be text, json, md, swe-jsonl, or junit");
|
|
2605
4477
|
}
|
|
4478
|
+
function comparisonFormat(value) {
|
|
4479
|
+
if (value === "text" || value === "json" || value === "md" || value === "junit") return value;
|
|
4480
|
+
throw new Error("eval compare --format must be text, json, md, or junit");
|
|
4481
|
+
}
|
|
2606
4482
|
export {
|
|
2607
4483
|
runEvalCommand
|
|
2608
4484
|
};
|
|
2609
|
-
//# sourceMappingURL=eval-
|
|
4485
|
+
//# sourceMappingURL=eval-TFBYQH4H.js.map
|