@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -1,29 +1,69 @@
|
|
|
1
1
|
import { createRequire as __clioCreateRequire } from "node:module"; const require = __clioCreateRequire(import.meta.url);
|
|
2
2
|
import {
|
|
3
|
+
discoverAgentRecipes
|
|
4
|
+
} from "./chunk-HCBCAYZU.js";
|
|
5
|
+
import "./chunk-AD7Y7STJ.js";
|
|
6
|
+
import "./chunk-AMKHQW3C.js";
|
|
7
|
+
import {
|
|
8
|
+
loadFragments,
|
|
3
9
|
renderCodewikiDigest
|
|
4
|
-
} from "./chunk-
|
|
10
|
+
} from "./chunk-47CMYGET.js";
|
|
5
11
|
import {
|
|
12
|
+
EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
|
|
13
|
+
EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
14
|
+
EVAL_TRACKED_METRIC_NAMES,
|
|
15
|
+
EVAL_VERDICT_SCHEMA_V1,
|
|
6
16
|
EvalTaskFileError,
|
|
7
17
|
TRUST_STATUS_AXES,
|
|
8
18
|
adaptRunReceiptTrustStatus,
|
|
19
|
+
assertComparableTrackedMetricSources,
|
|
20
|
+
assertEvalBehaviorReferencesVerdictV1,
|
|
21
|
+
buildEvalBehaviorMetricsV1,
|
|
9
22
|
createEvalId,
|
|
10
23
|
evalClioProvenance,
|
|
11
24
|
evalEnvironmentProvenance,
|
|
25
|
+
evalHarnessMetricsFromReceipt,
|
|
26
|
+
evalServingConfiguration,
|
|
27
|
+
evalServingObservationFrom,
|
|
12
28
|
formatTrustSummary,
|
|
13
29
|
inspectRunReceiptTrustStatus,
|
|
30
|
+
judgeEvalBehaviorV1,
|
|
14
31
|
loadEvalArtifactV4,
|
|
15
32
|
loadEvalTaskFile,
|
|
33
|
+
parseEvalBehaviorScenarioV1,
|
|
34
|
+
parseEvalExecutionMatrixDimensionsV1,
|
|
35
|
+
parseEvalVerdictEnvelopeV1,
|
|
36
|
+
renderEvalServingConfiguration,
|
|
37
|
+
sameEvalServingConfiguration,
|
|
16
38
|
summarizeTrustStatus,
|
|
17
39
|
verifyReceiptIntegrity,
|
|
18
40
|
writeEvalArtifactV4
|
|
19
|
-
} from "./chunk-
|
|
41
|
+
} from "./chunk-MLOK6ZOS.js";
|
|
42
|
+
import {
|
|
43
|
+
listSessionLedgerRefs,
|
|
44
|
+
parseSessionEntries
|
|
45
|
+
} from "./chunk-LXPJXFM5.js";
|
|
46
|
+
import "./chunk-3DPEIQKN.js";
|
|
47
|
+
import "./chunk-W6GROXXM.js";
|
|
48
|
+
import "./chunk-RKKLTLYB.js";
|
|
49
|
+
import "./chunk-5QKCQQ3E.js";
|
|
20
50
|
import {
|
|
21
51
|
shellQuote
|
|
22
52
|
} from "./chunk-TXOTCRLG.js";
|
|
23
|
-
import "./chunk-
|
|
53
|
+
import "./chunk-HVDIIIQW.js";
|
|
54
|
+
import "./chunk-VPTUJU4P.js";
|
|
55
|
+
import "./chunk-2JDWVJND.js";
|
|
24
56
|
import "./chunk-H7IXIC72.js";
|
|
25
|
-
import
|
|
57
|
+
import {
|
|
58
|
+
createSafetyPolicyEngine
|
|
59
|
+
} from "./chunk-N3PBVRTZ.js";
|
|
60
|
+
import "./chunk-A2NJGIB3.js";
|
|
61
|
+
import {
|
|
62
|
+
agentSpecFingerprint,
|
|
63
|
+
normalizeAgentSpec
|
|
64
|
+
} from "./chunk-S4COXYBG.js";
|
|
26
65
|
import "./chunk-MV3K5QF2.js";
|
|
66
|
+
import "./chunk-RAPCMZL4.js";
|
|
27
67
|
import "./chunk-UL3WSD3F.js";
|
|
28
68
|
import "./chunk-ECH6PKUQ.js";
|
|
29
69
|
import "./chunk-CGKSTWHD.js";
|
|
@@ -38,15 +78,29 @@ import {
|
|
|
38
78
|
enumerateWorkspaceFiles
|
|
39
79
|
} from "./chunk-33YXPOE3.js";
|
|
40
80
|
import "./chunk-7CR24IG7.js";
|
|
81
|
+
import "./chunk-XPLRXC72.js";
|
|
41
82
|
import "./chunk-IFBNV6H6.js";
|
|
42
83
|
import {
|
|
43
84
|
printError
|
|
44
85
|
} from "./chunk-XK56QHLX.js";
|
|
45
86
|
import "./chunk-5TSRNF4G.js";
|
|
87
|
+
import "./chunk-CFGTUFWB.js";
|
|
88
|
+
import {
|
|
89
|
+
extractReasoningTokens
|
|
90
|
+
} from "./chunk-AEYBF3TB.js";
|
|
91
|
+
import "./chunk-IHXBNWMM.js";
|
|
92
|
+
import "./chunk-B5CSFE7B.js";
|
|
93
|
+
import "./chunk-PNY46YEY.js";
|
|
94
|
+
import "./chunk-FQ4SKYE4.js";
|
|
95
|
+
import "./chunk-RKRLDWD3.js";
|
|
46
96
|
import {
|
|
47
97
|
InvalidIdError
|
|
48
|
-
} from "./chunk-
|
|
98
|
+
} from "./chunk-KV2AOLDF.js";
|
|
99
|
+
import "./chunk-6EJMN2Y3.js";
|
|
49
100
|
import "./chunk-IWHMRKLL.js";
|
|
101
|
+
import "./chunk-XDOQXGFO.js";
|
|
102
|
+
import "./chunk-LL4KHSZI.js";
|
|
103
|
+
import "./chunk-4ZG3XFUR.js";
|
|
50
104
|
import "./chunk-EQ63NRB7.js";
|
|
51
105
|
import "./chunk-SST6Z5JA.js";
|
|
52
106
|
import "./chunk-IKCO5N3L.js";
|
|
@@ -67,31 +121,585 @@ import {
|
|
|
67
121
|
|
|
68
122
|
// src/cli/eval.ts
|
|
69
123
|
init_esm_shims();
|
|
70
|
-
import { resolve as
|
|
124
|
+
import { resolve as resolve8 } from "node:path";
|
|
71
125
|
|
|
72
126
|
// src/domains/eval/compare/compare.ts
|
|
73
127
|
init_esm_shims();
|
|
74
|
-
|
|
128
|
+
|
|
129
|
+
// src/domains/eval/metrics/aggregate.ts
|
|
130
|
+
init_esm_shims();
|
|
131
|
+
function aggregateEvalVerdicts(verdicts) {
|
|
132
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
133
|
+
for (const verdict of verdicts) {
|
|
134
|
+
const group = byScenario.get(verdict.scenarioId) ?? [];
|
|
135
|
+
group.push(verdict);
|
|
136
|
+
byScenario.set(verdict.scenarioId, group);
|
|
137
|
+
}
|
|
138
|
+
return [...byScenario.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([scenarioId, group]) => aggregateScenario(scenarioId, group));
|
|
139
|
+
}
|
|
140
|
+
function aggregateScenario(scenarioId, verdicts) {
|
|
141
|
+
const ordered = [...verdicts].sort((left, right) => left.trialIndex - right.trialIndex);
|
|
142
|
+
const passed = ordered.filter((verdict) => verdict.outcome === "pass").length;
|
|
143
|
+
const failed = ordered.filter((verdict) => verdict.outcome === "fail").length;
|
|
144
|
+
const unmeasured = ordered.filter((verdict) => verdict.outcome === "unmeasured").length;
|
|
145
|
+
const machineryFailures = ordered.filter((verdict) => verdict.machinery === "infrastructure_failure").length;
|
|
146
|
+
const fixed = Object.fromEntries(
|
|
147
|
+
EVAL_TRACKED_METRIC_NAMES.map((name) => [name, distribution(ordered.map((verdict) => verdict.trackedMetrics[name]))])
|
|
148
|
+
);
|
|
149
|
+
const reasons = new Set(ordered.flatMap((verdict) => Object.keys(verdict.trackedMetrics.expectedColdReasons)));
|
|
150
|
+
const expectedColdReasons = Object.fromEntries(
|
|
151
|
+
[...reasons].sort((left, right) => left.localeCompare(right)).map((reason) => [
|
|
152
|
+
reason,
|
|
153
|
+
distribution(
|
|
154
|
+
ordered.map(
|
|
155
|
+
(verdict) => verdict.trackedMetrics.expectedColdReasons[reason] ?? { value: 0, source: "ledger" }
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
])
|
|
159
|
+
);
|
|
160
|
+
const k = ordered.length;
|
|
161
|
+
return {
|
|
162
|
+
scenarioId,
|
|
163
|
+
trials: k,
|
|
164
|
+
k,
|
|
165
|
+
passed,
|
|
166
|
+
failed,
|
|
167
|
+
unmeasured,
|
|
168
|
+
machineryFailures,
|
|
169
|
+
passAtK: k > 0 && passed > 0 ? 1 : 0,
|
|
170
|
+
passPowK: k > 0 && passed === k ? 1 : 0,
|
|
171
|
+
trackedMetrics: { ...fixed, expectedColdReasons }
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
function distribution(metrics) {
|
|
175
|
+
const values = metrics.flatMap((metric) => metric.value === null ? [] : [metric.value]);
|
|
176
|
+
const sources = [...new Set(metrics.map((metric) => metric.source))].sort(compareSources);
|
|
177
|
+
if (values.length === 0) {
|
|
178
|
+
return {
|
|
179
|
+
observations: metrics.length,
|
|
180
|
+
measured: 0,
|
|
181
|
+
unmeasured: metrics.length,
|
|
182
|
+
mean: null,
|
|
183
|
+
min: null,
|
|
184
|
+
max: null,
|
|
185
|
+
p90: null,
|
|
186
|
+
variance: null,
|
|
187
|
+
standardDeviation: null,
|
|
188
|
+
sources
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
const ordered = [...values].sort((left, right) => left - right);
|
|
192
|
+
const p90Index = Math.max(0, Math.ceil(ordered.length * 0.9) - 1);
|
|
193
|
+
const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
|
|
194
|
+
const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
|
|
195
|
+
return {
|
|
196
|
+
observations: metrics.length,
|
|
197
|
+
measured: values.length,
|
|
198
|
+
unmeasured: metrics.length - values.length,
|
|
199
|
+
mean,
|
|
200
|
+
min: ordered[0] ?? null,
|
|
201
|
+
max: ordered.at(-1) ?? null,
|
|
202
|
+
p90: ordered[p90Index] ?? null,
|
|
203
|
+
variance,
|
|
204
|
+
standardDeviation: Math.sqrt(variance),
|
|
205
|
+
sources
|
|
206
|
+
};
|
|
207
|
+
}
|
|
208
|
+
function compareSources(left, right) {
|
|
209
|
+
return sourceOrder(left) - sourceOrder(right);
|
|
210
|
+
}
|
|
211
|
+
function sourceOrder(source) {
|
|
212
|
+
if (source === "ledger") return 0;
|
|
213
|
+
if (source === "receipt") return 1;
|
|
214
|
+
return 2;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// src/domains/eval/compare/behavioral.ts
|
|
218
|
+
init_esm_shims();
|
|
219
|
+
|
|
220
|
+
// src/domains/eval/compare/envelope.ts
|
|
221
|
+
init_esm_shims();
|
|
222
|
+
function compareEvalExecutionEnvelopesV1(identity, baseline, candidate, baselineDimensions, candidateDimensions) {
|
|
223
|
+
const leftDimensions = [...baselineDimensions].sort();
|
|
224
|
+
const rightDimensions = [...candidateDimensions].sort();
|
|
225
|
+
if (stableJson(leftDimensions) !== stableJson(rightDimensions)) {
|
|
226
|
+
return { ...identity, fields: ["matrix.dimensions"] };
|
|
227
|
+
}
|
|
228
|
+
if (baseline.some((result) => result.executionEnvelope !== void 0) && baseline.some((result) => result.executionEnvelope === void 0) || candidate.some((result) => result.executionEnvelope !== void 0) && candidate.some((result) => result.executionEnvelope === void 0)) {
|
|
229
|
+
return { ...identity, fields: ["executionEnvelope.missingTrial"] };
|
|
230
|
+
}
|
|
231
|
+
const ignored = new Set(leftDimensions);
|
|
232
|
+
const baselineEnvelopes = uniqueEnvelopes(baseline, ignored);
|
|
233
|
+
const candidateEnvelopes = uniqueEnvelopes(candidate, ignored);
|
|
234
|
+
if (baselineEnvelopes.length === 0 && candidateEnvelopes.length === 0) return null;
|
|
235
|
+
if (baselineEnvelopes.length === 0 || candidateEnvelopes.length === 0) {
|
|
236
|
+
return { ...identity, fields: ["executionEnvelope"] };
|
|
237
|
+
}
|
|
238
|
+
if (baselineEnvelopes.length > 1 || candidateEnvelopes.length > 1) {
|
|
239
|
+
return { ...identity, fields: ["executionEnvelope.withinRunVariance"] };
|
|
240
|
+
}
|
|
241
|
+
const left = baselineEnvelopes[0];
|
|
242
|
+
const right = candidateEnvelopes[0];
|
|
243
|
+
if (left === void 0 || right === void 0 || stableJson(left) === stableJson(right)) return null;
|
|
244
|
+
return { ...identity, fields: differingFields(left, right, ignored) };
|
|
245
|
+
}
|
|
246
|
+
function uniqueEnvelopes(results, ignored) {
|
|
247
|
+
const byIdentity = /* @__PURE__ */ new Map();
|
|
248
|
+
for (const result of results) {
|
|
249
|
+
if (result.executionEnvelope === void 0) continue;
|
|
250
|
+
const normalized = normalizedEnvelope(result.executionEnvelope, ignored);
|
|
251
|
+
byIdentity.set(stableJson(normalized), normalized);
|
|
252
|
+
}
|
|
253
|
+
return [...byIdentity.values()];
|
|
254
|
+
}
|
|
255
|
+
function normalizedEnvelope(envelope, ignored) {
|
|
256
|
+
return {
|
|
257
|
+
...envelope,
|
|
258
|
+
prompt: ignored.has("prompt") ? { fragments: [], compositionHash: null } : envelope.prompt,
|
|
259
|
+
recipe: ignored.has("recipe") ? null : envelope.recipe,
|
|
260
|
+
target: ignored.has("target") ? "<matrix>" : envelope.target,
|
|
261
|
+
wireModel: ignored.has("wireModel") ? null : envelope.wireModel,
|
|
262
|
+
runtime: ignored.has("runtime") ? null : envelope.runtime,
|
|
263
|
+
thinkingLevel: ignored.has("thinkingLevel") ? null : envelope.thinkingLevel,
|
|
264
|
+
toolSignature: ignored.has("toolSignature") ? null : envelope.toolSignature,
|
|
265
|
+
autonomy: ignored.has("autonomy") ? null : envelope.autonomy,
|
|
266
|
+
policyHashes: ignored.has("policy") ? { rulePack: null, project: null } : envelope.policyHashes,
|
|
267
|
+
projectContext: ignored.has("projectContext") ? {
|
|
268
|
+
kind: "none",
|
|
269
|
+
tier: null,
|
|
270
|
+
contentHash: null,
|
|
271
|
+
chars: null,
|
|
272
|
+
sections: [],
|
|
273
|
+
rulesApplied: [],
|
|
274
|
+
operatorProfileApplied: null
|
|
275
|
+
} : envelope.projectContext,
|
|
276
|
+
corpus: ignored.has("corpus") ? { id: "<matrix>", version: "<matrix>" } : envelope.corpus
|
|
277
|
+
};
|
|
278
|
+
}
|
|
279
|
+
function differingFields(left, right, ignored) {
|
|
280
|
+
const fields = [
|
|
281
|
+
["prompt", "prompt", left.prompt, right.prompt],
|
|
282
|
+
["recipe", "recipe", left.recipe, right.recipe],
|
|
283
|
+
["target", "target", left.target, right.target],
|
|
284
|
+
["wireModel", "wireModel", left.wireModel, right.wireModel],
|
|
285
|
+
["runtime", "runtime", left.runtime, right.runtime],
|
|
286
|
+
["thinkingLevel", "thinkingLevel", left.thinkingLevel, right.thinkingLevel],
|
|
287
|
+
["toolSignature", "toolSignature", left.toolSignature, right.toolSignature],
|
|
288
|
+
["autonomy", "autonomy", left.autonomy, right.autonomy],
|
|
289
|
+
["policy", "policyHashes", left.policyHashes, right.policyHashes],
|
|
290
|
+
["projectContext", "projectContext", left.projectContext, right.projectContext],
|
|
291
|
+
["corpus", "corpus", left.corpus, right.corpus]
|
|
292
|
+
];
|
|
293
|
+
return fields.flatMap(
|
|
294
|
+
([dimension, field, baseline, candidate]) => ignored.has(dimension) || stableJson(baseline) === stableJson(candidate) ? [] : [field]
|
|
295
|
+
);
|
|
296
|
+
}
|
|
297
|
+
function stableJson(value) {
|
|
298
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
|
|
299
|
+
if (typeof value === "object" && value !== null) {
|
|
300
|
+
return `{${Object.entries(value).filter(([, entry]) => entry !== void 0).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
|
|
301
|
+
}
|
|
302
|
+
return JSON.stringify(value);
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// src/domains/eval/compare/behavioral.ts
|
|
306
|
+
function compareEvalBehaviorMetricsV1(baseline, candidate) {
|
|
307
|
+
const baselineGroups = behaviorGroups(baseline);
|
|
308
|
+
const candidateGroups = behaviorGroups(candidate);
|
|
309
|
+
const keys = /* @__PURE__ */ new Set([...baselineGroups.keys(), ...candidateGroups.keys()]);
|
|
310
|
+
const comparisons = [];
|
|
311
|
+
const envelopeMismatches = [];
|
|
312
|
+
const baselineDimensions = baseline.matrix.dimensions ?? [];
|
|
313
|
+
const candidateDimensions = candidate.matrix.dimensions ?? [];
|
|
314
|
+
for (const key of [...keys].sort((left, right) => left.localeCompare(right))) {
|
|
315
|
+
const baselineGroup = baselineGroups.get(key);
|
|
316
|
+
const candidateGroup = candidateGroups.get(key);
|
|
317
|
+
const identity = baselineGroup ?? candidateGroup;
|
|
318
|
+
if (identity === void 0) continue;
|
|
319
|
+
const envelopeMismatch = compareEvalExecutionEnvelopesV1(
|
|
320
|
+
identity,
|
|
321
|
+
baselineGroup?.results ?? [],
|
|
322
|
+
candidateGroup?.results ?? [],
|
|
323
|
+
baselineDimensions,
|
|
324
|
+
candidateDimensions
|
|
325
|
+
);
|
|
326
|
+
if (envelopeMismatch !== null) envelopeMismatches.push(envelopeMismatch);
|
|
327
|
+
const comparability = {
|
|
328
|
+
comparable: envelopeMismatch === null,
|
|
329
|
+
mismatchedFields: envelopeMismatch?.fields ?? []
|
|
330
|
+
};
|
|
331
|
+
for (const definition of EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1) {
|
|
332
|
+
const baselineDistribution = behaviorDistribution(baselineGroup?.results ?? [], definition);
|
|
333
|
+
const candidateDistribution = behaviorDistribution(candidateGroup?.results ?? [], definition);
|
|
334
|
+
comparisons.push({
|
|
335
|
+
scenarioId: identity.scenarioId,
|
|
336
|
+
role: identity.role,
|
|
337
|
+
target: identity.target,
|
|
338
|
+
metric: definition.name,
|
|
339
|
+
family: definition.family,
|
|
340
|
+
direction: definition.direction,
|
|
341
|
+
hardGate: definition.hardGate,
|
|
342
|
+
baseline: baselineDistribution,
|
|
343
|
+
candidate: candidateDistribution,
|
|
344
|
+
change: envelopeMismatch === null ? classifyChange(baselineDistribution.mean, candidateDistribution.mean, definition.direction) : "incomparable",
|
|
345
|
+
meanDelta: subtractNullable(candidateDistribution.mean, baselineDistribution.mean),
|
|
346
|
+
varianceChange: envelopeMismatch === null ? classifyChange(baselineDistribution.variance, candidateDistribution.variance, "lower") : "incomparable",
|
|
347
|
+
varianceDelta: subtractNullable(candidateDistribution.variance, baselineDistribution.variance),
|
|
348
|
+
comparability
|
|
349
|
+
});
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
const failures = comparisons.flatMap(
|
|
353
|
+
(comparison) => comparison.hardGate && (comparison.change === "regressed" || comparison.change === "incomparable" && comparison.baseline.mean !== null && comparison.candidate.mean === null) ? [
|
|
354
|
+
{
|
|
355
|
+
scenarioId: comparison.scenarioId,
|
|
356
|
+
role: comparison.role,
|
|
357
|
+
target: comparison.target,
|
|
358
|
+
metric: comparison.metric,
|
|
359
|
+
change: comparison.change
|
|
360
|
+
}
|
|
361
|
+
] : []
|
|
362
|
+
);
|
|
363
|
+
return {
|
|
364
|
+
comparisons,
|
|
365
|
+
hardGate: {
|
|
366
|
+
pass: failures.length === 0 && envelopeMismatches.length === 0,
|
|
367
|
+
failures,
|
|
368
|
+
envelopeFailures: envelopeMismatches
|
|
369
|
+
},
|
|
370
|
+
envelopeMismatches
|
|
371
|
+
};
|
|
372
|
+
}
|
|
373
|
+
function classifyChange(baseline, candidate, direction) {
|
|
374
|
+
if (baseline === null || candidate === null) return "incomparable";
|
|
375
|
+
if (baseline === candidate) return "unchanged";
|
|
376
|
+
if (direction === "higher") return candidate > baseline ? "improved" : "regressed";
|
|
377
|
+
return candidate < baseline ? "improved" : "regressed";
|
|
378
|
+
}
|
|
379
|
+
function behaviorGroups(artifact) {
|
|
380
|
+
const groups = /* @__PURE__ */ new Map();
|
|
381
|
+
for (const result of artifact.results) {
|
|
382
|
+
const behavioral = result.behavioralMetrics;
|
|
383
|
+
if (behavioral === void 0) continue;
|
|
384
|
+
const key = groupKey(behavioral.scenarioId, behavioral.role, behavioral.target);
|
|
385
|
+
const group = groups.get(key) ?? {
|
|
386
|
+
scenarioId: behavioral.scenarioId,
|
|
387
|
+
role: behavioral.role,
|
|
388
|
+
target: behavioral.target,
|
|
389
|
+
results: []
|
|
390
|
+
};
|
|
391
|
+
group.results.push(result);
|
|
392
|
+
groups.set(key, group);
|
|
393
|
+
}
|
|
394
|
+
return groups;
|
|
395
|
+
}
|
|
396
|
+
function behaviorDistribution(results, definition) {
|
|
397
|
+
const observations = results.map((result) => result.behavioralMetrics?.metrics[definition.name].value ?? null);
|
|
398
|
+
const values = observations.flatMap((value) => value === null ? [] : [value]);
|
|
399
|
+
if (values.length === 0) {
|
|
400
|
+
return {
|
|
401
|
+
observations: observations.length,
|
|
402
|
+
measured: 0,
|
|
403
|
+
unmeasured: observations.length,
|
|
404
|
+
mean: null,
|
|
405
|
+
min: null,
|
|
406
|
+
max: null,
|
|
407
|
+
p90: null,
|
|
408
|
+
variance: null,
|
|
409
|
+
standardDeviation: null,
|
|
410
|
+
source: definition.source
|
|
411
|
+
};
|
|
412
|
+
}
|
|
413
|
+
const ordered = [...values].sort((left, right) => left - right);
|
|
414
|
+
const mean = ordered.reduce((sum2, value) => sum2 + value, 0) / ordered.length;
|
|
415
|
+
const variance = ordered.reduce((sum2, value) => sum2 + (value - mean) ** 2, 0) / ordered.length;
|
|
416
|
+
return {
|
|
417
|
+
observations: observations.length,
|
|
418
|
+
measured: values.length,
|
|
419
|
+
unmeasured: observations.length - values.length,
|
|
420
|
+
mean,
|
|
421
|
+
min: ordered[0] ?? null,
|
|
422
|
+
max: ordered.at(-1) ?? null,
|
|
423
|
+
p90: ordered[Math.max(0, Math.ceil(ordered.length * 0.9) - 1)] ?? null,
|
|
424
|
+
variance,
|
|
425
|
+
standardDeviation: Math.sqrt(variance),
|
|
426
|
+
source: definition.source
|
|
427
|
+
};
|
|
428
|
+
}
|
|
429
|
+
function groupKey(scenarioId, role, target) {
|
|
430
|
+
return JSON.stringify([scenarioId, role, target.id, target.model]);
|
|
431
|
+
}
|
|
432
|
+
function subtractNullable(left, right) {
|
|
433
|
+
return left === null || right === null ? null : left - right;
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
// src/domains/eval/compare/compare.ts
|
|
437
|
+
var EvalServingConfigurationDriftError = class extends Error {
|
|
438
|
+
baseline;
|
|
439
|
+
candidate;
|
|
440
|
+
constructor(baseline, candidate) {
|
|
441
|
+
super(
|
|
442
|
+
[
|
|
443
|
+
"serving configuration drift; pass --allow-config-drift to compare these runs",
|
|
444
|
+
`baseline serving: ${renderEvalServingConfiguration(baseline)}`,
|
|
445
|
+
`candidate serving: ${renderEvalServingConfiguration(candidate)}`
|
|
446
|
+
].join("\n")
|
|
447
|
+
);
|
|
448
|
+
this.name = "EvalServingConfigurationDriftError";
|
|
449
|
+
this.baseline = baseline;
|
|
450
|
+
this.candidate = candidate;
|
|
451
|
+
}
|
|
452
|
+
};
|
|
453
|
+
function compareEvalArtifactsV4(baseline, candidate, options = {}) {
|
|
75
454
|
const baselineTokens = baseline.summary.tokens;
|
|
76
455
|
const candidateTokens = candidate.summary.tokens;
|
|
456
|
+
const baselineServing = servingConfigurationOf(baseline);
|
|
457
|
+
const candidateServing = servingConfigurationOf(candidate);
|
|
458
|
+
const configDrift = !sameEvalServingConfiguration(baselineServing, candidateServing);
|
|
459
|
+
if (configDrift && options.allowConfigDrift !== true) {
|
|
460
|
+
throw new EvalServingConfigurationDriftError(baselineServing, candidateServing);
|
|
461
|
+
}
|
|
462
|
+
const behavioral = compareEvalBehaviorMetricsV1(baseline, candidate);
|
|
463
|
+
const trackedMetrics = compareTrackedMetrics(baseline, candidate, options.metric);
|
|
464
|
+
const normalizedFilter = normalizeMetricFilter(options.metric);
|
|
465
|
+
const behavioralMetrics = normalizedFilter === void 0 ? behavioral.comparisons : behavioral.comparisons.filter((row) => row.metric === normalizedFilter || row.family === normalizedFilter);
|
|
466
|
+
if (options.metric !== void 0 && trackedMetrics.length === 0 && behavioralMetrics.length === 0) {
|
|
467
|
+
throw new Error(`eval metric not found: ${options.metric}`);
|
|
468
|
+
}
|
|
77
469
|
return {
|
|
78
470
|
baselineEvalId: baseline.evalId,
|
|
79
471
|
candidateEvalId: candidate.evalId,
|
|
472
|
+
baselineServingConfiguration: baselineServing,
|
|
473
|
+
candidateServingConfiguration: candidateServing,
|
|
474
|
+
configDrift,
|
|
80
475
|
passRateDelta: candidate.summary.passRate - baseline.summary.passRate,
|
|
81
476
|
tokenDelta: baselineTokens.measured && candidateTokens.measured ? candidateTokens.total - baselineTokens.total : null,
|
|
82
|
-
wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs
|
|
477
|
+
wallTimeDelta: candidate.summary.wallTimeMs - baseline.summary.wallTimeMs,
|
|
478
|
+
trackedMetrics,
|
|
479
|
+
behavioralMetrics,
|
|
480
|
+
hardGate: behavioral.hardGate,
|
|
481
|
+
envelopeMismatches: behavioral.envelopeMismatches,
|
|
482
|
+
scenarioReports: behaviorRollups(behavioralMetrics, (row) => row.scenarioId),
|
|
483
|
+
roleReports: behaviorRollups(behavioralMetrics, (row) => row.role),
|
|
484
|
+
affectedCorpusResults: behavioral.envelopeMismatches.flatMap((mismatch) => {
|
|
485
|
+
const changedFields = mismatch.fields.filter((field) => field === "prompt" || field === "recipe");
|
|
486
|
+
return changedFields.length === 0 ? [] : [{ scenarioId: mismatch.scenarioId, role: mismatch.role, changedFields }];
|
|
487
|
+
})
|
|
83
488
|
};
|
|
84
489
|
}
|
|
85
490
|
function renderEvalComparisonV4(summary) {
|
|
491
|
+
const envelopeFailures = summary.envelopeMismatches.map(
|
|
492
|
+
(mismatch) => ` incomparable envelope: ${mismatch.scenarioId} ${mismatch.role} ${mismatch.target.id}/${mismatch.target.model ?? "none"} fields=${mismatch.fields.join(",")}`
|
|
493
|
+
);
|
|
494
|
+
const affected = summary.affectedCorpusResults.map(
|
|
495
|
+
(result) => ` affected corpus result: ${result.scenarioId} role=${result.role} changed=${result.changedFields.join(",")}`
|
|
496
|
+
);
|
|
497
|
+
const scenarioReports = renderRollups("per-scenario baseline/candidate report", summary.scenarioReports);
|
|
498
|
+
const roleReports = renderRollups("per-role baseline/candidate report", summary.roleReports);
|
|
499
|
+
const hardFailures = summary.hardGate.failures.map(
|
|
500
|
+
(failure) => ` hard failure: ${failure.scenarioId} ${failure.role} ${failure.target.id}/${failure.target.model ?? "none"} ${failure.metric} ${failure.change}`
|
|
501
|
+
);
|
|
502
|
+
const tracked = summary.trackedMetrics.flatMap((row, index) => [
|
|
503
|
+
...index === 0 ? [
|
|
504
|
+
"tracked metrics:",
|
|
505
|
+
"scenario metric baseline_mean baseline_p90 baseline_variance candidate_mean candidate_p90 candidate_variance mean_delta p90_delta variance_delta change variance_change sources"
|
|
506
|
+
] : [],
|
|
507
|
+
[
|
|
508
|
+
row.scenarioId,
|
|
509
|
+
row.metric,
|
|
510
|
+
formatMetric(row.baseline.mean),
|
|
511
|
+
formatMetric(row.baseline.p90),
|
|
512
|
+
formatMetric(row.baseline.variance ?? null),
|
|
513
|
+
formatMetric(row.candidate.mean),
|
|
514
|
+
formatMetric(row.candidate.p90),
|
|
515
|
+
formatMetric(row.candidate.variance ?? null),
|
|
516
|
+
formatSignedMetric(row.meanDelta),
|
|
517
|
+
formatSignedMetric(row.p90Delta),
|
|
518
|
+
formatSignedMetric(row.varianceDelta),
|
|
519
|
+
row.change,
|
|
520
|
+
row.varianceChange,
|
|
521
|
+
`${row.baseline.sources.join("+") || "none"}->${row.candidate.sources.join("+") || "none"}`
|
|
522
|
+
].join(" ")
|
|
523
|
+
]);
|
|
524
|
+
const behavioral = summary.behavioralMetrics.flatMap((row, index) => [
|
|
525
|
+
...index === 0 ? [
|
|
526
|
+
"behavioral metrics:",
|
|
527
|
+
"scenario role target model family metric baseline_mean baseline_variance baseline_coverage candidate_mean candidate_variance candidate_coverage mean_delta variance_delta change variance_change comparability gate source"
|
|
528
|
+
] : [],
|
|
529
|
+
[
|
|
530
|
+
row.scenarioId,
|
|
531
|
+
row.role,
|
|
532
|
+
row.target.id,
|
|
533
|
+
row.target.model ?? "none",
|
|
534
|
+
row.family,
|
|
535
|
+
row.metric,
|
|
536
|
+
formatMetric(row.baseline.mean),
|
|
537
|
+
formatMetric(row.baseline.variance),
|
|
538
|
+
`${row.baseline.measured}/${row.baseline.observations}`,
|
|
539
|
+
formatMetric(row.candidate.mean),
|
|
540
|
+
formatMetric(row.candidate.variance),
|
|
541
|
+
`${row.candidate.measured}/${row.candidate.observations}`,
|
|
542
|
+
formatSignedMetric(row.meanDelta),
|
|
543
|
+
formatSignedMetric(row.varianceDelta),
|
|
544
|
+
row.change,
|
|
545
|
+
row.varianceChange,
|
|
546
|
+
row.comparability.comparable ? "comparable" : `incomparable:${row.comparability.mismatchedFields.join(",")}`,
|
|
547
|
+
row.hardGate ? "hard" : "informational",
|
|
548
|
+
row.baseline.source
|
|
549
|
+
].join(" ")
|
|
550
|
+
]);
|
|
86
551
|
return [
|
|
87
552
|
`baseline eval: ${summary.baselineEvalId}`,
|
|
88
553
|
`candidate eval: ${summary.candidateEvalId}`,
|
|
554
|
+
`baseline serving: ${renderEvalServingConfiguration(summary.baselineServingConfiguration)}`,
|
|
555
|
+
`candidate serving: ${renderEvalServingConfiguration(summary.candidateServingConfiguration)}`,
|
|
556
|
+
`config drift: ${summary.configDrift ? "allowed" : "none"}`,
|
|
89
557
|
`pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
|
|
90
558
|
`token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
|
|
91
559
|
`wall-time delta ms: ${summary.wallTimeDelta}`,
|
|
560
|
+
`behavioral hard gate: ${summary.hardGate.pass ? "pass" : `fail (${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length})`}`,
|
|
561
|
+
...hardFailures,
|
|
562
|
+
...envelopeFailures,
|
|
563
|
+
...affected,
|
|
564
|
+
...scenarioReports,
|
|
565
|
+
...roleReports,
|
|
566
|
+
...tracked,
|
|
567
|
+
...behavioral,
|
|
92
568
|
""
|
|
93
569
|
].join("\n");
|
|
94
570
|
}
|
|
571
|
+
function behaviorRollups(rows, keyOf) {
|
|
572
|
+
const groups = /* @__PURE__ */ new Map();
|
|
573
|
+
for (const row of rows) groups.set(keyOf(row), [...groups.get(keyOf(row)) ?? [], row]);
|
|
574
|
+
return [...groups.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([id, grouped]) => ({
|
|
575
|
+
id,
|
|
576
|
+
metrics: changeCounts(grouped.map((row) => row.change)),
|
|
577
|
+
variance: changeCounts(grouped.map((row) => row.varianceChange))
|
|
578
|
+
}));
|
|
579
|
+
}
|
|
580
|
+
function changeCounts(changes) {
|
|
581
|
+
return {
|
|
582
|
+
improved: changes.filter((change) => change === "improved").length,
|
|
583
|
+
regressed: changes.filter((change) => change === "regressed").length,
|
|
584
|
+
unchanged: changes.filter((change) => change === "unchanged").length,
|
|
585
|
+
incomparable: changes.filter((change) => change === "incomparable").length
|
|
586
|
+
};
|
|
587
|
+
}
|
|
588
|
+
function renderRollups(title, reports) {
|
|
589
|
+
if (reports.length === 0) return [];
|
|
590
|
+
return [
|
|
591
|
+
`${title}:`,
|
|
592
|
+
...reports.map(
|
|
593
|
+
(report) => ` ${report.id}: metrics ${renderChangeCounts(report.metrics)}; variance ${renderChangeCounts(report.variance)}`
|
|
594
|
+
)
|
|
595
|
+
];
|
|
596
|
+
}
|
|
597
|
+
function renderChangeCounts(counts) {
|
|
598
|
+
return `improved=${counts.improved} regressed=${counts.regressed} unchanged=${counts.unchanged} incomparable=${counts.incomparable}`;
|
|
599
|
+
}
|
|
600
|
+
function compareTrackedMetrics(baseline, candidate, metricFilter) {
|
|
601
|
+
const baselineAggregates = baseline.aggregates ?? aggregateEvalVerdicts(baseline.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
|
|
602
|
+
const candidateAggregates = candidate.aggregates ?? aggregateEvalVerdicts(candidate.results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict]));
|
|
603
|
+
const baselineByScenario = new Map(baselineAggregates.map((entry) => [entry.scenarioId, entry]));
|
|
604
|
+
const candidateByScenario = new Map(candidateAggregates.map((entry) => [entry.scenarioId, entry]));
|
|
605
|
+
const scenarioIds = [...baselineByScenario.keys()].filter((scenarioId) => candidateByScenario.has(scenarioId)).sort((left, right) => left.localeCompare(right));
|
|
606
|
+
const filter = normalizeMetricFilter(metricFilter);
|
|
607
|
+
const rows = [];
|
|
608
|
+
for (const scenarioId of scenarioIds) {
|
|
609
|
+
const baselineAggregate = baselineByScenario.get(scenarioId);
|
|
610
|
+
const candidateAggregate = candidateByScenario.get(scenarioId);
|
|
611
|
+
if (baselineAggregate === void 0 || candidateAggregate === void 0) continue;
|
|
612
|
+
for (const metric of EVAL_TRACKED_METRIC_NAMES) {
|
|
613
|
+
if (filter !== void 0 && filter !== metric) continue;
|
|
614
|
+
rows.push(
|
|
615
|
+
metricComparison(
|
|
616
|
+
scenarioId,
|
|
617
|
+
metric,
|
|
618
|
+
baselineAggregate.trackedMetrics[metric],
|
|
619
|
+
candidateAggregate.trackedMetrics[metric]
|
|
620
|
+
)
|
|
621
|
+
);
|
|
622
|
+
}
|
|
623
|
+
const reasons = /* @__PURE__ */ new Set([
|
|
624
|
+
...Object.keys(baselineAggregate.trackedMetrics.expectedColdReasons),
|
|
625
|
+
...Object.keys(candidateAggregate.trackedMetrics.expectedColdReasons)
|
|
626
|
+
]);
|
|
627
|
+
for (const reason of [...reasons].sort((left, right) => left.localeCompare(right))) {
|
|
628
|
+
const metric = `expectedColdReasons.${reason}`;
|
|
629
|
+
if (filter !== void 0 && filter !== metric && filter !== "expectedColdReasons") continue;
|
|
630
|
+
rows.push(
|
|
631
|
+
metricComparison(
|
|
632
|
+
scenarioId,
|
|
633
|
+
metric,
|
|
634
|
+
baselineAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(baselineAggregate.k),
|
|
635
|
+
candidateAggregate.trackedMetrics.expectedColdReasons[reason] ?? zeroDistribution(candidateAggregate.k)
|
|
636
|
+
)
|
|
637
|
+
);
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
return rows;
|
|
641
|
+
}
|
|
642
|
+
function metricComparison(scenarioId, metric, baseline, candidate) {
|
|
643
|
+
assertComparableTrackedMetricSources(`${scenarioId}.${metric}`, baseline.sources, candidate.sources);
|
|
644
|
+
const direction = trackedMetricDirection(metric);
|
|
645
|
+
return {
|
|
646
|
+
scenarioId,
|
|
647
|
+
metric,
|
|
648
|
+
baseline,
|
|
649
|
+
candidate,
|
|
650
|
+
meanDelta: subtractNullable2(candidate.mean, baseline.mean),
|
|
651
|
+
p90Delta: subtractNullable2(candidate.p90, baseline.p90),
|
|
652
|
+
varianceDelta: subtractNullable2(candidate.variance ?? null, baseline.variance ?? null),
|
|
653
|
+
change: classifyChange(baseline.mean, candidate.mean, direction),
|
|
654
|
+
varianceChange: classifyChange(baseline.variance ?? null, candidate.variance ?? null, "lower")
|
|
655
|
+
};
|
|
656
|
+
}
|
|
657
|
+
function servingConfigurationOf(artifact) {
|
|
658
|
+
return artifact.servingConfiguration ?? {
|
|
659
|
+
targetId: artifact.matrix.target,
|
|
660
|
+
runtimeId: null,
|
|
661
|
+
modelId: artifact.matrix.model,
|
|
662
|
+
serverBuild: null,
|
|
663
|
+
total_slots: null,
|
|
664
|
+
thinkingLevel: artifact.matrix.thinking,
|
|
665
|
+
compiledPromptHash: null
|
|
666
|
+
};
|
|
667
|
+
}
|
|
668
|
+
function normalizeMetricFilter(metric) {
|
|
669
|
+
if (metric === void 0) return void 0;
|
|
670
|
+
const trimmed = metric.trim();
|
|
671
|
+
if (trimmed.startsWith("trackedMetrics.")) return trimmed.slice("trackedMetrics.".length);
|
|
672
|
+
if (trimmed.startsWith("behavioralMetrics.")) return trimmed.slice("behavioralMetrics.".length);
|
|
673
|
+
return trimmed;
|
|
674
|
+
}
|
|
675
|
+
function trackedMetricDirection(metric) {
|
|
676
|
+
return metric === "cacheReadTokens" ? "higher" : "lower";
|
|
677
|
+
}
|
|
678
|
+
function zeroDistribution(observations) {
|
|
679
|
+
return {
|
|
680
|
+
observations,
|
|
681
|
+
measured: observations,
|
|
682
|
+
unmeasured: 0,
|
|
683
|
+
mean: 0,
|
|
684
|
+
min: 0,
|
|
685
|
+
max: 0,
|
|
686
|
+
p90: 0,
|
|
687
|
+
variance: 0,
|
|
688
|
+
standardDeviation: 0,
|
|
689
|
+
sources: ["ledger"]
|
|
690
|
+
};
|
|
691
|
+
}
|
|
692
|
+
function subtractNullable2(left, right) {
|
|
693
|
+
return left === null || right === null ? null : left - right;
|
|
694
|
+
}
|
|
695
|
+
function formatMetric(value) {
|
|
696
|
+
return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(2);
|
|
697
|
+
}
|
|
698
|
+
function formatSignedMetric(value) {
|
|
699
|
+
if (value === null) return "null";
|
|
700
|
+
const formatted = formatMetric(value);
|
|
701
|
+
return value > 0 ? `+${formatted}` : formatted;
|
|
702
|
+
}
|
|
95
703
|
|
|
96
704
|
// src/domains/eval/compare/gates.ts
|
|
97
705
|
init_esm_shims();
|
|
@@ -102,12 +710,34 @@ var import_yaml = __toESM(require_dist(), 1);
|
|
|
102
710
|
import { readFileSync } from "node:fs";
|
|
103
711
|
function loadThresholds(path) {
|
|
104
712
|
const parsed = (0, import_yaml.parse)(readFileSync(path, "utf8"));
|
|
105
|
-
|
|
106
|
-
if (isRecord(
|
|
107
|
-
return {
|
|
713
|
+
const root = isRecord(parsed) && isRecord(parsed.thresholds) ? parsed.thresholds : parsed;
|
|
714
|
+
if (isRecord(root) && (Array.isArray(root.fail) || Array.isArray(root.informational))) {
|
|
715
|
+
return {
|
|
716
|
+
fail: parseAssertions(root.fail, `${path}.fail`),
|
|
717
|
+
informational: parseAssertions(root.informational, `${path}.informational`)
|
|
718
|
+
};
|
|
108
719
|
}
|
|
109
720
|
throw new Error(`invalid thresholds file: ${path}`);
|
|
110
721
|
}
|
|
722
|
+
function parseAssertions(value, source) {
|
|
723
|
+
if (value === void 0) return [];
|
|
724
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
725
|
+
return value.map((entry, index) => {
|
|
726
|
+
if (!isRecord(entry)) throw new Error(`${source}[${index}]: expected object`);
|
|
727
|
+
if (typeof entry.metric !== "string" || entry.metric.length === 0) {
|
|
728
|
+
throw new Error(`${source}[${index}].metric: expected non-empty string`);
|
|
729
|
+
}
|
|
730
|
+
if (!isOp(entry.op)) throw new Error(`${source}[${index}].op: expected lt, lte, gt, gte, eq, or neq`);
|
|
731
|
+
if (!isScalar(entry.value)) throw new Error(`${source}[${index}].value: expected scalar`);
|
|
732
|
+
return { metric: entry.metric, op: entry.op, value: entry.value };
|
|
733
|
+
});
|
|
734
|
+
}
|
|
735
|
+
function isOp(value) {
|
|
736
|
+
return value === "lt" || value === "lte" || value === "gt" || value === "gte" || value === "eq" || value === "neq";
|
|
737
|
+
}
|
|
738
|
+
function isScalar(value) {
|
|
739
|
+
return typeof value === "number" && Number.isFinite(value) || typeof value === "string" || typeof value === "boolean";
|
|
740
|
+
}
|
|
111
741
|
function resolveMetricAssertion(assertion, metrics, artifact) {
|
|
112
742
|
const actual = metricValue(assertion.metric, metrics, artifact);
|
|
113
743
|
return { actual, unresolved: actual === null, holds: comparisonHolds(assertion, actual) };
|
|
@@ -152,21 +782,26 @@ function isRecord(value) {
|
|
|
152
782
|
|
|
153
783
|
// src/domains/eval/compare/gates.ts
|
|
154
784
|
function evaluateGate(artifact, thresholds) {
|
|
155
|
-
const failures =
|
|
156
|
-
|
|
785
|
+
const failures = evaluateAssertions(artifact, thresholds.fail);
|
|
786
|
+
const informational = evaluateAssertions(artifact, thresholds.informational ?? []);
|
|
787
|
+
return { pass: failures.length === 0, failures, informational };
|
|
788
|
+
}
|
|
789
|
+
function evaluateAssertions(artifact, assertions) {
|
|
790
|
+
const findings = [];
|
|
791
|
+
for (const assertion of assertions) {
|
|
157
792
|
const whole = resolveMetricAssertion(assertion, {}, artifact);
|
|
158
793
|
if (!whole.unresolved) {
|
|
159
|
-
if (whole.holds)
|
|
794
|
+
if (whole.holds) findings.push({ assertion, actual: whole.actual, unresolved: false });
|
|
160
795
|
continue;
|
|
161
796
|
}
|
|
162
797
|
if (artifact.results.length === 0) {
|
|
163
|
-
|
|
798
|
+
findings.push({ assertion, actual: null, unresolved: true });
|
|
164
799
|
continue;
|
|
165
800
|
}
|
|
166
801
|
for (const result of artifact.results) {
|
|
167
802
|
const perRun = resolveMetricAssertion(assertion, result.metrics);
|
|
168
803
|
if (!perRun.unresolved && !perRun.holds) continue;
|
|
169
|
-
|
|
804
|
+
findings.push({
|
|
170
805
|
assertion,
|
|
171
806
|
actual: perRun.actual,
|
|
172
807
|
unresolved: perRun.unresolved,
|
|
@@ -175,7 +810,7 @@ function evaluateGate(artifact, thresholds) {
|
|
|
175
810
|
});
|
|
176
811
|
}
|
|
177
812
|
}
|
|
178
|
-
return
|
|
813
|
+
return findings;
|
|
179
814
|
}
|
|
180
815
|
function renderGateFailure(failure) {
|
|
181
816
|
const run = failure.taskId === void 0 ? "" : ` [${failure.taskId}#${failure.repeatIndex ?? 0}]`;
|
|
@@ -184,6 +819,121 @@ function renderGateFailure(failure) {
|
|
|
184
819
|
` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(failure.actual)}
|
|
185
820
|
`;
|
|
186
821
|
}
|
|
822
|
+
function renderInformationalBudget(finding) {
|
|
823
|
+
const run = finding.taskId === void 0 ? "" : ` [${finding.taskId}#${finding.repeatIndex ?? 0}]`;
|
|
824
|
+
const { metric, op, value } = finding.assertion;
|
|
825
|
+
return finding.unresolved ? ` ${metric}${run}: unmeasured informational budget
|
|
826
|
+
` : ` ${metric} ${op} ${JSON.stringify(value)}${run}: actual ${JSON.stringify(finding.actual)}
|
|
827
|
+
`;
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
// src/domains/eval/reports/comparison.ts
|
|
831
|
+
init_esm_shims();
|
|
832
|
+
function renderEvalComparisonReportV1(summary, format2) {
|
|
833
|
+
if (format2 === "json") return `${JSON.stringify(summary, null, 2)}
|
|
834
|
+
`;
|
|
835
|
+
if (format2 === "md") return renderMarkdown(summary);
|
|
836
|
+
if (format2 === "junit") return renderJunit(summary);
|
|
837
|
+
return renderEvalComparisonV4(summary);
|
|
838
|
+
}
|
|
839
|
+
function renderMarkdown(summary) {
|
|
840
|
+
const rows = summary.behavioralMetrics.map(
|
|
841
|
+
(row) => `| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`
|
|
842
|
+
);
|
|
843
|
+
const scenarioRows = summary.scenarioReports.map(
|
|
844
|
+
(report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
|
|
845
|
+
);
|
|
846
|
+
const roleRows = summary.roleReports.map(
|
|
847
|
+
(report) => `| ${cell(report.id)} | ${changeCounts2(report.metrics)} | ${changeCounts2(report.variance)} |`
|
|
848
|
+
);
|
|
849
|
+
return [
|
|
850
|
+
`# Eval comparison ${summary.baselineEvalId} \u2192 ${summary.candidateEvalId}`,
|
|
851
|
+
"",
|
|
852
|
+
`Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
|
|
853
|
+
...summary.hardGate.failures.map(
|
|
854
|
+
(failure) => `- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`
|
|
855
|
+
),
|
|
856
|
+
...summary.envelopeMismatches.map(
|
|
857
|
+
(mismatch) => `- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`
|
|
858
|
+
),
|
|
859
|
+
...summary.affectedCorpusResults.map(
|
|
860
|
+
(result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`
|
|
861
|
+
),
|
|
862
|
+
`Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
|
|
863
|
+
`Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
|
|
864
|
+
`Wall-time delta ms: ${summary.wallTimeDelta}`,
|
|
865
|
+
"",
|
|
866
|
+
"| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
|
|
867
|
+
"|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
|
|
868
|
+
...rows,
|
|
869
|
+
"",
|
|
870
|
+
"## Per-scenario baseline/candidate report",
|
|
871
|
+
"",
|
|
872
|
+
"| Scenario | Metric changes | Variance changes |",
|
|
873
|
+
"|---|---|---|",
|
|
874
|
+
...scenarioRows,
|
|
875
|
+
"",
|
|
876
|
+
"## Per-role baseline/candidate report",
|
|
877
|
+
"",
|
|
878
|
+
"| Role | Metric changes | Variance changes |",
|
|
879
|
+
"|---|---|---|",
|
|
880
|
+
...roleRows,
|
|
881
|
+
""
|
|
882
|
+
].join("\n");
|
|
883
|
+
}
|
|
884
|
+
function renderJunit(summary) {
|
|
885
|
+
const failures = new Set(
|
|
886
|
+
summary.hardGate.failures.map(
|
|
887
|
+
(failure) => JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric])
|
|
888
|
+
)
|
|
889
|
+
);
|
|
890
|
+
const represented = /* @__PURE__ */ new Set();
|
|
891
|
+
const cases = summary.behavioralMetrics.map((row) => {
|
|
892
|
+
const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
|
|
893
|
+
const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
|
|
894
|
+
represented.add(key);
|
|
895
|
+
const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
|
|
896
|
+
return failures.has(key) ? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>` : ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
|
|
897
|
+
});
|
|
898
|
+
for (const failure of summary.hardGate.failures) {
|
|
899
|
+
const key = JSON.stringify([
|
|
900
|
+
failure.scenarioId,
|
|
901
|
+
failure.role,
|
|
902
|
+
failure.target.id,
|
|
903
|
+
failure.target.model,
|
|
904
|
+
failure.metric
|
|
905
|
+
]);
|
|
906
|
+
if (represented.has(key)) continue;
|
|
907
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
|
|
908
|
+
cases.push(
|
|
909
|
+
` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`
|
|
910
|
+
);
|
|
911
|
+
}
|
|
912
|
+
for (const failure of summary.hardGate.envelopeFailures) {
|
|
913
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
|
|
914
|
+
cases.push(
|
|
915
|
+
` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`
|
|
916
|
+
);
|
|
917
|
+
}
|
|
918
|
+
return [
|
|
919
|
+
`<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
|
|
920
|
+
...cases,
|
|
921
|
+
"</testsuite>",
|
|
922
|
+
""
|
|
923
|
+
].join("\n");
|
|
924
|
+
}
|
|
925
|
+
function changeCounts2(counts) {
|
|
926
|
+
return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
|
|
927
|
+
}
|
|
928
|
+
function format(value) {
|
|
929
|
+
return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
|
|
930
|
+
}
|
|
931
|
+
function cell(value) {
|
|
932
|
+
return value.replaceAll("|", "\\|");
|
|
933
|
+
}
|
|
934
|
+
function escapeXml(value) {
|
|
935
|
+
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll('"', """);
|
|
936
|
+
}
|
|
187
937
|
|
|
188
938
|
// src/domains/eval/reports/json.ts
|
|
189
939
|
init_esm_shims();
|
|
@@ -195,21 +945,35 @@ function renderEvalJsonReportV4(artifact) {
|
|
|
195
945
|
// src/domains/eval/reports/junit.ts
|
|
196
946
|
init_esm_shims();
|
|
197
947
|
function renderEvalJunitReportV4(artifact) {
|
|
948
|
+
let failures = 0;
|
|
949
|
+
let skipped = 0;
|
|
198
950
|
const cases = artifact.results.map((result) => {
|
|
199
|
-
const name =
|
|
951
|
+
const name = escapeXml2(
|
|
200
952
|
`${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`
|
|
201
953
|
);
|
|
202
|
-
if (result.pass)
|
|
203
|
-
|
|
954
|
+
if (!result.pass) {
|
|
955
|
+
failures += 1;
|
|
956
|
+
return ` <testcase name="${name}"><failure message="${escapeXml2(result.failureClass ?? "failed")}" /></testcase>`;
|
|
957
|
+
}
|
|
958
|
+
const outcome = result.behavioral?.outcome;
|
|
959
|
+
if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
|
|
960
|
+
failures += 1;
|
|
961
|
+
return ` <testcase name="${name}"><failure message="${escapeXml2(outcome)}" /></testcase>`;
|
|
962
|
+
}
|
|
963
|
+
if (outcome === "unknown" || outcome === "unmeasured") {
|
|
964
|
+
skipped += 1;
|
|
965
|
+
return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml2(outcome)}" /></testcase>`;
|
|
966
|
+
}
|
|
967
|
+
return ` <testcase name="${name}" />`;
|
|
204
968
|
}).join("\n");
|
|
205
969
|
return [
|
|
206
|
-
`<testsuite name="${
|
|
970
|
+
`<testsuite name="${escapeXml2(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
|
|
207
971
|
cases,
|
|
208
972
|
"</testsuite>",
|
|
209
973
|
""
|
|
210
974
|
].join("\n");
|
|
211
975
|
}
|
|
212
|
-
function
|
|
976
|
+
function escapeXml2(value) {
|
|
213
977
|
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll('"', """);
|
|
214
978
|
}
|
|
215
979
|
|
|
@@ -223,10 +987,10 @@ function renderEvalMarkdownReportV4(artifact) {
|
|
|
223
987
|
`Target: ${artifact.matrix.target}`,
|
|
224
988
|
`Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
|
|
225
989
|
"",
|
|
226
|
-
"| Task | Target | Model | Repeat |
|
|
227
|
-
"
|
|
990
|
+
"| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
|
|
991
|
+
"|---|---|---|---|---:|---|---|---|",
|
|
228
992
|
...artifact.results.map(
|
|
229
|
-
(result) => `| ${result.taskId} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.failureClass ?? ""} |`
|
|
993
|
+
(result) => `| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`
|
|
230
994
|
),
|
|
231
995
|
""
|
|
232
996
|
];
|
|
@@ -251,6 +1015,12 @@ function renderEvalSweJsonlReportV4(artifact) {
|
|
|
251
1015
|
init_esm_shims();
|
|
252
1016
|
function renderEvalTextReportV4(artifact) {
|
|
253
1017
|
const tokens = artifact.summary.tokens;
|
|
1018
|
+
const behavioral = artifact.results.flatMap(
|
|
1019
|
+
(result) => result.behavioral === void 0 ? [] : [result.behavioral.outcome]
|
|
1020
|
+
);
|
|
1021
|
+
const behavioralSummary = behavioral.length === 0 ? [] : [
|
|
1022
|
+
`behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`
|
|
1023
|
+
];
|
|
254
1024
|
return [
|
|
255
1025
|
`eval: ${artifact.evalId}`,
|
|
256
1026
|
`suite: ${artifact.suite.id}`,
|
|
@@ -265,9 +1035,13 @@ function renderEvalTextReportV4(artifact) {
|
|
|
265
1035
|
// reported next to how many runs it actually covers.
|
|
266
1036
|
!tokens.measured ? `tokens total: unmeasured (0 of ${tokens.runs} runs reported usage)` : tokens.measuredRuns === tokens.runs ? `tokens total: ${tokens.total}` : `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
|
|
267
1037
|
`wall time ms: ${artifact.summary.wallTimeMs}`,
|
|
1038
|
+
...behavioralSummary,
|
|
268
1039
|
""
|
|
269
1040
|
].join("\n");
|
|
270
1041
|
}
|
|
1042
|
+
function count(values, wanted) {
|
|
1043
|
+
return values.filter((value) => value === wanted).length;
|
|
1044
|
+
}
|
|
271
1045
|
|
|
272
1046
|
// src/domains/eval/suites/load.ts
|
|
273
1047
|
init_esm_shims();
|
|
@@ -351,12 +1125,26 @@ function readMatrix(value, path, issues) {
|
|
|
351
1125
|
];
|
|
352
1126
|
});
|
|
353
1127
|
if (repeats === null || targets.length === 0) return null;
|
|
1128
|
+
let dimensions;
|
|
1129
|
+
if (value.dimensions !== void 0) {
|
|
1130
|
+
try {
|
|
1131
|
+
dimensions = parseEvalExecutionMatrixDimensionsV1(value.dimensions, `${path}.dimensions`);
|
|
1132
|
+
} catch (error) {
|
|
1133
|
+
issues.push({ path: `${path}.dimensions`, message: error instanceof Error ? error.message : String(error) });
|
|
1134
|
+
return null;
|
|
1135
|
+
}
|
|
1136
|
+
}
|
|
354
1137
|
const maxCostUsd = value.maxCostUsd;
|
|
355
1138
|
if (maxCostUsd !== void 0 && (typeof maxCostUsd !== "number" || !Number.isFinite(maxCostUsd) || maxCostUsd < 0)) {
|
|
356
1139
|
issues.push({ path: `${path}.maxCostUsd`, message: "expected non-negative number" });
|
|
357
1140
|
return null;
|
|
358
1141
|
}
|
|
359
|
-
return {
|
|
1142
|
+
return {
|
|
1143
|
+
targets,
|
|
1144
|
+
repeats,
|
|
1145
|
+
...dimensions === void 0 ? {} : { dimensions },
|
|
1146
|
+
...maxCostUsd === void 0 ? {} : { maxCostUsd }
|
|
1147
|
+
};
|
|
360
1148
|
}
|
|
361
1149
|
function readTasks(value, path, issues) {
|
|
362
1150
|
if (!Array.isArray(value) || value.length === 0) {
|
|
@@ -383,10 +1171,12 @@ function readTask(value, path, issues) {
|
|
|
383
1171
|
const workspace = readWorkspace(value.workspace, `${path}.workspace`, issues);
|
|
384
1172
|
const runner = readRunner(value.runner, `${path}.runner`, issues);
|
|
385
1173
|
const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
|
|
1174
|
+
const behavioral = readBehavioral(value.behavioral, `${path}.behavioral`, issues);
|
|
386
1175
|
if (id === null || workspace === null || runner === null || timeoutMs === null) return null;
|
|
387
1176
|
return {
|
|
388
1177
|
id,
|
|
389
1178
|
tags: readOptionalStringArray(value, "tags", `${path}.tags`, issues),
|
|
1179
|
+
...behavioral === void 0 ? {} : { behavioral },
|
|
390
1180
|
workspace,
|
|
391
1181
|
runner,
|
|
392
1182
|
verify: readVerify(value.verify, `${path}.verify`, issues),
|
|
@@ -394,6 +1184,15 @@ function readTask(value, path, issues) {
|
|
|
394
1184
|
timeoutMs
|
|
395
1185
|
};
|
|
396
1186
|
}
|
|
1187
|
+
function readBehavioral(value, path, issues) {
|
|
1188
|
+
if (value === void 0) return void 0;
|
|
1189
|
+
try {
|
|
1190
|
+
return parseEvalBehaviorScenarioV1(value, path);
|
|
1191
|
+
} catch (error) {
|
|
1192
|
+
issues.push({ path, message: error instanceof Error ? error.message : String(error) });
|
|
1193
|
+
return void 0;
|
|
1194
|
+
}
|
|
1195
|
+
}
|
|
397
1196
|
function readWorkspace(value, path, issues) {
|
|
398
1197
|
if (!isRecord2(value)) {
|
|
399
1198
|
issues.push({ path, message: "expected object" });
|
|
@@ -430,6 +1229,10 @@ function readRunner(value, path, issues) {
|
|
|
430
1229
|
}
|
|
431
1230
|
const prompt = optionalString(value, "prompt");
|
|
432
1231
|
const agent = optionalString(value, "agent");
|
|
1232
|
+
const autonomy = optionalString(value, "autonomy");
|
|
1233
|
+
if (autonomy !== void 0 && !["read-only", "suggest", "auto-edit", "full-auto"].includes(autonomy)) {
|
|
1234
|
+
issues.push({ path: `${path}.autonomy`, message: "expected read-only, suggest, auto-edit, or full-auto" });
|
|
1235
|
+
}
|
|
433
1236
|
if (agent !== void 0 && kind !== "clio-run") {
|
|
434
1237
|
issues.push({ path: `${path}.agent`, message: "agent is only valid on the clio-run runner" });
|
|
435
1238
|
}
|
|
@@ -437,6 +1240,7 @@ function readRunner(value, path, issues) {
|
|
|
437
1240
|
return {
|
|
438
1241
|
kind,
|
|
439
1242
|
...prompt === void 0 ? {} : { prompt },
|
|
1243
|
+
...autonomy === void 0 ? {} : { autonomy },
|
|
440
1244
|
...agent === void 0 ? {} : { agent },
|
|
441
1245
|
...command === void 0 ? {} : { command },
|
|
442
1246
|
commands: readOptionalStringArray(value, "commands", `${path}.commands`, issues),
|
|
@@ -463,7 +1267,22 @@ function readMetrics(value, path, issues) {
|
|
|
463
1267
|
issues.push({ path, message: "expected object" });
|
|
464
1268
|
return { collect: [] };
|
|
465
1269
|
}
|
|
466
|
-
|
|
1270
|
+
const observation = value.readObservation;
|
|
1271
|
+
let readObservation;
|
|
1272
|
+
if (observation !== void 0) {
|
|
1273
|
+
if (!isRecord2(observation)) {
|
|
1274
|
+
issues.push({ path: `${path}.readObservation`, message: "expected object" });
|
|
1275
|
+
} else {
|
|
1276
|
+
readObservation = {
|
|
1277
|
+
allowedPaths: readOptionalStringArray(observation, "allowedPaths", `${path}.readObservation.allowedPaths`, issues),
|
|
1278
|
+
decoyPaths: readOptionalStringArray(observation, "decoyPaths", `${path}.readObservation.decoyPaths`, issues)
|
|
1279
|
+
};
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
return {
|
|
1283
|
+
collect: readOptionalStringArray(value, "collect", `${path}.collect`, issues),
|
|
1284
|
+
...readObservation === void 0 ? {} : { readObservation }
|
|
1285
|
+
};
|
|
467
1286
|
}
|
|
468
1287
|
function readThresholds(value, path, issues) {
|
|
469
1288
|
if (value === void 0) return void 0;
|
|
@@ -471,7 +1290,10 @@ function readThresholds(value, path, issues) {
|
|
|
471
1290
|
issues.push({ path, message: "expected object" });
|
|
472
1291
|
return void 0;
|
|
473
1292
|
}
|
|
474
|
-
return {
|
|
1293
|
+
return {
|
|
1294
|
+
fail: readAssertions(value.fail, `${path}.fail`, issues),
|
|
1295
|
+
informational: readAssertions(value.informational, `${path}.informational`, issues)
|
|
1296
|
+
};
|
|
475
1297
|
}
|
|
476
1298
|
function readAssertions(value, path, issues) {
|
|
477
1299
|
if (value === void 0) return [];
|
|
@@ -620,7 +1442,8 @@ function resolveSuiteForRun(suite, options) {
|
|
|
620
1442
|
...suite,
|
|
621
1443
|
matrix: {
|
|
622
1444
|
...suite.matrix,
|
|
623
|
-
targets
|
|
1445
|
+
targets,
|
|
1446
|
+
...options.trials === void 0 ? {} : { repeats: options.trials }
|
|
624
1447
|
}
|
|
625
1448
|
};
|
|
626
1449
|
}
|
|
@@ -646,9 +1469,166 @@ function resolveTargets(targets, options) {
|
|
|
646
1469
|
|
|
647
1470
|
// src/domains/eval/suites/run.ts
|
|
648
1471
|
init_esm_shims();
|
|
649
|
-
import { mkdtemp as mkdtemp3, rm as rm3 } from "node:fs/promises";
|
|
1472
|
+
import { mkdtemp as mkdtemp3, rm as rm3, writeFile } from "node:fs/promises";
|
|
650
1473
|
import { tmpdir as tmpdir3 } from "node:os";
|
|
651
|
-
import { resolve as
|
|
1474
|
+
import { resolve as resolve7 } from "node:path";
|
|
1475
|
+
|
|
1476
|
+
// src/domains/eval/execution-provenance.ts
|
|
1477
|
+
init_esm_shims();
|
|
1478
|
+
import { createHash as createHash2 } from "node:crypto";
|
|
1479
|
+
function buildEvalExecutionEnvelopeV1(input) {
|
|
1480
|
+
const scenario = input.task.behavioral;
|
|
1481
|
+
if (scenario === void 0) throw new Error(`behavioral task ${input.task.id} has no behavioral scenario`);
|
|
1482
|
+
const manifest = input.ledger.promptManifests.at(-1) ?? null;
|
|
1483
|
+
const contextSnapshot = input.ledger.contextSnapshots.at(-1) ?? null;
|
|
1484
|
+
const recipe = recipeIdentity(
|
|
1485
|
+
input,
|
|
1486
|
+
scenario.execution.subject.kind === "worker" ? scenario.execution.subject.role : null
|
|
1487
|
+
);
|
|
1488
|
+
const policy = policyIdentity(input.cwd, input.receipt, input.observation);
|
|
1489
|
+
const autonomy = input.receipt?.autonomyEnforcement?.autonomy ?? input.observation?.autonomy ?? input.task.runner.autonomy ?? null;
|
|
1490
|
+
const promptFragments = promptFragmentIdentities(manifest, recipe, autonomy);
|
|
1491
|
+
const compositionHash = input.receipt?.staticCompositionHash ?? input.observation?.compositionHash ?? manifest?.systemPromptHash ?? contextSnapshot?.promptHash ?? null;
|
|
1492
|
+
const projectContext = projectContextIdentity(input, manifest, promptFragments);
|
|
1493
|
+
return {
|
|
1494
|
+
schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
1495
|
+
prompt: { fragments: promptFragments, compositionHash },
|
|
1496
|
+
recipe: recipe === null ? null : { id: recipe.id, version: recipe.version, contentHash: recipe.contentHash },
|
|
1497
|
+
target: input.receipt?.targetId ?? input.observation?.target ?? input.target.id,
|
|
1498
|
+
wireModel: input.receipt?.wireModelId ?? input.observation?.wireModel ?? contextSnapshot?.modelId ?? input.target.model ?? null,
|
|
1499
|
+
runtime: input.receipt?.runtimeId ?? input.observation?.runtime ?? contextSnapshot?.runtimeId ?? null,
|
|
1500
|
+
thinkingLevel: input.receipt?.runtimeResolution?.effectiveThinkingLevel ?? input.observation?.thinkingLevel ?? manifest?.thinkingLevel ?? input.target.thinking ?? null,
|
|
1501
|
+
toolSignature: input.receipt?.toolSignature ?? input.observation?.toolSignature ?? contextSnapshot?.toolSignature ?? null,
|
|
1502
|
+
autonomy,
|
|
1503
|
+
policyHashes: policy,
|
|
1504
|
+
projectContext,
|
|
1505
|
+
corpus: { ...scenario.corpus }
|
|
1506
|
+
};
|
|
1507
|
+
}
|
|
1508
|
+
function recipeIdentity(input, role) {
|
|
1509
|
+
const id = input.receipt?.agentId ?? input.task.runner.agent ?? role;
|
|
1510
|
+
if (id === null || input.cwd === null) return null;
|
|
1511
|
+
try {
|
|
1512
|
+
const recipe = discoverAgentRecipes(input.cwd).find((entry) => entry.id === id);
|
|
1513
|
+
if (recipe === void 0) return null;
|
|
1514
|
+
return {
|
|
1515
|
+
id: recipe.id,
|
|
1516
|
+
version: recipe.version,
|
|
1517
|
+
contentHash: agentSpecFingerprint(normalizeAgentSpec(recipe)),
|
|
1518
|
+
personaHash: sha256(recipe.body)
|
|
1519
|
+
};
|
|
1520
|
+
} catch {
|
|
1521
|
+
return null;
|
|
1522
|
+
}
|
|
1523
|
+
}
|
|
1524
|
+
function promptFragmentIdentities(manifest, recipe, autonomy) {
|
|
1525
|
+
let versions = /* @__PURE__ */ new Map();
|
|
1526
|
+
try {
|
|
1527
|
+
versions = new Map([...loadFragments().byId.values()].map((fragment) => [fragment.id, fragment.version]));
|
|
1528
|
+
} catch {
|
|
1529
|
+
}
|
|
1530
|
+
if (manifest !== null) {
|
|
1531
|
+
return manifest.fragments.map((fragment) => ({
|
|
1532
|
+
id: fragment.id,
|
|
1533
|
+
version: versions.get(fragment.id) ?? "unversioned",
|
|
1534
|
+
contentHash: fragment.contentHash
|
|
1535
|
+
})).sort((left, right) => left.id.localeCompare(right.id));
|
|
1536
|
+
}
|
|
1537
|
+
if (recipe === null) return [];
|
|
1538
|
+
const selected = ["identity.clio-worker", "operating.contract", "operating.worker"];
|
|
1539
|
+
if (autonomy !== null) selected.push(`safety.${autonomy}`);
|
|
1540
|
+
const fragments = [];
|
|
1541
|
+
try {
|
|
1542
|
+
const table = loadFragments();
|
|
1543
|
+
for (const id of selected) {
|
|
1544
|
+
const fragment = table.byId.get(id);
|
|
1545
|
+
if (fragment !== void 0) {
|
|
1546
|
+
fragments.push({ id, version: fragment.version, contentHash: fragment.contentHash });
|
|
1547
|
+
}
|
|
1548
|
+
}
|
|
1549
|
+
} catch {
|
|
1550
|
+
}
|
|
1551
|
+
fragments.push({ id: `persona.${recipe.id}`, version: recipe.version, contentHash: recipe.personaHash });
|
|
1552
|
+
return fragments.sort((left, right) => left.id.localeCompare(right.id));
|
|
1553
|
+
}
|
|
1554
|
+
function policyIdentity(cwd, receipt, observation) {
|
|
1555
|
+
const sealed = receipt?.reproducibility?.safetyPolicy;
|
|
1556
|
+
if (sealed !== void 0) return { rulePack: sealed.rulePackHash, project: sealed.projectPolicyHash };
|
|
1557
|
+
if (observation !== void 0) return { ...observation.policyHashes };
|
|
1558
|
+
if (cwd === null) return { rulePack: null, project: null };
|
|
1559
|
+
try {
|
|
1560
|
+
const metadata = createSafetyPolicyEngine({ cwd }).metadata();
|
|
1561
|
+
return { rulePack: metadata.rulePackHash, project: metadata.projectPolicyHash };
|
|
1562
|
+
} catch {
|
|
1563
|
+
return { rulePack: null, project: null };
|
|
1564
|
+
}
|
|
1565
|
+
}
|
|
1566
|
+
function projectContextIdentity(input, manifest, fragments) {
|
|
1567
|
+
const receipt = input.receipt;
|
|
1568
|
+
if (receipt?.projectContext !== void 0) {
|
|
1569
|
+
const sections = [...receipt.projectContext.sections ?? []].sort();
|
|
1570
|
+
const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
|
|
1571
|
+
return {
|
|
1572
|
+
kind: "worker",
|
|
1573
|
+
tier: receipt.projectContext.tier,
|
|
1574
|
+
contentHash: hasContentBearingContext ? receipt.projectContext.contentHash ?? null : null,
|
|
1575
|
+
chars: hasContentBearingContext ? receipt.projectContext.chars ?? null : null,
|
|
1576
|
+
sections,
|
|
1577
|
+
rulesApplied: [...receipt.rulesApplied ?? []].sort(),
|
|
1578
|
+
operatorProfileApplied: receipt.operatorProfileApplied ?? null
|
|
1579
|
+
};
|
|
1580
|
+
}
|
|
1581
|
+
const observed = input.observation?.projectContext;
|
|
1582
|
+
if (observed !== void 0 && observed !== null) {
|
|
1583
|
+
const sections = [...observed.sections].sort();
|
|
1584
|
+
const hasContentBearingContext = sections.some((section) => section !== "workspace-root");
|
|
1585
|
+
return {
|
|
1586
|
+
kind: "worker",
|
|
1587
|
+
tier: observed.tier,
|
|
1588
|
+
contentHash: hasContentBearingContext ? observed.contentHash : null,
|
|
1589
|
+
chars: hasContentBearingContext ? observed.chars : null,
|
|
1590
|
+
sections,
|
|
1591
|
+
rulesApplied: [...observed.rulesApplied].sort(),
|
|
1592
|
+
operatorProfileApplied: observed.operatorProfileApplied
|
|
1593
|
+
};
|
|
1594
|
+
}
|
|
1595
|
+
if (manifest !== null) {
|
|
1596
|
+
const contextFragments = fragments.filter((fragment) => fragment.id.startsWith("context."));
|
|
1597
|
+
const preload = manifest.projectPreload;
|
|
1598
|
+
const identity = {
|
|
1599
|
+
preload,
|
|
1600
|
+
fragments: contextFragments.map((fragment) => [fragment.id, fragment.contentHash])
|
|
1601
|
+
};
|
|
1602
|
+
return {
|
|
1603
|
+
kind: "session",
|
|
1604
|
+
tier: preload?.mode ?? null,
|
|
1605
|
+
contentHash: sha256(stableJson2(identity)),
|
|
1606
|
+
chars: preload?.chars ?? null,
|
|
1607
|
+
sections: contextFragments.map((fragment) => fragment.id).sort(),
|
|
1608
|
+
rulesApplied: [],
|
|
1609
|
+
operatorProfileApplied: contextFragments.some((fragment) => fragment.id === "context.operator-profile")
|
|
1610
|
+
};
|
|
1611
|
+
}
|
|
1612
|
+
return {
|
|
1613
|
+
kind: "none",
|
|
1614
|
+
tier: null,
|
|
1615
|
+
contentHash: null,
|
|
1616
|
+
chars: null,
|
|
1617
|
+
sections: [],
|
|
1618
|
+
rulesApplied: [],
|
|
1619
|
+
operatorProfileApplied: null
|
|
1620
|
+
};
|
|
1621
|
+
}
|
|
1622
|
+
function stableJson2(value) {
|
|
1623
|
+
if (Array.isArray(value)) return `[${value.map(stableJson2).join(",")}]`;
|
|
1624
|
+
if (typeof value === "object" && value !== null) {
|
|
1625
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson2(entry)}`).join(",")}}`;
|
|
1626
|
+
}
|
|
1627
|
+
return JSON.stringify(value);
|
|
1628
|
+
}
|
|
1629
|
+
function sha256(value) {
|
|
1630
|
+
return createHash2("sha256").update(value, "utf8").digest("hex");
|
|
1631
|
+
}
|
|
652
1632
|
|
|
653
1633
|
// src/domains/eval/metrics/context.ts
|
|
654
1634
|
init_esm_shims();
|
|
@@ -1295,19 +2275,317 @@ function zeroToolCallMetrics() {
|
|
|
1295
2275
|
};
|
|
1296
2276
|
}
|
|
1297
2277
|
|
|
2278
|
+
// src/domains/eval/metrics/tracked.ts
|
|
2279
|
+
init_esm_shims();
|
|
2280
|
+
import { readFile as readFile2 } from "node:fs/promises";
|
|
2281
|
+
import { dirname as dirname2, join as join2 } from "node:path";
|
|
2282
|
+
async function readEvalLedgerSnapshot(stateDir) {
|
|
2283
|
+
const refs = await listSessionLedgerRefs(stateDir);
|
|
2284
|
+
const entries = [];
|
|
2285
|
+
const compiledPromptHashes = [];
|
|
2286
|
+
const promptManifests = [];
|
|
2287
|
+
const contextSnapshots = [];
|
|
2288
|
+
for (const ref of refs) {
|
|
2289
|
+
try {
|
|
2290
|
+
const raw = await readFile2(ref.path, "utf8");
|
|
2291
|
+
entries.push(...parseSessionEntries(raw, ref.path).entries);
|
|
2292
|
+
} catch {
|
|
2293
|
+
}
|
|
2294
|
+
try {
|
|
2295
|
+
const manifest = await readFile2(join2(dirname2(ref.path), "prompt-manifest.jsonl"), "utf8");
|
|
2296
|
+
for (const line of manifest.split(/\r?\n/u)) {
|
|
2297
|
+
const record = parseJsonRecord2(line);
|
|
2298
|
+
if (record === null) continue;
|
|
2299
|
+
const hash = record?.systemPromptHash;
|
|
2300
|
+
if (typeof hash !== "string" || !/^[a-f0-9]{64}$/u.test(hash)) continue;
|
|
2301
|
+
compiledPromptHashes.push(hash);
|
|
2302
|
+
const observation = promptManifestObservation(record);
|
|
2303
|
+
if (observation !== null) promptManifests.push(observation);
|
|
2304
|
+
}
|
|
2305
|
+
} catch {
|
|
2306
|
+
}
|
|
2307
|
+
try {
|
|
2308
|
+
const snapshots = await readFile2(join2(dirname2(ref.path), "context-snapshots.jsonl"), "utf8");
|
|
2309
|
+
for (const line of snapshots.split(/\r?\n/u)) {
|
|
2310
|
+
const record = parseJsonRecord2(line);
|
|
2311
|
+
if (record === null) continue;
|
|
2312
|
+
contextSnapshots.push({
|
|
2313
|
+
runtimeId: nullableString(record.runtimeId),
|
|
2314
|
+
modelId: nullableString(record.modelId),
|
|
2315
|
+
promptHash: nullableDigest(record.promptHash),
|
|
2316
|
+
toolSignature: nullableDigest(record.toolSignature)
|
|
2317
|
+
});
|
|
2318
|
+
}
|
|
2319
|
+
} catch {
|
|
2320
|
+
}
|
|
2321
|
+
}
|
|
2322
|
+
return { entries, compiledPromptHashes: [...new Set(compiledPromptHashes)], promptManifests, contextSnapshots };
|
|
2323
|
+
}
|
|
2324
|
+
function buildEvalTrackedMetrics(input) {
|
|
2325
|
+
const calls = assistantCalls(input.ledgerEntries);
|
|
2326
|
+
const compactionEntries = input.ledgerEntries.filter((entry) => entry.kind === "compactionSummary");
|
|
2327
|
+
const compactionUsage = compactionEntries.flatMap((entry) => {
|
|
2328
|
+
if (entry.kind !== "compactionSummary" || !isRecord5(entry.usage)) return [];
|
|
2329
|
+
return [entry.usage];
|
|
2330
|
+
});
|
|
2331
|
+
const modelCallReadings = calls.map(() => ledgerReading(1));
|
|
2332
|
+
for (const entry of compactionEntries) {
|
|
2333
|
+
if (entry.kind !== "compactionSummary") continue;
|
|
2334
|
+
const apiCalls = isRecord5(entry.usage) ? nonNegativeNumber(entry.usage.apiCalls) : null;
|
|
2335
|
+
modelCallReadings.push(apiCalls === null ? estimatedReading(1) : ledgerReading(apiCalls));
|
|
2336
|
+
}
|
|
2337
|
+
const uncachedReadings = calls.map(uncachedPrefillForCall);
|
|
2338
|
+
const cacheReadings = calls.map(cacheReadForCall);
|
|
2339
|
+
const generatedReadings = calls.map(generatedForCall);
|
|
2340
|
+
for (const usage of compactionUsage) {
|
|
2341
|
+
uncachedReadings.push(readingFromUsage(usage, "input"));
|
|
2342
|
+
cacheReadings.push(readingFromUsage(usage, "cacheRead"));
|
|
2343
|
+
generatedReadings.push(readingFromUsage(usage, "output"));
|
|
2344
|
+
}
|
|
2345
|
+
const reasoning = reasoningMetric(input.receipt, calls, compactionUsage);
|
|
2346
|
+
const receiptToolMetrics = input.receipt === null ? null : evalHarnessMetricsFromReceipt(input.receipt);
|
|
2347
|
+
const ledgerToolCalls = input.ledgerEntries.filter(
|
|
2348
|
+
(entry) => entry.kind === "message" && entry.role === "tool_call"
|
|
2349
|
+
).length;
|
|
2350
|
+
const ledgerToolErrors = input.ledgerEntries.filter((entry) => {
|
|
2351
|
+
if (entry.kind !== "message" || entry.role !== "tool_result" || !isRecord5(entry.payload)) return false;
|
|
2352
|
+
return entry.payload.isError === true || entry.payload.outcome === "error";
|
|
2353
|
+
}).length;
|
|
2354
|
+
const receiptToolErrors = input.receipt?.toolStats.reduce((sum2, stat) => sum2 + finiteNonNegative(stat.errors), 0);
|
|
2355
|
+
const expectedColdReasons = expectedColdReasonMetrics(calls);
|
|
2356
|
+
return {
|
|
2357
|
+
modelCalls: sumReadings(modelCallReadings, "ledger"),
|
|
2358
|
+
uncachedPrefillTokens: sumReadings(uncachedReadings, "estimated"),
|
|
2359
|
+
cacheReadTokens: sumReadings(cacheReadings, "estimated"),
|
|
2360
|
+
generatedTokens: sumReadings(generatedReadings, "estimated"),
|
|
2361
|
+
reasoningTokens: reasoning,
|
|
2362
|
+
toolCalls: receiptToolMetrics === null ? { value: ledgerToolCalls, source: "ledger" } : { value: receiptToolMetrics.toolCalls, source: "receipt" },
|
|
2363
|
+
toolErrors: receiptToolErrors === void 0 ? { value: ledgerToolErrors, source: "ledger" } : { value: receiptToolErrors, source: "receipt" },
|
|
2364
|
+
ttftMsFirstCall: firstCallTtft(calls),
|
|
2365
|
+
wallClockMs: wallClockMetric(input.receipt, input.fallbackWallClockMs),
|
|
2366
|
+
contextTokensAtEnd: contextTokensAtEnd(calls, compactionEntries),
|
|
2367
|
+
compactions: { value: compactionEntries.length, source: "ledger" },
|
|
2368
|
+
expectedColdReasons
|
|
2369
|
+
};
|
|
2370
|
+
}
|
|
2371
|
+
function emptyEvalTrackedMetrics(source = "estimated") {
|
|
2372
|
+
const zero = () => ({ value: 0, source });
|
|
2373
|
+
return {
|
|
2374
|
+
modelCalls: zero(),
|
|
2375
|
+
uncachedPrefillTokens: zero(),
|
|
2376
|
+
cacheReadTokens: zero(),
|
|
2377
|
+
generatedTokens: zero(),
|
|
2378
|
+
reasoningTokens: { value: null, source },
|
|
2379
|
+
toolCalls: zero(),
|
|
2380
|
+
toolErrors: zero(),
|
|
2381
|
+
ttftMsFirstCall: zero(),
|
|
2382
|
+
wallClockMs: zero(),
|
|
2383
|
+
contextTokensAtEnd: zero(),
|
|
2384
|
+
compactions: zero(),
|
|
2385
|
+
expectedColdReasons: {}
|
|
2386
|
+
};
|
|
2387
|
+
}
|
|
2388
|
+
function assistantCalls(entries) {
|
|
2389
|
+
return entries.flatMap((entry) => {
|
|
2390
|
+
if (entry.kind !== "message" || entry.role !== "assistant" || !isRecord5(entry.payload)) return [];
|
|
2391
|
+
const promptCache = recordField(entry.payload, "promptCache");
|
|
2392
|
+
const timing = recordField(entry.payload, "timing");
|
|
2393
|
+
const usage = recordField(entry.payload, "usage");
|
|
2394
|
+
if (promptCache === null && timing === null && usage === null) return [];
|
|
2395
|
+
return [
|
|
2396
|
+
{
|
|
2397
|
+
payload: entry.payload,
|
|
2398
|
+
promptCache,
|
|
2399
|
+
backend: promptCache === null ? null : recordField(promptCache, "backend"),
|
|
2400
|
+
timing,
|
|
2401
|
+
usage
|
|
2402
|
+
}
|
|
2403
|
+
];
|
|
2404
|
+
});
|
|
2405
|
+
}
|
|
2406
|
+
function uncachedPrefillForCall(call) {
|
|
2407
|
+
const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
|
|
2408
|
+
const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
|
|
2409
|
+
if (promptTokens !== null && cachedTokens !== null && cachedTokens <= promptTokens) {
|
|
2410
|
+
return ledgerReading(promptTokens - cachedTokens);
|
|
2411
|
+
}
|
|
2412
|
+
const piInput = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
|
|
2413
|
+
if (piInput !== null) return ledgerReading(piInput);
|
|
2414
|
+
const legacyInput = call.usage === null ? null : nonNegativeNumber(call.usage.input);
|
|
2415
|
+
return legacyInput === null ? estimatedReading(0) : estimatedReading(legacyInput);
|
|
2416
|
+
}
|
|
2417
|
+
function cacheReadForCall(call) {
|
|
2418
|
+
const cachedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.cachedTokens);
|
|
2419
|
+
if (cachedTokens !== null) return ledgerReading(cachedTokens);
|
|
2420
|
+
const piCacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
|
|
2421
|
+
if (piCacheRead !== null) return ledgerReading(piCacheRead);
|
|
2422
|
+
const legacyCacheRead = call.usage === null ? null : nonNegativeNumber(call.usage.cacheRead);
|
|
2423
|
+
return legacyCacheRead === null ? estimatedReading(0) : estimatedReading(legacyCacheRead);
|
|
2424
|
+
}
|
|
2425
|
+
function generatedForCall(call) {
|
|
2426
|
+
const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
|
|
2427
|
+
if (predictedTokens !== null) return ledgerReading(predictedTokens);
|
|
2428
|
+
const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
|
|
2429
|
+
return output === null ? estimatedReading(0) : ledgerReading(output);
|
|
2430
|
+
}
|
|
2431
|
+
function readingFromUsage(usage, field) {
|
|
2432
|
+
const value = nonNegativeNumber(usage[field]);
|
|
2433
|
+
return value === null ? estimatedReading(0) : ledgerReading(value);
|
|
2434
|
+
}
|
|
2435
|
+
function reasoningMetric(receipt, calls, compactionUsage) {
|
|
2436
|
+
if (receipt !== null && typeof receipt.reasoningTokenCount === "number") {
|
|
2437
|
+
return { value: finiteNonNegative(receipt.reasoningTokenCount), source: "receipt" };
|
|
2438
|
+
}
|
|
2439
|
+
let total = 0;
|
|
2440
|
+
let measured = false;
|
|
2441
|
+
for (const call of calls) {
|
|
2442
|
+
const value = extractReasoningTokens(call.usage);
|
|
2443
|
+
if (value === null) continue;
|
|
2444
|
+
measured = true;
|
|
2445
|
+
total += finiteNonNegative(value);
|
|
2446
|
+
}
|
|
2447
|
+
for (const usage of compactionUsage) {
|
|
2448
|
+
const value = nonNegativeNumber(usage.reasoning);
|
|
2449
|
+
if (value === null) continue;
|
|
2450
|
+
measured = true;
|
|
2451
|
+
total += value;
|
|
2452
|
+
}
|
|
2453
|
+
return measured ? { value: total, source: "ledger" } : { value: null, source: "estimated" };
|
|
2454
|
+
}
|
|
2455
|
+
function firstCallTtft(calls) {
|
|
2456
|
+
const first = calls[0];
|
|
2457
|
+
const value = first?.timing === null || first?.timing === void 0 ? null : nonNegativeNumber(first.timing.ttftMs);
|
|
2458
|
+
return value === null ? { value: 0, source: "estimated" } : { value, source: "ledger" };
|
|
2459
|
+
}
|
|
2460
|
+
function wallClockMetric(receipt, fallback) {
|
|
2461
|
+
if (receipt !== null) {
|
|
2462
|
+
const started = Date.parse(receipt.startedAt);
|
|
2463
|
+
const ended = Date.parse(receipt.endedAt);
|
|
2464
|
+
if (Number.isFinite(started) && Number.isFinite(ended) && ended >= started) {
|
|
2465
|
+
return { value: ended - started, source: "receipt" };
|
|
2466
|
+
}
|
|
2467
|
+
}
|
|
2468
|
+
return { value: finiteNonNegative(fallback), source: "estimated" };
|
|
2469
|
+
}
|
|
2470
|
+
function contextTokensAtEnd(calls, compactions) {
|
|
2471
|
+
const lastCall = calls.at(-1);
|
|
2472
|
+
if (lastCall !== void 0) {
|
|
2473
|
+
const lastReading = contextForCall(lastCall);
|
|
2474
|
+
if (lastReading !== null && lastReading > 0) return { value: lastReading, source: "ledger" };
|
|
2475
|
+
if (lastReading === 0) {
|
|
2476
|
+
for (const call of [...calls.slice(0, -1)].reverse()) {
|
|
2477
|
+
const reading = contextForCall(call);
|
|
2478
|
+
if (reading !== null && reading > 0) return { value: reading, source: "ledger" };
|
|
2479
|
+
}
|
|
2480
|
+
}
|
|
2481
|
+
}
|
|
2482
|
+
const lastCompaction = compactions.at(-1);
|
|
2483
|
+
if (lastCompaction?.kind === "compactionSummary") {
|
|
2484
|
+
const tokensAfter = nonNegativeNumber(lastCompaction.tokensAfter);
|
|
2485
|
+
if (tokensAfter !== null) return { value: tokensAfter, source: "ledger" };
|
|
2486
|
+
}
|
|
2487
|
+
return { value: 0, source: "estimated" };
|
|
2488
|
+
}
|
|
2489
|
+
function contextForCall(call) {
|
|
2490
|
+
const promptTokens = call.backend === null ? null : nonNegativeNumber(call.backend.promptTokens);
|
|
2491
|
+
const predictedTokens = call.backend === null ? null : nonNegativeNumber(call.backend.predictedTokens);
|
|
2492
|
+
if (promptTokens !== null && predictedTokens !== null) return promptTokens + predictedTokens;
|
|
2493
|
+
const input = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.input);
|
|
2494
|
+
const cacheRead = call.promptCache === null ? null : nonNegativeNumber(call.promptCache.cacheRead);
|
|
2495
|
+
const output = call.usage === null ? null : nonNegativeNumber(call.usage.output);
|
|
2496
|
+
return input === null || cacheRead === null || output === null ? null : input + cacheRead + output;
|
|
2497
|
+
}
|
|
2498
|
+
function expectedColdReasonMetrics(calls) {
|
|
2499
|
+
const counts = /* @__PURE__ */ new Map();
|
|
2500
|
+
for (const call of calls) {
|
|
2501
|
+
const reasons = call.promptCache?.expectedColdReasons;
|
|
2502
|
+
if (!Array.isArray(reasons)) continue;
|
|
2503
|
+
const unique = new Set(reasons.filter((reason) => typeof reason === "string" && reason.length > 0));
|
|
2504
|
+
for (const reason of unique) counts.set(reason, (counts.get(reason) ?? 0) + 1);
|
|
2505
|
+
}
|
|
2506
|
+
return Object.fromEntries(
|
|
2507
|
+
[...counts.entries()].sort(([left], [right]) => left.localeCompare(right)).map(([reason, value]) => [reason, { value, source: "ledger" }])
|
|
2508
|
+
);
|
|
2509
|
+
}
|
|
2510
|
+
function sumReadings(readings, emptySource) {
|
|
2511
|
+
if (readings.length === 0) return { value: 0, source: emptySource };
|
|
2512
|
+
return {
|
|
2513
|
+
value: readings.reduce((sum2, reading) => sum2 + reading.value, 0),
|
|
2514
|
+
source: readings.some((reading) => reading.source === "estimated") ? "estimated" : "ledger"
|
|
2515
|
+
};
|
|
2516
|
+
}
|
|
2517
|
+
function ledgerReading(value) {
|
|
2518
|
+
return { value: finiteNonNegative(value), source: "ledger" };
|
|
2519
|
+
}
|
|
2520
|
+
function estimatedReading(value) {
|
|
2521
|
+
return { value: finiteNonNegative(value), source: "estimated" };
|
|
2522
|
+
}
|
|
2523
|
+
function finiteNonNegative(value) {
|
|
2524
|
+
return Number.isFinite(value) && value >= 0 ? value : 0;
|
|
2525
|
+
}
|
|
2526
|
+
function nonNegativeNumber(value) {
|
|
2527
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
|
|
2528
|
+
}
|
|
2529
|
+
function recordField(record, field) {
|
|
2530
|
+
return isRecord5(record[field]) ? record[field] : null;
|
|
2531
|
+
}
|
|
2532
|
+
function parseJsonRecord2(line) {
|
|
2533
|
+
if (line.trim().length === 0) return null;
|
|
2534
|
+
try {
|
|
2535
|
+
const parsed = JSON.parse(line);
|
|
2536
|
+
return isRecord5(parsed) ? parsed : null;
|
|
2537
|
+
} catch {
|
|
2538
|
+
return null;
|
|
2539
|
+
}
|
|
2540
|
+
}
|
|
2541
|
+
function promptManifestObservation(record) {
|
|
2542
|
+
const systemPromptHash = nullableDigest(record.systemPromptHash);
|
|
2543
|
+
if (systemPromptHash === null || !Array.isArray(record.fragments)) return null;
|
|
2544
|
+
const fragments = record.fragments.flatMap((entry) => {
|
|
2545
|
+
if (!isRecord5(entry) || typeof entry.id !== "string") return [];
|
|
2546
|
+
const contentHash = nullableDigest(entry.contentHash);
|
|
2547
|
+
return contentHash === null ? [] : [{ id: entry.id, contentHash }];
|
|
2548
|
+
});
|
|
2549
|
+
const preload = record.projectPreload;
|
|
2550
|
+
const projectPreload = preload === null ? null : isRecord5(preload) && (preload.mode === "full" || preload.mode === "synopsis" || preload.mode === "none") && typeof preload.chars === "number" && Number.isInteger(preload.chars) && typeof preload.lines === "number" && Number.isInteger(preload.lines) && typeof preload.nearLimit === "boolean" && typeof preload.label === "string" ? {
|
|
2551
|
+
mode: preload.mode,
|
|
2552
|
+
chars: preload.chars,
|
|
2553
|
+
lines: preload.lines,
|
|
2554
|
+
reason: nullableString(preload.reason),
|
|
2555
|
+
nearLimit: preload.nearLimit,
|
|
2556
|
+
label: preload.label
|
|
2557
|
+
} : null;
|
|
2558
|
+
return {
|
|
2559
|
+
systemPromptHash,
|
|
2560
|
+
thinkingLevel: nullableString(record.thinkingLevel),
|
|
2561
|
+
projectPreload,
|
|
2562
|
+
fragments
|
|
2563
|
+
};
|
|
2564
|
+
}
|
|
2565
|
+
function nullableString(value) {
|
|
2566
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
2567
|
+
}
|
|
2568
|
+
function nullableDigest(value) {
|
|
2569
|
+
return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
|
|
2570
|
+
}
|
|
2571
|
+
function isRecord5(value) {
|
|
2572
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2573
|
+
}
|
|
2574
|
+
|
|
1298
2575
|
// src/domains/eval/runners/clio-run.ts
|
|
1299
2576
|
init_esm_shims();
|
|
2577
|
+
import { isAbsolute, relative, resolve as resolve2, sep } from "node:path";
|
|
1300
2578
|
|
|
1301
2579
|
// src/domains/eval/metrics/evidence.ts
|
|
1302
2580
|
init_esm_shims();
|
|
1303
2581
|
import { readFileSync as readFileSync3 } from "node:fs";
|
|
1304
|
-
import { join as
|
|
2582
|
+
import { join as join3 } from "node:path";
|
|
1305
2583
|
function dispatchScopeMetrics(receipt) {
|
|
1306
2584
|
const scope = receipt.pathScope;
|
|
1307
2585
|
if (scope === void 0) return {};
|
|
1308
2586
|
const entries = [...scope.workingContextPaths, ...scope.writeBoundaries];
|
|
1309
2587
|
const evidence = entries.flatMap((entry) => entry.evidence);
|
|
1310
|
-
const
|
|
2588
|
+
const count2 = (source) => evidence.filter((entry) => entry.source === source).length;
|
|
1311
2589
|
return {
|
|
1312
2590
|
"dispatch.scope.mode": scope.mode,
|
|
1313
2591
|
"dispatch.scope.inferredPathCount": entries.filter(
|
|
@@ -1316,9 +2594,9 @@ function dispatchScopeMetrics(receipt) {
|
|
|
1316
2594
|
"dispatch.scope.derivedPathCount": entries.filter(
|
|
1317
2595
|
(entry) => entry.evidence.some((item) => item.provenance === "derived")
|
|
1318
2596
|
).length,
|
|
1319
|
-
"dispatch.scope.source.task":
|
|
1320
|
-
"dispatch.scope.source.briefing":
|
|
1321
|
-
"dispatch.scope.source.writeRoots":
|
|
2597
|
+
"dispatch.scope.source.task": count2("task"),
|
|
2598
|
+
"dispatch.scope.source.briefing": count2("briefing"),
|
|
2599
|
+
"dispatch.scope.source.writeRoots": count2("writeRoots")
|
|
1322
2600
|
};
|
|
1323
2601
|
}
|
|
1324
2602
|
function receiptFromRunJsonStdout(stdout) {
|
|
@@ -1365,7 +2643,7 @@ function evidenceTrustMetrics(receipt, envelope) {
|
|
|
1365
2643
|
}
|
|
1366
2644
|
function readRunEnvelopeForReceipt(receipt, stateDir) {
|
|
1367
2645
|
try {
|
|
1368
|
-
const parsed = JSON.parse(readFileSync3(
|
|
2646
|
+
const parsed = JSON.parse(readFileSync3(join3(stateDir, "runs.json"), "utf8"));
|
|
1369
2647
|
if (!Array.isArray(parsed)) return null;
|
|
1370
2648
|
const row = parsed.find(
|
|
1371
2649
|
(entry) => typeof entry === "object" && entry !== null && entry.id === receipt.runId
|
|
@@ -1396,10 +2674,10 @@ function createTokenUsageFold() {
|
|
|
1396
2674
|
} catch {
|
|
1397
2675
|
return;
|
|
1398
2676
|
}
|
|
1399
|
-
if (!
|
|
1400
|
-
const message =
|
|
2677
|
+
if (!isRecord6(event) || event.type !== "message_end") return;
|
|
2678
|
+
const message = isRecord6(event.message) ? event.message : void 0;
|
|
1401
2679
|
if (message === void 0 || message.role !== "assistant") return;
|
|
1402
|
-
const usage =
|
|
2680
|
+
const usage = isRecord6(message.usage) ? message.usage : void 0;
|
|
1403
2681
|
if (usage === void 0) return;
|
|
1404
2682
|
measured = true;
|
|
1405
2683
|
const input = numberField2(usage, "input");
|
|
@@ -1412,7 +2690,7 @@ function createTokenUsageFold() {
|
|
|
1412
2690
|
tokens.cacheRead += cacheRead;
|
|
1413
2691
|
tokens.cacheWrite += cacheWrite;
|
|
1414
2692
|
tokens.total += totalTokens > 0 ? totalTokens : input + output + cacheRead + cacheWrite;
|
|
1415
|
-
if (
|
|
2693
|
+
if (isRecord6(usage.cost)) costUsd += numberField2(usage.cost, "total");
|
|
1416
2694
|
};
|
|
1417
2695
|
return {
|
|
1418
2696
|
push(chunk) {
|
|
@@ -1462,14 +2740,115 @@ function numberField2(record, field) {
|
|
|
1462
2740
|
const value = record[field];
|
|
1463
2741
|
return typeof value === "number" && Number.isFinite(value) ? value : 0;
|
|
1464
2742
|
}
|
|
1465
|
-
function
|
|
2743
|
+
function isRecord6(value) {
|
|
1466
2744
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1467
2745
|
}
|
|
1468
2746
|
|
|
1469
2747
|
// src/domains/eval/runners/external-command.ts
|
|
1470
2748
|
init_esm_shims();
|
|
1471
2749
|
import { spawn } from "node:child_process";
|
|
2750
|
+
import { performance as performance2 } from "node:perf_hooks";
|
|
2751
|
+
|
|
2752
|
+
// src/domains/eval/metrics/call-ledger-stream.ts
|
|
2753
|
+
init_esm_shims();
|
|
1472
2754
|
import { performance } from "node:perf_hooks";
|
|
2755
|
+
function createEvalCallLedgerFold(now = () => performance.now()) {
|
|
2756
|
+
const entries = [];
|
|
2757
|
+
let pending = "";
|
|
2758
|
+
let activeStartedAt = null;
|
|
2759
|
+
let activeFirstOutputAt = null;
|
|
2760
|
+
const consume = (line) => {
|
|
2761
|
+
const event = parseRecord(line);
|
|
2762
|
+
if (event === null) return;
|
|
2763
|
+
if (event.type === "message_start" && isAssistantMessage(event.message)) {
|
|
2764
|
+
activeStartedAt = now();
|
|
2765
|
+
activeFirstOutputAt = null;
|
|
2766
|
+
return;
|
|
2767
|
+
}
|
|
2768
|
+
if (event.type === "message_update" && activeStartedAt !== null && activeFirstOutputAt === null) {
|
|
2769
|
+
activeFirstOutputAt = now();
|
|
2770
|
+
return;
|
|
2771
|
+
}
|
|
2772
|
+
if (event.type !== "message_end" || !isAssistantMessage(event.message)) return;
|
|
2773
|
+
const message = event.message;
|
|
2774
|
+
const usage = isRecord7(message.usage) ? message.usage : null;
|
|
2775
|
+
if (usage === null) {
|
|
2776
|
+
activeStartedAt = null;
|
|
2777
|
+
activeFirstOutputAt = null;
|
|
2778
|
+
return;
|
|
2779
|
+
}
|
|
2780
|
+
const endedAt = now();
|
|
2781
|
+
const promptCache = {
|
|
2782
|
+
input: nonNegativeNumber2(usage.input) ?? 0,
|
|
2783
|
+
cacheRead: nonNegativeNumber2(usage.cacheRead) ?? 0,
|
|
2784
|
+
cacheWrite: nonNegativeNumber2(usage.cacheWrite) ?? 0,
|
|
2785
|
+
backendVerdict: "unknown"
|
|
2786
|
+
};
|
|
2787
|
+
if (isRecord7(message.backendTimings)) promptCache.backend = structuredClone(message.backendTimings);
|
|
2788
|
+
const previous = entries.at(-1);
|
|
2789
|
+
entries.push({
|
|
2790
|
+
kind: "message",
|
|
2791
|
+
role: "assistant",
|
|
2792
|
+
turnId: `eval-call-${entries.length + 1}`,
|
|
2793
|
+
parentTurnId: previous?.turnId ?? null,
|
|
2794
|
+
timestamp: messageTimestamp(message.timestamp),
|
|
2795
|
+
payload: {
|
|
2796
|
+
promptCache,
|
|
2797
|
+
timing: {
|
|
2798
|
+
ttftMs: activeStartedAt === null ? null : Math.round(Math.max(0, (activeFirstOutputAt ?? endedAt) - activeStartedAt)),
|
|
2799
|
+
apiMs: activeStartedAt === null ? 0 : Math.round(Math.max(0, endedAt - activeStartedAt))
|
|
2800
|
+
},
|
|
2801
|
+
usage: structuredClone(usage)
|
|
2802
|
+
}
|
|
2803
|
+
});
|
|
2804
|
+
activeStartedAt = null;
|
|
2805
|
+
activeFirstOutputAt = null;
|
|
2806
|
+
};
|
|
2807
|
+
return {
|
|
2808
|
+
push(chunk) {
|
|
2809
|
+
pending += chunk;
|
|
2810
|
+
for (; ; ) {
|
|
2811
|
+
const newline = pending.indexOf("\n");
|
|
2812
|
+
if (newline === -1) break;
|
|
2813
|
+
consume(pending.slice(0, newline).replace(/\r$/u, ""));
|
|
2814
|
+
pending = pending.slice(newline + 1);
|
|
2815
|
+
}
|
|
2816
|
+
},
|
|
2817
|
+
entries() {
|
|
2818
|
+
if (pending.length > 0) {
|
|
2819
|
+
consume(pending.replace(/\r$/u, ""));
|
|
2820
|
+
pending = "";
|
|
2821
|
+
}
|
|
2822
|
+
return structuredClone(entries);
|
|
2823
|
+
}
|
|
2824
|
+
};
|
|
2825
|
+
}
|
|
2826
|
+
function isAssistantMessage(value) {
|
|
2827
|
+
return isRecord7(value) && value.role === "assistant";
|
|
2828
|
+
}
|
|
2829
|
+
function messageTimestamp(value) {
|
|
2830
|
+
const milliseconds = nonNegativeNumber2(value);
|
|
2831
|
+
if (milliseconds === null) return (/* @__PURE__ */ new Date(0)).toISOString();
|
|
2832
|
+
const date = new Date(milliseconds);
|
|
2833
|
+
return Number.isFinite(date.getTime()) ? date.toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
|
|
2834
|
+
}
|
|
2835
|
+
function nonNegativeNumber2(value) {
|
|
2836
|
+
return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null;
|
|
2837
|
+
}
|
|
2838
|
+
function parseRecord(line) {
|
|
2839
|
+
if (line.trim().length === 0) return null;
|
|
2840
|
+
try {
|
|
2841
|
+
const value = JSON.parse(line);
|
|
2842
|
+
return isRecord7(value) ? value : null;
|
|
2843
|
+
} catch {
|
|
2844
|
+
return null;
|
|
2845
|
+
}
|
|
2846
|
+
}
|
|
2847
|
+
function isRecord7(value) {
|
|
2848
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2849
|
+
}
|
|
2850
|
+
|
|
2851
|
+
// src/domains/eval/runners/external-command.ts
|
|
1473
2852
|
var OUTPUT_LIMIT = 2e5;
|
|
1474
2853
|
var OUTPUT_HEAD_LIMIT = 2e4;
|
|
1475
2854
|
var OUTPUT_TRUNCATION_MARKER = "\n[output middle truncated; tail preserved]\n";
|
|
@@ -1484,6 +2863,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1484
2863
|
let usage = UNMEASURED_TOKEN_USAGE;
|
|
1485
2864
|
let streamInvariants = EMPTY_STREAM_INVARIANTS;
|
|
1486
2865
|
let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
|
|
2866
|
+
const ledgerEntries = [];
|
|
1487
2867
|
for (const command of commands) {
|
|
1488
2868
|
const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1489
2869
|
stdout = appendLimited(stdout, result.stdout);
|
|
@@ -1492,6 +2872,7 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1492
2872
|
usage = addTokenStreamUsage(usage, result.usage);
|
|
1493
2873
|
streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
|
|
1494
2874
|
fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
|
|
2875
|
+
ledgerEntries.push(...result.ledgerEntries);
|
|
1495
2876
|
if (result.exitCode !== 0) {
|
|
1496
2877
|
return {
|
|
1497
2878
|
assignmentId: null,
|
|
@@ -1507,7 +2888,8 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1507
2888
|
...fleetLoopMetricEntries(fleetLoops),
|
|
1508
2889
|
"verifier.exitCode": result.exitCode
|
|
1509
2890
|
},
|
|
1510
|
-
artifacts: {}
|
|
2891
|
+
artifacts: {},
|
|
2892
|
+
ledgerEntries
|
|
1511
2893
|
};
|
|
1512
2894
|
}
|
|
1513
2895
|
}
|
|
@@ -1525,18 +2907,20 @@ async function runExternalCommandRunner(runner, cwd, timeoutMs, env) {
|
|
|
1525
2907
|
...fleetLoopMetricEntries(fleetLoops),
|
|
1526
2908
|
"verifier.exitCode": 0
|
|
1527
2909
|
},
|
|
1528
|
-
artifacts: {}
|
|
2910
|
+
artifacts: {},
|
|
2911
|
+
ledgerEntries
|
|
1529
2912
|
};
|
|
1530
2913
|
}
|
|
1531
2914
|
function runShellCommand(command, cwd, timeoutMs, env) {
|
|
1532
|
-
const started =
|
|
1533
|
-
return new Promise((
|
|
2915
|
+
const started = performance2.now();
|
|
2916
|
+
return new Promise((resolve9) => {
|
|
1534
2917
|
let stdout = "";
|
|
1535
2918
|
let stderr = "";
|
|
1536
2919
|
const metricCapture = createJsonlMetricCapture();
|
|
1537
2920
|
const usageFold = createTokenUsageFold();
|
|
1538
2921
|
const streamFold = createStreamInvariantFold();
|
|
1539
2922
|
const fleetLoopFold = createFleetLoopFold();
|
|
2923
|
+
const callLedgerFold = createEvalCallLedgerFold();
|
|
1540
2924
|
let timedOut = false;
|
|
1541
2925
|
let settled = false;
|
|
1542
2926
|
const child = spawn(command, {
|
|
@@ -1555,6 +2939,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1555
2939
|
usageFold.push(chunk);
|
|
1556
2940
|
streamFold.push(chunk);
|
|
1557
2941
|
fleetLoopFold.push(chunk);
|
|
2942
|
+
callLedgerFold.push(chunk);
|
|
1558
2943
|
});
|
|
1559
2944
|
child.stderr.on("data", (chunk) => {
|
|
1560
2945
|
stderr = appendLimited(stderr, chunk);
|
|
@@ -1568,7 +2953,7 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1568
2953
|
if (settled) return;
|
|
1569
2954
|
settled = true;
|
|
1570
2955
|
clearTimeout(timer);
|
|
1571
|
-
|
|
2956
|
+
resolve9({
|
|
1572
2957
|
command,
|
|
1573
2958
|
exitCode,
|
|
1574
2959
|
stdout,
|
|
@@ -1576,8 +2961,9 @@ function runShellCommand(command, cwd, timeoutMs, env) {
|
|
|
1576
2961
|
usage: usageFold.usage(),
|
|
1577
2962
|
streamInvariants: streamFold.invariants(),
|
|
1578
2963
|
fleetLoops: fleetLoopFold.observation(),
|
|
2964
|
+
ledgerEntries: callLedgerFold.entries(),
|
|
1579
2965
|
stderr,
|
|
1580
|
-
wallTimeMs: Math.round(
|
|
2966
|
+
wallTimeMs: Math.round(performance2.now() - started),
|
|
1581
2967
|
timedOut
|
|
1582
2968
|
});
|
|
1583
2969
|
};
|
|
@@ -1609,7 +2995,7 @@ function createJsonlMetricCapture() {
|
|
|
1609
2995
|
} catch {
|
|
1610
2996
|
return;
|
|
1611
2997
|
}
|
|
1612
|
-
if (!
|
|
2998
|
+
if (!isRecord8(parsed)) return;
|
|
1613
2999
|
const compact = compactMetricEvent(parsed);
|
|
1614
3000
|
if (compact === null) return;
|
|
1615
3001
|
const encoded = JSON.stringify(compact);
|
|
@@ -1663,7 +3049,7 @@ function compactMetricEvent(event) {
|
|
|
1663
3049
|
type,
|
|
1664
3050
|
...stringField(event, "toolCallId") !== void 0 ? { toolCallId: stringField(event, "toolCallId") } : {},
|
|
1665
3051
|
toolName,
|
|
1666
|
-
...toolName === "dispatch" &&
|
|
3052
|
+
...toolName === "dispatch" && isRecord8(event.args) ? { args: event.args } : toolName === "read" && isRecord8(event.args) ? { args: boundedReadArgs(event.args) } : toolName === "code_nav" && isRecord8(event.args) ? { args: { mode: event.args.mode } } : {}
|
|
1667
3053
|
};
|
|
1668
3054
|
}
|
|
1669
3055
|
if (type === "tool_execution_end") {
|
|
@@ -1675,7 +3061,7 @@ function compactMetricEvent(event) {
|
|
|
1675
3061
|
...stringField(event, "outcome") !== void 0 ? { outcome: stringField(event, "outcome") } : {}
|
|
1676
3062
|
};
|
|
1677
3063
|
}
|
|
1678
|
-
if (type !== "clio_tool_finish" || !
|
|
3064
|
+
if (type !== "clio_tool_finish" || !isRecord8(event.payload)) return null;
|
|
1679
3065
|
return {
|
|
1680
3066
|
type,
|
|
1681
3067
|
payload: {
|
|
@@ -1685,7 +3071,14 @@ function compactMetricEvent(event) {
|
|
|
1685
3071
|
}
|
|
1686
3072
|
};
|
|
1687
3073
|
}
|
|
1688
|
-
function
|
|
3074
|
+
function boundedReadArgs(args) {
|
|
3075
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
3076
|
+
const value = args[field];
|
|
3077
|
+
if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4096) };
|
|
3078
|
+
}
|
|
3079
|
+
return {};
|
|
3080
|
+
}
|
|
3081
|
+
function isRecord8(value) {
|
|
1689
3082
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1690
3083
|
}
|
|
1691
3084
|
function stringField(record, field) {
|
|
@@ -1694,7 +3087,7 @@ function stringField(record, field) {
|
|
|
1694
3087
|
}
|
|
1695
3088
|
|
|
1696
3089
|
// src/domains/eval/runners/clio-run.ts
|
|
1697
|
-
async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
|
|
3090
|
+
async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env, readObservation) {
|
|
1698
3091
|
const prompt = runner.prompt ?? "";
|
|
1699
3092
|
const args = [
|
|
1700
3093
|
shellQuote(clioEntry),
|
|
@@ -1705,12 +3098,14 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1705
3098
|
shellQuote(target.id),
|
|
1706
3099
|
...target.model === void 0 ? [] : ["--model", shellQuote(target.model)],
|
|
1707
3100
|
...target.thinking === void 0 ? [] : ["--thinking", shellQuote(target.thinking)],
|
|
3101
|
+
...runner.autonomy === void 0 ? [] : ["--autonomy", runner.autonomy],
|
|
1708
3102
|
shellQuote(prompt)
|
|
1709
3103
|
];
|
|
1710
3104
|
const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1711
3105
|
const tokens = result.usage;
|
|
1712
3106
|
const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
|
|
1713
3107
|
const tools = toolCallMetricsFromJsonl(toolMetricStream);
|
|
3108
|
+
const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
|
|
1714
3109
|
const receipt = receiptFromRunJsonStdout(result.stdout);
|
|
1715
3110
|
const envelope = receipt === null ? null : readRunEnvelopeForReceipt(receipt, env?.CLIO_CODER_STATE_DIR ?? clioStateDir());
|
|
1716
3111
|
return {
|
|
@@ -1729,6 +3124,7 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1729
3124
|
"tools.totalCalls": tools.totalCalls,
|
|
1730
3125
|
"tools.failed": tools.failed,
|
|
1731
3126
|
"tools.blocked": tools.blocked,
|
|
3127
|
+
...behavioralTools,
|
|
1732
3128
|
"verifier.exitCode": result.exitCode,
|
|
1733
3129
|
...receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope }),
|
|
1734
3130
|
...receipt === null ? {} : { "evidence.qualityLabel": receipt.quality.typedValidations.length > 0 ? "measured" : "unmeasured" }
|
|
@@ -1736,8 +3132,11 @@ async function runClioRunRunner(runner, cwd, clioEntry, timeoutMs, target, env)
|
|
|
1736
3132
|
artifacts: {
|
|
1737
3133
|
stdout: result.stdout,
|
|
1738
3134
|
stderr: result.stderr,
|
|
3135
|
+
callLedger: JSON.stringify(result.ledgerEntries),
|
|
1739
3136
|
...receipt === null ? {} : { receipt: JSON.stringify(receipt) }
|
|
1740
|
-
}
|
|
3137
|
+
},
|
|
3138
|
+
receipt,
|
|
3139
|
+
ledgerEntries: result.ledgerEntries
|
|
1741
3140
|
};
|
|
1742
3141
|
}
|
|
1743
3142
|
function toolCallMetricsFromJsonl(stdout) {
|
|
@@ -1750,7 +3149,7 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1750
3149
|
let event;
|
|
1751
3150
|
try {
|
|
1752
3151
|
const parsed = JSON.parse(line);
|
|
1753
|
-
if (!
|
|
3152
|
+
if (!isRecord9(parsed)) continue;
|
|
1754
3153
|
event = parsed;
|
|
1755
3154
|
} catch {
|
|
1756
3155
|
continue;
|
|
@@ -1764,7 +3163,7 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1764
3163
|
recordToolOutcome(executionEnds, toolOutcome(event) ?? (event.isError === true ? "error" : "ok"));
|
|
1765
3164
|
continue;
|
|
1766
3165
|
}
|
|
1767
|
-
if (event.type !== "clio_tool_finish" || !
|
|
3166
|
+
if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
|
|
1768
3167
|
const outcome = toolOutcome(event.payload);
|
|
1769
3168
|
if (outcome === void 0) continue;
|
|
1770
3169
|
const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId");
|
|
@@ -1776,6 +3175,95 @@ function toolCallMetricsFromJsonl(stdout) {
|
|
|
1776
3175
|
}
|
|
1777
3176
|
return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
|
|
1778
3177
|
}
|
|
3178
|
+
function toolBehaviorMetricEntriesFromJsonl(stdout, cwd, readObservation) {
|
|
3179
|
+
const starts = /* @__PURE__ */ new Map();
|
|
3180
|
+
const readPaths = /* @__PURE__ */ new Set();
|
|
3181
|
+
const executionEnds = [];
|
|
3182
|
+
const canonicalFinishes = [];
|
|
3183
|
+
const seenExecution = /* @__PURE__ */ new Set();
|
|
3184
|
+
const seenCanonical = /* @__PURE__ */ new Set();
|
|
3185
|
+
for (const line of stdout.split(/\r?\n/)) {
|
|
3186
|
+
if (line.trim().length === 0) continue;
|
|
3187
|
+
let event;
|
|
3188
|
+
try {
|
|
3189
|
+
const parsed = JSON.parse(line);
|
|
3190
|
+
if (!isRecord9(parsed)) continue;
|
|
3191
|
+
event = parsed;
|
|
3192
|
+
} catch {
|
|
3193
|
+
continue;
|
|
3194
|
+
}
|
|
3195
|
+
if (event.type === "tool_execution_start") {
|
|
3196
|
+
const callId2 = stringField2(event, "toolCallId");
|
|
3197
|
+
const tool2 = stringField2(event, "toolName");
|
|
3198
|
+
if (callId2 === void 0 || tool2 === void 0) continue;
|
|
3199
|
+
const path = tool2 === "read" && isRecord9(event.args) ? toolPath(event.args) : null;
|
|
3200
|
+
starts.set(callId2, { tool: tool2, path });
|
|
3201
|
+
if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
|
|
3202
|
+
continue;
|
|
3203
|
+
}
|
|
3204
|
+
if (event.type === "tool_execution_end") {
|
|
3205
|
+
const callId2 = stringField2(event, "toolCallId") ?? null;
|
|
3206
|
+
if (callId2 !== null && seenExecution.has(callId2)) continue;
|
|
3207
|
+
if (callId2 !== null) seenExecution.add(callId2);
|
|
3208
|
+
const tool2 = stringField2(event, "toolName") ?? (callId2 === null ? void 0 : starts.get(callId2)?.tool);
|
|
3209
|
+
if (tool2 === void 0) continue;
|
|
3210
|
+
executionEnds.push({ callId: callId2, tool: tool2, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
|
|
3211
|
+
continue;
|
|
3212
|
+
}
|
|
3213
|
+
if (event.type !== "clio_tool_finish" || !isRecord9(event.payload)) continue;
|
|
3214
|
+
const outcome = toolOutcome(event.payload);
|
|
3215
|
+
const tool = stringField2(event.payload, "tool");
|
|
3216
|
+
if (outcome === void 0 || tool === void 0) continue;
|
|
3217
|
+
const callId = stringField2(event.payload, "toolCallId") ?? stringField2(event, "toolCallId") ?? null;
|
|
3218
|
+
if (callId !== null && seenCanonical.has(callId)) continue;
|
|
3219
|
+
if (callId !== null) seenCanonical.add(callId);
|
|
3220
|
+
canonicalFinishes.push({ callId, tool, outcome });
|
|
3221
|
+
}
|
|
3222
|
+
const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
|
|
3223
|
+
const calls = /* @__PURE__ */ new Map();
|
|
3224
|
+
const blocked = /* @__PURE__ */ new Map();
|
|
3225
|
+
for (const terminal of terminals) {
|
|
3226
|
+
const tool = metricToolName(terminal.tool);
|
|
3227
|
+
calls.set(tool, (calls.get(tool) ?? 0) + 1);
|
|
3228
|
+
if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
|
|
3229
|
+
}
|
|
3230
|
+
const namedTools = /* @__PURE__ */ new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
|
|
3231
|
+
const entries = { "tools.read.distinctPaths": readPaths.size };
|
|
3232
|
+
for (const tool of [...namedTools].sort()) {
|
|
3233
|
+
entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
|
|
3234
|
+
entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
|
|
3235
|
+
}
|
|
3236
|
+
if (readObservation !== void 0) {
|
|
3237
|
+
const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
3238
|
+
const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
3239
|
+
entries["tools.read.outsideAllowed"] = [...readPaths].filter(
|
|
3240
|
+
(path) => !allowed.some((root) => pathWithin(path, root))
|
|
3241
|
+
).length;
|
|
3242
|
+
entries["tools.read.decoyHits"] = [...readPaths].filter(
|
|
3243
|
+
(path) => decoys.some((root) => pathWithin(path, root))
|
|
3244
|
+
).length;
|
|
3245
|
+
}
|
|
3246
|
+
return entries;
|
|
3247
|
+
}
|
|
3248
|
+
function toolPath(args) {
|
|
3249
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
3250
|
+
const value = args[field];
|
|
3251
|
+
if (typeof value === "string" && value.length > 0 && value.length <= 4096) return value;
|
|
3252
|
+
}
|
|
3253
|
+
return null;
|
|
3254
|
+
}
|
|
3255
|
+
function normalizeObservedPath(cwd, path) {
|
|
3256
|
+
const absolute = resolve2(cwd, path);
|
|
3257
|
+
const local = relative(cwd, absolute);
|
|
3258
|
+
return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".").split(sep).join("/");
|
|
3259
|
+
}
|
|
3260
|
+
function pathWithin(path, root) {
|
|
3261
|
+
if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
|
|
3262
|
+
return path === root || path.startsWith(`${root}/`);
|
|
3263
|
+
}
|
|
3264
|
+
function metricToolName(tool) {
|
|
3265
|
+
return tool.toLowerCase().replaceAll(/[^a-z0-9_-]/gu, "_").slice(0, 64) || "unknown";
|
|
3266
|
+
}
|
|
1779
3267
|
function recordToolOutcome(metrics, outcome) {
|
|
1780
3268
|
metrics.totalCalls += 1;
|
|
1781
3269
|
if (outcome === "error") metrics.failed += 1;
|
|
@@ -1789,7 +3277,7 @@ function stringField2(record, field) {
|
|
|
1789
3277
|
const value = record[field];
|
|
1790
3278
|
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
1791
3279
|
}
|
|
1792
|
-
function
|
|
3280
|
+
function isRecord9(value) {
|
|
1793
3281
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1794
3282
|
}
|
|
1795
3283
|
|
|
@@ -1840,7 +3328,7 @@ function parseContextIndexOutput(stdout) {
|
|
|
1840
3328
|
// src/domains/eval/runners/context-init.ts
|
|
1841
3329
|
init_esm_shims();
|
|
1842
3330
|
import { existsSync as existsSync2, statSync } from "node:fs";
|
|
1843
|
-
import { join as
|
|
3331
|
+
import { join as join4 } from "node:path";
|
|
1844
3332
|
async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, env) {
|
|
1845
3333
|
const extraArgs = runner.args ?? [];
|
|
1846
3334
|
const command = [
|
|
@@ -1858,22 +3346,22 @@ async function runContextInitRunner(runner, cwd, clioEntry, timeoutMs, target, e
|
|
|
1858
3346
|
].map(shellQuote).join(" ");
|
|
1859
3347
|
const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
1860
3348
|
const payload = parseInitPayload(result.stdout);
|
|
1861
|
-
const candidateGeneration =
|
|
3349
|
+
const candidateGeneration = recordField2(payload, "generation");
|
|
1862
3350
|
const generation = isValidGenerationPayload(payload, candidateGeneration) ? candidateGeneration : null;
|
|
1863
3351
|
const routeError = generation ? generationRouteError(generation, target) : null;
|
|
1864
3352
|
const payloadError = generation ? routeError : "context-init runner did not receive a valid JSON generation result";
|
|
1865
3353
|
const exitCode = result.exitCode === 0 && payloadError ? 1 : result.exitCode;
|
|
1866
3354
|
const stderr = payloadError ? `${result.stderr}${result.stderr.endsWith("\n") || result.stderr.length === 0 ? "" : "\n"}${payloadError}
|
|
1867
3355
|
` : result.stderr;
|
|
1868
|
-
const run =
|
|
1869
|
-
const tokens =
|
|
3356
|
+
const run = recordField2(generation, "run");
|
|
3357
|
+
const tokens = recordField2(run, "tokens");
|
|
1870
3358
|
const effectiveTarget = stringField3(run, "targetId");
|
|
1871
3359
|
const effectiveModel = stringField3(run, "wireModelId");
|
|
1872
3360
|
const effectiveRuntime = stringField3(run, "runtimeId");
|
|
1873
3361
|
const effectiveRuntimeKind = stringField3(run, "runtimeKind");
|
|
1874
3362
|
const effectiveThinking = stringField3(run, "thinkingLevel");
|
|
1875
3363
|
const structuredOutputMode = stringField3(run, "structuredOutputMode");
|
|
1876
|
-
const clioMdPath =
|
|
3364
|
+
const clioMdPath = join4(cwd, "CLIO-CODER.md");
|
|
1877
3365
|
const clioMdBytes = existsSync2(clioMdPath) ? statSync(clioMdPath).size : 0;
|
|
1878
3366
|
return {
|
|
1879
3367
|
assignmentId: null,
|
|
@@ -1927,7 +3415,7 @@ function isNonnegativeFiniteNumber(value) {
|
|
|
1927
3415
|
return typeof value === "number" && Number.isFinite(value) && value >= 0;
|
|
1928
3416
|
}
|
|
1929
3417
|
function isValidRunPayload(value) {
|
|
1930
|
-
const run =
|
|
3418
|
+
const run = recordField2(value);
|
|
1931
3419
|
if (!run) return false;
|
|
1932
3420
|
for (const key of ["durationMs", "promptBytes", "outputBytes"]) {
|
|
1933
3421
|
if (!isNonnegativeFiniteNumber(run[key])) return false;
|
|
@@ -1936,7 +3424,7 @@ function isValidRunPayload(value) {
|
|
|
1936
3424
|
for (const key of ["toolCalls", "toolFailures", "toolBlocked"]) {
|
|
1937
3425
|
if (run[key] !== void 0 && !isNonnegativeFiniteNumber(run[key])) return false;
|
|
1938
3426
|
}
|
|
1939
|
-
const tokens =
|
|
3427
|
+
const tokens = recordField2(run, "tokens");
|
|
1940
3428
|
if (run.tokens !== void 0 && !tokens) return false;
|
|
1941
3429
|
if (tokens) {
|
|
1942
3430
|
for (const key of ["total", "input", "output", "cacheRead", "cacheWrite", "reasoning"]) {
|
|
@@ -1962,14 +3450,14 @@ function isValidGenerationPayload(payload, generation) {
|
|
|
1962
3450
|
}
|
|
1963
3451
|
const runPresent = generation.run !== void 0;
|
|
1964
3452
|
if (runPresent && !isValidRunPayload(generation.run)) return false;
|
|
1965
|
-
const run =
|
|
3453
|
+
const run = recordField2(generation, "run");
|
|
1966
3454
|
if (mode === "model" && (parserOutcome !== "parsed" || !hasReceiptIdentity(run))) return false;
|
|
1967
3455
|
if ((parserOutcome === "parsed" || parserOutcome === "rejected") && !runPresent) return false;
|
|
1968
3456
|
if (parserOutcome === "rejected" && !hasReceiptIdentity(run)) return false;
|
|
1969
3457
|
return true;
|
|
1970
3458
|
}
|
|
1971
3459
|
function generationRouteError(generation, target) {
|
|
1972
|
-
const run =
|
|
3460
|
+
const run = recordField2(generation, "run");
|
|
1973
3461
|
if (!run) return null;
|
|
1974
3462
|
const actualTarget = stringField3(run, "targetId");
|
|
1975
3463
|
const actualModel = stringField3(run, "wireModelId");
|
|
@@ -1988,34 +3476,101 @@ function generationRouteError(generation, target) {
|
|
|
1988
3476
|
function parseInitPayload(stdout) {
|
|
1989
3477
|
try {
|
|
1990
3478
|
const parsed = JSON.parse(stdout);
|
|
1991
|
-
return
|
|
3479
|
+
return recordField2(parsed);
|
|
1992
3480
|
} catch {
|
|
1993
3481
|
return null;
|
|
1994
3482
|
}
|
|
1995
3483
|
}
|
|
1996
|
-
function
|
|
3484
|
+
function recordField2(value, field) {
|
|
1997
3485
|
const selected = field && typeof value === "object" && value !== null && !Array.isArray(value) ? value[field] : value;
|
|
1998
3486
|
return typeof selected === "object" && selected !== null && !Array.isArray(selected) ? selected : null;
|
|
1999
3487
|
}
|
|
2000
3488
|
function numberField3(value, field) {
|
|
2001
|
-
const record =
|
|
3489
|
+
const record = recordField2(value);
|
|
2002
3490
|
const selected = record?.[field];
|
|
2003
3491
|
return typeof selected === "number" && Number.isFinite(selected) ? selected : null;
|
|
2004
3492
|
}
|
|
2005
3493
|
function stringField3(value, field) {
|
|
2006
|
-
const record =
|
|
3494
|
+
const record = recordField2(value);
|
|
2007
3495
|
const selected = record?.[field];
|
|
2008
3496
|
return typeof selected === "string" && selected.length > 0 ? selected : null;
|
|
2009
3497
|
}
|
|
2010
3498
|
|
|
3499
|
+
// src/domains/eval/schema/adapter.ts
|
|
3500
|
+
init_esm_shims();
|
|
3501
|
+
import { createHash as createHash3 } from "node:crypto";
|
|
3502
|
+
function adaptSuiteV2ResultToVerdictV1(result, trackedMetrics) {
|
|
3503
|
+
const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
|
|
3504
|
+
const outcome = result.pass ? "pass" : "fail";
|
|
3505
|
+
const graderExitCode = result.metrics["task.exitCode"];
|
|
3506
|
+
return parseEvalVerdictEnvelopeV1({
|
|
3507
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
3508
|
+
scenarioId: result.taskId,
|
|
3509
|
+
trialIndex: result.repeatIndex,
|
|
3510
|
+
outcome,
|
|
3511
|
+
machinery,
|
|
3512
|
+
reason: result.pass ? null : result.failureClass ?? "result_failed",
|
|
3513
|
+
trackedMetrics,
|
|
3514
|
+
behavioral: null,
|
|
3515
|
+
evidence: {
|
|
3516
|
+
assignmentId: result.assignmentId,
|
|
3517
|
+
terminalReceiptDigest: result.terminalReceiptDigest,
|
|
3518
|
+
graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null
|
|
3519
|
+
}
|
|
3520
|
+
});
|
|
3521
|
+
}
|
|
3522
|
+
function adaptSuiteV2ResultToBehaviorV1(result, verdict, scenario) {
|
|
3523
|
+
const requestedFacts = new Set(
|
|
3524
|
+
[...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
|
|
3525
|
+
(rule) => `${rule.fact.source}\0${rule.fact.key}`
|
|
3526
|
+
)
|
|
3527
|
+
);
|
|
3528
|
+
const observedSources = /* @__PURE__ */ new Set();
|
|
3529
|
+
const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
|
|
3530
|
+
if (value === null) return [];
|
|
3531
|
+
const source = metricFactSource(key);
|
|
3532
|
+
observedSources.add(source);
|
|
3533
|
+
if (!requestedFacts.has(`${source}\0${key}`)) return [];
|
|
3534
|
+
const serialized = JSON.stringify({ source, key, value });
|
|
3535
|
+
const digest = createHash3("sha256").update(serialized, "utf8").digest("hex");
|
|
3536
|
+
const fact = {
|
|
3537
|
+
id: `metric-${digest.slice(0, 16)}`,
|
|
3538
|
+
source,
|
|
3539
|
+
key,
|
|
3540
|
+
value,
|
|
3541
|
+
evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1e3) }
|
|
3542
|
+
};
|
|
3543
|
+
return [fact];
|
|
3544
|
+
});
|
|
3545
|
+
const allSources = ["transcript", "tool", "receipt", "grader"];
|
|
3546
|
+
const unavailableSources = allSources.filter(
|
|
3547
|
+
(source) => !observedSources.has(source) || source === "tool" && scenario.execution.toolTarget === "none"
|
|
3548
|
+
);
|
|
3549
|
+
const behavior = judgeEvalBehaviorV1(scenario, verdict, {
|
|
3550
|
+
facts,
|
|
3551
|
+
unavailableSources,
|
|
3552
|
+
infrastructureFailure: verdict.machinery === "infrastructure_failure"
|
|
3553
|
+
});
|
|
3554
|
+
assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
|
|
3555
|
+
return behavior;
|
|
3556
|
+
}
|
|
3557
|
+
function metricFactSource(key) {
|
|
3558
|
+
if (key.startsWith("tools.")) return "tool";
|
|
3559
|
+
if (key.startsWith("task.") || key.startsWith("claims.") || key.startsWith("completion.") || key === "result.pass" || key === "verifier.exitCode")
|
|
3560
|
+
return "grader";
|
|
3561
|
+
if (key.startsWith("receipt.") || key.startsWith("evidence.") || key.startsWith("boundary.") || key.startsWith("loop.") || key.startsWith("cost."))
|
|
3562
|
+
return "receipt";
|
|
3563
|
+
return "transcript";
|
|
3564
|
+
}
|
|
3565
|
+
|
|
2011
3566
|
// src/domains/eval/verifiers/command.ts
|
|
2012
3567
|
init_esm_shims();
|
|
2013
|
-
async function runCommandVerifiers(commands, cwd, timeoutMs) {
|
|
3568
|
+
async function runCommandVerifiers(commands, cwd, timeoutMs, env) {
|
|
2014
3569
|
let stdout = "";
|
|
2015
3570
|
let stderr = "";
|
|
2016
3571
|
let wallTimeMs = 0;
|
|
2017
3572
|
for (const command of commands) {
|
|
2018
|
-
const result = await runShellCommand(command, cwd, timeoutMs);
|
|
3573
|
+
const result = await runShellCommand(command, cwd, timeoutMs, env);
|
|
2019
3574
|
stdout += result.stdout;
|
|
2020
3575
|
stderr += result.stderr;
|
|
2021
3576
|
wallTimeMs += result.wallTimeMs;
|
|
@@ -2027,9 +3582,9 @@ async function runCommandVerifiers(commands, cwd, timeoutMs) {
|
|
|
2027
3582
|
// src/domains/eval/verifiers/file-exists.ts
|
|
2028
3583
|
init_esm_shims();
|
|
2029
3584
|
import { existsSync as existsSync3 } from "node:fs";
|
|
2030
|
-
import { resolve as
|
|
3585
|
+
import { resolve as resolve3 } from "node:path";
|
|
2031
3586
|
function forbiddenPathHits(cwd, paths) {
|
|
2032
|
-
return paths.filter((path) => existsSync3(
|
|
3587
|
+
return paths.filter((path) => existsSync3(resolve3(cwd, path)));
|
|
2033
3588
|
}
|
|
2034
3589
|
|
|
2035
3590
|
// src/domains/eval/verifiers/patch.ts
|
|
@@ -2051,10 +3606,10 @@ init_esm_shims();
|
|
|
2051
3606
|
import { spawn as spawn2 } from "node:child_process";
|
|
2052
3607
|
import { mkdtemp, rm } from "node:fs/promises";
|
|
2053
3608
|
import { tmpdir } from "node:os";
|
|
2054
|
-
import { resolve as
|
|
3609
|
+
import { resolve as resolve4 } from "node:path";
|
|
2055
3610
|
async function prepareGitWorkspace(workspace) {
|
|
2056
3611
|
if (workspace.url === void 0) throw new Error("git workspace requires url");
|
|
2057
|
-
const dest = await mkdtemp(
|
|
3612
|
+
const dest = await mkdtemp(resolve4(tmpdir(), "clio-eval-git-"));
|
|
2058
3613
|
try {
|
|
2059
3614
|
await runGit(["clone", "--quiet", workspace.url, dest], process.cwd());
|
|
2060
3615
|
const ref = workspace.checkout ?? workspace.commit;
|
|
@@ -2089,9 +3644,9 @@ function runGit(args, cwd) {
|
|
|
2089
3644
|
// src/domains/eval/workspaces/local.ts
|
|
2090
3645
|
init_esm_shims();
|
|
2091
3646
|
import { access } from "node:fs/promises";
|
|
2092
|
-
import { resolve as
|
|
3647
|
+
import { resolve as resolve5 } from "node:path";
|
|
2093
3648
|
async function prepareLocalWorkspace(baseDir, workspace) {
|
|
2094
|
-
const dir =
|
|
3649
|
+
const dir = resolve5(baseDir, workspace.path ?? ".");
|
|
2095
3650
|
await access(dir);
|
|
2096
3651
|
return { dir, cleanup: async () => {
|
|
2097
3652
|
} };
|
|
@@ -2099,28 +3654,113 @@ async function prepareLocalWorkspace(baseDir, workspace) {
|
|
|
2099
3654
|
|
|
2100
3655
|
// src/domains/eval/workspaces/temp-copy.ts
|
|
2101
3656
|
init_esm_shims();
|
|
2102
|
-
import {
|
|
3657
|
+
import { execFile } from "node:child_process";
|
|
3658
|
+
import { cp, lstat, mkdtemp as mkdtemp2, rm as rm2 } from "node:fs/promises";
|
|
2103
3659
|
import { tmpdir as tmpdir2 } from "node:os";
|
|
2104
|
-
import { relative, resolve as
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2110
|
-
|
|
2111
|
-
|
|
2112
|
-
|
|
2113
|
-
|
|
2114
|
-
|
|
2115
|
-
|
|
2116
|
-
|
|
2117
|
-
|
|
2118
|
-
|
|
3660
|
+
import { relative as relative2, resolve as resolve6 } from "node:path";
|
|
3661
|
+
import { promisify } from "node:util";
|
|
3662
|
+
var execFileAsync = promisify(execFile);
|
|
3663
|
+
var GIT_FILE_LIST_LIMIT_BYTES = 128 * 1024 * 1024;
|
|
3664
|
+
async function prepareTempCopyWorkspace(baseDir, workspace, options = {}) {
|
|
3665
|
+
const source = resolve6(baseDir, workspace.path ?? ".");
|
|
3666
|
+
const dest = await mkdtemp2(resolve6(options.tempRoot ?? tmpdir2(), "clio-eval-workspace-"));
|
|
3667
|
+
try {
|
|
3668
|
+
const selection = await gitCopySelection(source);
|
|
3669
|
+
const excludes = workspace.excludes ?? [];
|
|
3670
|
+
const copyWorkspace = options.copy ?? defaultCopy;
|
|
3671
|
+
await copyWorkspace(source, dest, {
|
|
3672
|
+
recursive: true,
|
|
3673
|
+
filter: (path) => shouldCopy(relative2(source, path), excludes, selection)
|
|
3674
|
+
});
|
|
3675
|
+
return {
|
|
3676
|
+
dir: dest,
|
|
3677
|
+
cleanup: async () => {
|
|
3678
|
+
await rm2(dest, { recursive: true, force: true });
|
|
3679
|
+
}
|
|
3680
|
+
};
|
|
3681
|
+
} catch (error) {
|
|
3682
|
+
await rm2(dest, { recursive: true, force: true });
|
|
3683
|
+
throw error;
|
|
3684
|
+
}
|
|
2119
3685
|
}
|
|
2120
3686
|
function isExcluded(rel, excludes) {
|
|
2121
3687
|
const normalized = rel.replaceAll("\\", "/");
|
|
2122
3688
|
return excludes.some((entry) => normalized === entry || normalized.startsWith(`${entry.replaceAll("\\", "/")}/`));
|
|
2123
3689
|
}
|
|
3690
|
+
async function defaultCopy(source, destination, options) {
|
|
3691
|
+
await cp(source, destination, options);
|
|
3692
|
+
}
|
|
3693
|
+
function shouldCopy(relativePath, excludes, selection) {
|
|
3694
|
+
const normalized = relativePath.replaceAll("\\", "/");
|
|
3695
|
+
if (normalized.length === 0) return true;
|
|
3696
|
+
if (isExcluded(normalized, excludes)) return false;
|
|
3697
|
+
if (selection === null) return true;
|
|
3698
|
+
return selection.files.has(normalized) || selection.directories.has(normalized);
|
|
3699
|
+
}
|
|
3700
|
+
async function gitCopySelection(source) {
|
|
3701
|
+
let inside;
|
|
3702
|
+
try {
|
|
3703
|
+
inside = await gitOutput(source, ["rev-parse", "--is-inside-work-tree"]);
|
|
3704
|
+
} catch (error) {
|
|
3705
|
+
if (await hasGitMarker(source)) throw error;
|
|
3706
|
+
return null;
|
|
3707
|
+
}
|
|
3708
|
+
if (inside.trim() !== "true") return null;
|
|
3709
|
+
const output = await gitOutput(source, [
|
|
3710
|
+
"--literal-pathspecs",
|
|
3711
|
+
"ls-files",
|
|
3712
|
+
"-z",
|
|
3713
|
+
"--cached",
|
|
3714
|
+
"--others",
|
|
3715
|
+
"--exclude-standard",
|
|
3716
|
+
"--",
|
|
3717
|
+
"."
|
|
3718
|
+
]);
|
|
3719
|
+
const files = /* @__PURE__ */ new Set();
|
|
3720
|
+
const directories = /* @__PURE__ */ new Set();
|
|
3721
|
+
for (const path of output.split("\0")) {
|
|
3722
|
+
if (path.length === 0) continue;
|
|
3723
|
+
const normalized = normalizeGitPath(path);
|
|
3724
|
+
if (normalized === null) throw new Error("git ls-files returned a path outside the eval workspace");
|
|
3725
|
+
files.add(normalized);
|
|
3726
|
+
let separator = normalized.lastIndexOf("/");
|
|
3727
|
+
while (separator >= 0) {
|
|
3728
|
+
directories.add(normalized.slice(0, separator));
|
|
3729
|
+
separator = normalized.lastIndexOf("/", separator - 1);
|
|
3730
|
+
}
|
|
3731
|
+
}
|
|
3732
|
+
return { files, directories };
|
|
3733
|
+
}
|
|
3734
|
+
async function gitOutput(cwd, args) {
|
|
3735
|
+
const { stdout } = await execFileAsync("git", [...args], {
|
|
3736
|
+
cwd,
|
|
3737
|
+
encoding: "utf8",
|
|
3738
|
+
maxBuffer: GIT_FILE_LIST_LIMIT_BYTES
|
|
3739
|
+
});
|
|
3740
|
+
return stdout;
|
|
3741
|
+
}
|
|
3742
|
+
function normalizeGitPath(path) {
|
|
3743
|
+
const normalized = path.replaceAll("\\", "/").replace(/^\.\//u, "");
|
|
3744
|
+
if (normalized.length === 0 || normalized.startsWith("/") || /^[A-Za-z]:\//u.test(normalized)) return null;
|
|
3745
|
+
const segments = normalized.split("/");
|
|
3746
|
+
if (segments.some((segment) => segment.length === 0 || segment === "." || segment === "..")) return null;
|
|
3747
|
+
return normalized;
|
|
3748
|
+
}
|
|
3749
|
+
async function hasGitMarker(source) {
|
|
3750
|
+
let current = resolve6(source);
|
|
3751
|
+
while (true) {
|
|
3752
|
+
try {
|
|
3753
|
+
await lstat(resolve6(current, ".git"));
|
|
3754
|
+
return true;
|
|
3755
|
+
} catch (error) {
|
|
3756
|
+
const code = typeof error === "object" && error !== null && "code" in error ? error.code : void 0;
|
|
3757
|
+
if (code !== "ENOENT" && code !== "ENOTDIR") throw error;
|
|
3758
|
+
}
|
|
3759
|
+
const parent = resolve6(current, "..");
|
|
3760
|
+
if (parent === current) return false;
|
|
3761
|
+
current = parent;
|
|
3762
|
+
}
|
|
3763
|
+
}
|
|
2124
3764
|
|
|
2125
3765
|
// src/domains/eval/suites/matrix.ts
|
|
2126
3766
|
init_esm_shims();
|
|
@@ -2148,28 +3788,39 @@ async function runEvalSuiteV2(loaded, options) {
|
|
|
2148
3788
|
const started = now();
|
|
2149
3789
|
const evalId = createEvalId(started, loaded.hash);
|
|
2150
3790
|
const results = [];
|
|
3791
|
+
const servingObservations = [];
|
|
2151
3792
|
const maxCostUsd = loaded.suite.matrix.maxCostUsd;
|
|
2152
3793
|
let spentUsd = 0;
|
|
2153
3794
|
for (const item of expandEvalMatrix(loaded.suite)) {
|
|
2154
3795
|
if (maxCostUsd !== void 0 && spentUsd > maxCostUsd) {
|
|
2155
|
-
results.push(budgetExhaustedResult(item.task
|
|
3796
|
+
results.push(budgetExhaustedResult(loaded, item.task, item.target, item.repeatIndex, spentUsd, maxCostUsd));
|
|
2156
3797
|
continue;
|
|
2157
3798
|
}
|
|
2158
|
-
const
|
|
2159
|
-
|
|
2160
|
-
|
|
3799
|
+
const completed = await runMatrixItem(
|
|
3800
|
+
loaded,
|
|
3801
|
+
item.task,
|
|
3802
|
+
item.target,
|
|
3803
|
+
item.repeatIndex,
|
|
3804
|
+
options.clioEntry,
|
|
3805
|
+
options.freshWorkspaces === true,
|
|
3806
|
+
options.tempCopy
|
|
3807
|
+
);
|
|
3808
|
+
spentUsd += resultCostUsd(completed.result);
|
|
3809
|
+
results.push(completed.result);
|
|
3810
|
+
servingObservations.push(completed.serving);
|
|
2161
3811
|
}
|
|
2162
|
-
|
|
3812
|
+
const serving = await evalServingConfiguration(loaded.suite.matrix.targets, servingObservations);
|
|
3813
|
+
return buildArtifact(loaded, evalId, results, options.clioEntry, serving);
|
|
2163
3814
|
}
|
|
2164
3815
|
function resultCostUsd(result) {
|
|
2165
3816
|
const value = result.metrics["cost.usd"];
|
|
2166
3817
|
return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : 0;
|
|
2167
3818
|
}
|
|
2168
|
-
function budgetExhaustedResult(
|
|
2169
|
-
|
|
3819
|
+
function budgetExhaustedResult(loaded, task, target, repeatIndex, spentUsd, maxCostUsd) {
|
|
3820
|
+
const result = {
|
|
2170
3821
|
assignmentId: null,
|
|
2171
3822
|
terminalReceiptDigest: null,
|
|
2172
|
-
taskId,
|
|
3823
|
+
taskId: task.id,
|
|
2173
3824
|
repeatIndex,
|
|
2174
3825
|
target: { id: target.id, model: target.model ?? null, thinking: target.thinking ?? null },
|
|
2175
3826
|
pass: false,
|
|
@@ -2184,21 +3835,36 @@ function budgetExhaustedResult(taskId, target, repeatIndex, spentUsd, maxCostUsd
|
|
|
2184
3835
|
error: `matrix cost budget exhausted: spent $${spentUsd.toFixed(4)} of max $${maxCostUsd.toFixed(4)} before this item`
|
|
2185
3836
|
}
|
|
2186
3837
|
};
|
|
3838
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(result, emptyEvalTrackedMetrics());
|
|
3839
|
+
attachBehavioralResult(result, task);
|
|
3840
|
+
attachExecutionEnvelope(result, task, target, loaded.baseDir, null, emptyLedgerSnapshot());
|
|
3841
|
+
return result;
|
|
2187
3842
|
}
|
|
2188
|
-
async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
3843
|
+
async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry, freshWorkspace, tempCopy) {
|
|
2189
3844
|
let workspace = null;
|
|
2190
|
-
|
|
3845
|
+
let receipt = null;
|
|
3846
|
+
let runnerWallTimeMs = 0;
|
|
3847
|
+
let executionObservation;
|
|
3848
|
+
const stateDir = await mkdtemp3(resolve7(tempCopy?.tempRoot ?? tmpdir3(), "clio-eval-state-"));
|
|
2191
3849
|
try {
|
|
2192
|
-
workspace = await prepareWorkspace(loaded.baseDir, task);
|
|
3850
|
+
workspace = await prepareWorkspace(loaded.baseDir, task, freshWorkspace, tempCopy);
|
|
2193
3851
|
const setup = await runCommandVerifiers(task.workspace.setup ?? [], workspace.dir, task.timeoutMs);
|
|
2194
3852
|
if (!setup.pass) throw new EvalWorkspaceSetupError(setup.exitCode, setup.stderr);
|
|
2195
3853
|
const runner = await runTaskRunner(task, target, workspace.dir, clioEntry, {
|
|
2196
3854
|
CLIO_CODER_STATE_DIR: stateDir,
|
|
2197
3855
|
CLIO_CODER_ENTRY: clioEntry
|
|
2198
3856
|
});
|
|
3857
|
+
const runnerStdoutFile = resolve7(stateDir, "eval-runner-output.jsonl");
|
|
3858
|
+
await writeFile(runnerStdoutFile, runner.stdout, "utf8");
|
|
3859
|
+
receipt = runner.receipt ?? null;
|
|
3860
|
+
runnerWallTimeMs = runner.wallTimeMs;
|
|
2199
3861
|
const patch = collectPatchMetrics(workspace.dir);
|
|
2200
3862
|
const receiptExitCode = runner.exitCode;
|
|
2201
3863
|
const journalMetrics = invariantMetrics(stateDir, receiptExitCode);
|
|
3864
|
+
const measurement = await measureTaskOutcome(task, workspace.dir, {
|
|
3865
|
+
CLIO_EVAL_RUNNER_STDOUT_FILE: runnerStdoutFile
|
|
3866
|
+
});
|
|
3867
|
+
executionObservation = measurement.executionObservation;
|
|
2202
3868
|
const metrics = {
|
|
2203
3869
|
...zeroToolCallMetrics(),
|
|
2204
3870
|
...collectContextMetrics(workspace.dir),
|
|
@@ -2214,15 +3880,16 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2214
3880
|
"patch.testFilesModified": patch.testFilesModified,
|
|
2215
3881
|
"result.pass": runner.exitCode === 0,
|
|
2216
3882
|
"result.failureClass": runner.exitCode === 0 ? null : "runner_failed",
|
|
2217
|
-
...
|
|
3883
|
+
...measurement.metrics
|
|
2218
3884
|
};
|
|
2219
3885
|
const verifier = await runVerifiers(task, workspace.dir, metrics);
|
|
2220
|
-
const
|
|
2221
|
-
const
|
|
3886
|
+
const graderFailed = metrics["task.solved"] === false;
|
|
3887
|
+
const pass = runner.exitCode === 0 && verifier.pass && !graderFailed;
|
|
3888
|
+
const failureClass = pass ? null : runner.exitCode !== 0 ? "runner_failed" : !verifier.pass ? verifier.failureClass : "grader_failed";
|
|
2222
3889
|
metrics["verifier.exitCode"] = verifier.exitCode;
|
|
2223
3890
|
metrics["result.pass"] = pass;
|
|
2224
3891
|
metrics["result.failureClass"] = failureClass;
|
|
2225
|
-
|
|
3892
|
+
const result = {
|
|
2226
3893
|
assignmentId: runner.assignmentId,
|
|
2227
3894
|
terminalReceiptDigest: runner.terminalReceiptDigest,
|
|
2228
3895
|
taskId: task.id,
|
|
@@ -2233,13 +3900,30 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2233
3900
|
metrics,
|
|
2234
3901
|
artifacts: {
|
|
2235
3902
|
...runner.artifacts,
|
|
3903
|
+
workspace: workspace.dir,
|
|
2236
3904
|
...verifier.stdout.length > 0 ? { verifierStdout: verifier.stdout } : {},
|
|
2237
3905
|
...verifier.stderr.length > 0 ? { verifierStderr: verifier.stderr } : {}
|
|
2238
3906
|
}
|
|
2239
3907
|
};
|
|
3908
|
+
const snapshot = await readEvalLedgerSnapshot(stateDir);
|
|
3909
|
+
const ledgerEntries = [...snapshot.entries, ...runner.ledgerEntries ?? []];
|
|
3910
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(
|
|
3911
|
+
result,
|
|
3912
|
+
buildEvalTrackedMetrics({
|
|
3913
|
+
ledgerEntries,
|
|
3914
|
+
receipt: receipt ?? null,
|
|
3915
|
+
fallbackWallClockMs: runner.wallTimeMs
|
|
3916
|
+
})
|
|
3917
|
+
);
|
|
3918
|
+
attachBehavioralResult(result, task);
|
|
3919
|
+
attachExecutionEnvelope(result, task, target, workspace.dir, receipt ?? null, snapshot, executionObservation);
|
|
3920
|
+
return {
|
|
3921
|
+
result,
|
|
3922
|
+
serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
|
|
3923
|
+
};
|
|
2240
3924
|
} catch (error) {
|
|
2241
3925
|
const failureClass = error instanceof EvalWorkspaceSetupError ? "setup_failed" : "command_error";
|
|
2242
|
-
|
|
3926
|
+
const result = {
|
|
2243
3927
|
assignmentId: null,
|
|
2244
3928
|
terminalReceiptDigest: null,
|
|
2245
3929
|
taskId: task.id,
|
|
@@ -2253,13 +3937,61 @@ async function runMatrixItem(loaded, task, target, repeatIndex, clioEntry) {
|
|
|
2253
3937
|
"verifier.exitCode": 1,
|
|
2254
3938
|
"latency.wallMs": 0
|
|
2255
3939
|
},
|
|
2256
|
-
artifacts: {
|
|
3940
|
+
artifacts: {
|
|
3941
|
+
error: error instanceof Error ? error.message : String(error),
|
|
3942
|
+
...workspace === null ? {} : { workspace: workspace.dir }
|
|
3943
|
+
}
|
|
3944
|
+
};
|
|
3945
|
+
const snapshot = await readEvalLedgerSnapshot(stateDir);
|
|
3946
|
+
result.verdict = adaptSuiteV2ResultToVerdictV1(
|
|
3947
|
+
result,
|
|
3948
|
+
buildEvalTrackedMetrics({
|
|
3949
|
+
ledgerEntries: snapshot.entries,
|
|
3950
|
+
receipt: receipt ?? null,
|
|
3951
|
+
fallbackWallClockMs: runnerWallTimeMs
|
|
3952
|
+
})
|
|
3953
|
+
);
|
|
3954
|
+
attachBehavioralResult(result, task);
|
|
3955
|
+
attachExecutionEnvelope(
|
|
3956
|
+
result,
|
|
3957
|
+
task,
|
|
3958
|
+
target,
|
|
3959
|
+
workspace?.dir ?? loaded.baseDir,
|
|
3960
|
+
receipt ?? null,
|
|
3961
|
+
snapshot,
|
|
3962
|
+
executionObservation
|
|
3963
|
+
);
|
|
3964
|
+
return {
|
|
3965
|
+
result,
|
|
3966
|
+
serving: evalServingObservationFrom(target, receipt ?? null, snapshot.compiledPromptHashes)
|
|
2257
3967
|
};
|
|
2258
3968
|
} finally {
|
|
2259
|
-
|
|
2260
|
-
|
|
3969
|
+
try {
|
|
3970
|
+
await workspace?.cleanup();
|
|
3971
|
+
} finally {
|
|
3972
|
+
await rm3(stateDir, { recursive: true, force: true });
|
|
3973
|
+
}
|
|
2261
3974
|
}
|
|
2262
3975
|
}
|
|
3976
|
+
function attachBehavioralResult(result, task) {
|
|
3977
|
+
if (task.behavioral === void 0 || result.verdict === void 0) return;
|
|
3978
|
+
result.behavioral = adaptSuiteV2ResultToBehaviorV1(result, result.verdict, task.behavioral);
|
|
3979
|
+
result.behavioralMetrics = buildEvalBehaviorMetricsV1(result, task.behavioral.execution.subject.role);
|
|
3980
|
+
}
|
|
3981
|
+
function attachExecutionEnvelope(result, task, target, cwd, receipt, ledger, observation) {
|
|
3982
|
+
if (task.behavioral === void 0) return;
|
|
3983
|
+
result.executionEnvelope = buildEvalExecutionEnvelopeV1({
|
|
3984
|
+
task,
|
|
3985
|
+
target,
|
|
3986
|
+
cwd,
|
|
3987
|
+
receipt,
|
|
3988
|
+
ledger,
|
|
3989
|
+
...observation === void 0 ? {} : { observation }
|
|
3990
|
+
});
|
|
3991
|
+
}
|
|
3992
|
+
function emptyLedgerSnapshot() {
|
|
3993
|
+
return { entries: [], compiledPromptHashes: [], promptManifests: [], contextSnapshots: [] };
|
|
3994
|
+
}
|
|
2263
3995
|
function invariantMetrics(stateDir, runnerExitCode) {
|
|
2264
3996
|
const journal = readRunJournal(stateDir);
|
|
2265
3997
|
return {
|
|
@@ -2270,23 +4002,80 @@ function invariantMetrics(stateDir, runnerExitCode) {
|
|
|
2270
4002
|
...writeBoundaryInvariantMetrics(stateDir)
|
|
2271
4003
|
};
|
|
2272
4004
|
}
|
|
2273
|
-
async function prepareWorkspace(baseDir, task) {
|
|
4005
|
+
async function prepareWorkspace(baseDir, task, freshWorkspace, tempCopy) {
|
|
4006
|
+
if (task.workspace.kind === "local" && freshWorkspace) {
|
|
4007
|
+
return prepareTempCopyWorkspace(baseDir, { ...task.workspace, kind: "temp-copy" }, tempCopy);
|
|
4008
|
+
}
|
|
2274
4009
|
if (task.workspace.kind === "local") return prepareLocalWorkspace(baseDir, task.workspace);
|
|
2275
4010
|
if (task.workspace.kind === "git") return prepareGitWorkspace(task.workspace);
|
|
2276
|
-
return prepareTempCopyWorkspace(baseDir, task.workspace);
|
|
4011
|
+
return prepareTempCopyWorkspace(baseDir, task.workspace, tempCopy);
|
|
2277
4012
|
}
|
|
2278
4013
|
async function runTaskRunner(task, target, cwd, clioEntry, env) {
|
|
2279
4014
|
if (task.runner.kind === "external-command") return runExternalCommandRunner(task.runner, cwd, task.timeoutMs, env);
|
|
2280
4015
|
if (task.runner.kind === "context-index") return runContextIndexRunner(cwd, clioEntry, task.timeoutMs, target, env);
|
|
2281
4016
|
if (task.runner.kind === "context-init")
|
|
2282
4017
|
return runContextInitRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
|
|
2283
|
-
return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env);
|
|
4018
|
+
return runClioRunRunner(task.runner, cwd, clioEntry, task.timeoutMs, target, env, task.metrics.readObservation);
|
|
2284
4019
|
}
|
|
2285
|
-
async function measureTaskOutcome(task, cwd) {
|
|
4020
|
+
async function measureTaskOutcome(task, cwd, env) {
|
|
2286
4021
|
const commands = task.verify.measure ?? [];
|
|
2287
|
-
if (commands.length === 0) return {};
|
|
2288
|
-
const result = await runCommandVerifiers(commands, cwd, task.timeoutMs);
|
|
2289
|
-
|
|
4022
|
+
if (commands.length === 0) return { metrics: {} };
|
|
4023
|
+
const result = await runCommandVerifiers(commands, cwd, task.timeoutMs, env);
|
|
4024
|
+
const behavioral = graderBehaviorMeasurement(result.stdout);
|
|
4025
|
+
return {
|
|
4026
|
+
metrics: {
|
|
4027
|
+
"task.exitCode": result.exitCode,
|
|
4028
|
+
"task.solved": result.exitCode === 0,
|
|
4029
|
+
...behavioral.metrics
|
|
4030
|
+
},
|
|
4031
|
+
...behavioral.executionObservation === void 0 ? {} : { executionObservation: behavioral.executionObservation }
|
|
4032
|
+
};
|
|
4033
|
+
}
|
|
4034
|
+
function graderBehaviorMeasurement(stdout) {
|
|
4035
|
+
const metrics = {};
|
|
4036
|
+
let executionObservation;
|
|
4037
|
+
for (const line of stdout.split(/\r?\n/u)) {
|
|
4038
|
+
if (line.trim().length === 0) continue;
|
|
4039
|
+
let value;
|
|
4040
|
+
try {
|
|
4041
|
+
value = JSON.parse(line);
|
|
4042
|
+
} catch {
|
|
4043
|
+
continue;
|
|
4044
|
+
}
|
|
4045
|
+
if (!isRecord10(value)) continue;
|
|
4046
|
+
if (value.schema === "clio.eval.measure.v1" && isRecord10(value.metrics)) {
|
|
4047
|
+
for (const [key, metric] of Object.entries(value.metrics)) {
|
|
4048
|
+
if (key !== "claims.unsupported" && key !== "completion.reported") continue;
|
|
4049
|
+
if (typeof metric === "boolean" || typeof metric === "number" && Number.isFinite(metric)) metrics[key] = metric;
|
|
4050
|
+
}
|
|
4051
|
+
}
|
|
4052
|
+
if (value.schema === "clio.eval.execution-observation.v1") {
|
|
4053
|
+
executionObservation = parseExecutionObservation(value);
|
|
4054
|
+
}
|
|
4055
|
+
}
|
|
4056
|
+
return { metrics, ...executionObservation === void 0 ? {} : { executionObservation } };
|
|
4057
|
+
}
|
|
4058
|
+
function parseExecutionObservation(value) {
|
|
4059
|
+
const policies = isRecord10(value.policyHashes) ? value.policyHashes : {};
|
|
4060
|
+
const project = isRecord10(value.projectContext) ? value.projectContext : null;
|
|
4061
|
+
return {
|
|
4062
|
+
compositionHash: nullableDigest2(value.compositionHash),
|
|
4063
|
+
target: nullableString2(value.target),
|
|
4064
|
+
wireModel: nullableString2(value.wireModel),
|
|
4065
|
+
runtime: nullableString2(value.runtime),
|
|
4066
|
+
thinkingLevel: nullableString2(value.thinkingLevel),
|
|
4067
|
+
toolSignature: nullableDigest2(value.toolSignature),
|
|
4068
|
+
autonomy: nullableString2(value.autonomy),
|
|
4069
|
+
policyHashes: { rulePack: nullableDigest2(policies.rulePack), project: nullableDigest2(policies.project) },
|
|
4070
|
+
projectContext: project === null ? null : {
|
|
4071
|
+
tier: nullableString2(project.tier),
|
|
4072
|
+
contentHash: nullableDigest2(project.contentHash),
|
|
4073
|
+
chars: nullableNonNegativeInteger(project.chars),
|
|
4074
|
+
sections: stringArray(project.sections),
|
|
4075
|
+
rulesApplied: stringArray(project.rulesApplied),
|
|
4076
|
+
operatorProfileApplied: typeof project.operatorProfileApplied === "boolean" ? project.operatorProfileApplied : null
|
|
4077
|
+
}
|
|
4078
|
+
};
|
|
2290
4079
|
}
|
|
2291
4080
|
async function runVerifiers(task, cwd, metrics) {
|
|
2292
4081
|
const commandResult = await runCommandVerifiers(task.verify.commands ?? [], cwd, task.timeoutMs);
|
|
@@ -2322,7 +4111,7 @@ async function runVerifiers(task, cwd, metrics) {
|
|
|
2322
4111
|
}
|
|
2323
4112
|
return { pass: true, exitCode: 0, failureClass: null, stdout: commandResult.stdout, stderr: commandResult.stderr };
|
|
2324
4113
|
}
|
|
2325
|
-
function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
4114
|
+
function buildArtifact(loaded, evalId, results, clioEntry, servingConfiguration) {
|
|
2326
4115
|
const passed = results.filter((result) => result.pass).length;
|
|
2327
4116
|
return {
|
|
2328
4117
|
version: 4,
|
|
@@ -2330,7 +4119,11 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
|
2330
4119
|
suite: { id: loaded.suite.suite.id, hash: loaded.hash },
|
|
2331
4120
|
clio: evalClioProvenance({ entry: clioEntry }),
|
|
2332
4121
|
environment: evalEnvironmentProvenance(),
|
|
2333
|
-
matrix:
|
|
4122
|
+
matrix: {
|
|
4123
|
+
...artifactMatrixIdentity(loaded.suite.matrix.targets),
|
|
4124
|
+
...loaded.suite.matrix.dimensions === void 0 ? {} : { dimensions: loaded.suite.matrix.dimensions }
|
|
4125
|
+
},
|
|
4126
|
+
servingConfiguration,
|
|
2334
4127
|
summary: {
|
|
2335
4128
|
runs: results.length,
|
|
2336
4129
|
passed,
|
|
@@ -2339,26 +4132,50 @@ function buildArtifact(loaded, evalId, results, clioEntry) {
|
|
|
2339
4132
|
tokens: tokenAccountingFrom(results),
|
|
2340
4133
|
wallTimeMs: results.reduce((sum2, result) => sum2 + wallTimeMetric(result.metrics), 0)
|
|
2341
4134
|
},
|
|
4135
|
+
aggregates: aggregateEvalVerdicts(
|
|
4136
|
+
results.flatMap((result) => result.verdict === void 0 ? [] : [result.verdict])
|
|
4137
|
+
),
|
|
2342
4138
|
results
|
|
2343
4139
|
};
|
|
2344
4140
|
}
|
|
2345
4141
|
function assertionMessage(assertion, actual) {
|
|
2346
4142
|
return `assertion failed: ${assertion.metric} ${assertion.op} ${String(assertion.value)} (actual ${JSON.stringify(actual)})`;
|
|
2347
4143
|
}
|
|
4144
|
+
function isRecord10(value) {
|
|
4145
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
4146
|
+
}
|
|
4147
|
+
function nullableString2(value) {
|
|
4148
|
+
return typeof value === "string" && value.length > 0 ? value : null;
|
|
4149
|
+
}
|
|
4150
|
+
function nullableDigest2(value) {
|
|
4151
|
+
return typeof value === "string" && /^[a-f0-9]{64}$/u.test(value) ? value : null;
|
|
4152
|
+
}
|
|
4153
|
+
function nullableNonNegativeInteger(value) {
|
|
4154
|
+
return typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : null;
|
|
4155
|
+
}
|
|
4156
|
+
function stringArray(value) {
|
|
4157
|
+
return Array.isArray(value) ? value.filter((entry) => typeof entry === "string") : [];
|
|
4158
|
+
}
|
|
2348
4159
|
|
|
2349
4160
|
// src/cli/eval.ts
|
|
2350
4161
|
var HELP = `clio-coder eval <command>
|
|
2351
4162
|
|
|
2352
4163
|
Commands:
|
|
2353
4164
|
clio-coder eval validate --suite <suite.yaml>
|
|
2354
|
-
|
|
2355
|
-
|
|
4165
|
+
clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
|
|
4166
|
+
clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
|
|
2356
4167
|
clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
|
|
2357
|
-
|
|
4168
|
+
clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
|
|
2358
4169
|
clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
|
|
2359
4170
|
`;
|
|
2360
4171
|
function parseEvalArgs(args) {
|
|
2361
|
-
const parsed = {
|
|
4172
|
+
const parsed = {
|
|
4173
|
+
repeat: 1,
|
|
4174
|
+
compareIds: [],
|
|
4175
|
+
format: "text",
|
|
4176
|
+
allowConfigDrift: false,
|
|
4177
|
+
help: false
|
|
4178
|
+
};
|
|
2362
4179
|
for (let index = 0; index < args.length; index += 1) {
|
|
2363
4180
|
const arg = args[index];
|
|
2364
4181
|
if (arg === void 0) continue;
|
|
@@ -2417,6 +4234,11 @@ function parseEvalArgs(args) {
|
|
|
2417
4234
|
index += 1;
|
|
2418
4235
|
continue;
|
|
2419
4236
|
}
|
|
4237
|
+
if (arg === "--trials") {
|
|
4238
|
+
parsed.trials = positiveInteger(requiredValue(args, index, "--trials"), "--trials");
|
|
4239
|
+
index += 1;
|
|
4240
|
+
continue;
|
|
4241
|
+
}
|
|
2420
4242
|
throw new Error(`unknown eval run argument: ${arg}`);
|
|
2421
4243
|
}
|
|
2422
4244
|
if (parsed.command === "report") {
|
|
@@ -2432,6 +4254,20 @@ function parseEvalArgs(args) {
|
|
|
2432
4254
|
throw new Error(`unexpected eval report argument: ${arg}`);
|
|
2433
4255
|
}
|
|
2434
4256
|
if (parsed.command === "compare") {
|
|
4257
|
+
if (arg === "--format") {
|
|
4258
|
+
parsed.format = comparisonFormat(requiredValue(args, index, "--format"));
|
|
4259
|
+
index += 1;
|
|
4260
|
+
continue;
|
|
4261
|
+
}
|
|
4262
|
+
if (arg === "--metric") {
|
|
4263
|
+
parsed.metric = requiredValue(args, index, "--metric");
|
|
4264
|
+
index += 1;
|
|
4265
|
+
continue;
|
|
4266
|
+
}
|
|
4267
|
+
if (arg === "--allow-config-drift") {
|
|
4268
|
+
parsed.allowConfigDrift = true;
|
|
4269
|
+
continue;
|
|
4270
|
+
}
|
|
2435
4271
|
if (!arg.startsWith("-")) {
|
|
2436
4272
|
parsed.compareIds.push(arg);
|
|
2437
4273
|
continue;
|
|
@@ -2504,13 +4340,19 @@ async function runEvalValidate(parsed) {
|
|
|
2504
4340
|
}
|
|
2505
4341
|
async function runEvalRun(parsed) {
|
|
2506
4342
|
try {
|
|
2507
|
-
const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.repeat);
|
|
4343
|
+
const loaded = parsed.suite !== void 0 ? await loadEvalSuiteFile(parsed.suite) : await loadV1TaskFileAsSuite(parsed.taskFile ?? "", parsed.trials ?? parsed.repeat);
|
|
2508
4344
|
const resolveOptions = {};
|
|
2509
4345
|
if (parsed.target !== void 0) resolveOptions.target = parsed.target;
|
|
2510
4346
|
if (parsed.model !== void 0) resolveOptions.model = parsed.model;
|
|
2511
|
-
const suite = resolveSuiteForRun(loaded.suite,
|
|
2512
|
-
|
|
2513
|
-
|
|
4347
|
+
const suite = resolveSuiteForRun(loaded.suite, {
|
|
4348
|
+
...resolveOptions,
|
|
4349
|
+
...parsed.trials ? { trials: parsed.trials } : {}
|
|
4350
|
+
});
|
|
4351
|
+
const clioEntry = resolve8(parsed.clioEntry ?? process.argv[1] ?? "dist/cli/index.js");
|
|
4352
|
+
const artifact = await runEvalSuiteV2(
|
|
4353
|
+
{ ...loaded, suite },
|
|
4354
|
+
{ clioEntry, freshWorkspaces: parsed.trials !== void 0 }
|
|
4355
|
+
);
|
|
2514
4356
|
const artifactPath = await writeEvalArtifactV4(clioDataDir(), artifact, parsed.out);
|
|
2515
4357
|
process.stdout.write(`${renderEvalTextReportV4(artifact)}artifact: ${artifactPath}
|
|
2516
4358
|
`);
|
|
@@ -2520,6 +4362,11 @@ async function runEvalRun(parsed) {
|
|
|
2520
4362
|
`);
|
|
2521
4363
|
for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
|
|
2522
4364
|
}
|
|
4365
|
+
if (gate !== null && gate.informational.length > 0) {
|
|
4366
|
+
process.stdout.write(`informational budgets: ${gate.informational.length} notice
|
|
4367
|
+
`);
|
|
4368
|
+
for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
|
|
4369
|
+
}
|
|
2523
4370
|
return artifact.summary.failed === 0 && (gate === null || gate.pass) ? 0 : 1;
|
|
2524
4371
|
} catch (error) {
|
|
2525
4372
|
return handleEvalLoadError(error, 1);
|
|
@@ -2543,8 +4390,12 @@ async function runEvalCompareCommand(parsed) {
|
|
|
2543
4390
|
const dataDir = clioDataDir();
|
|
2544
4391
|
const baseline = await loadEvalArtifactV4(dataDir, baselineEvalId);
|
|
2545
4392
|
const candidate = await loadEvalArtifactV4(dataDir, candidateEvalId);
|
|
2546
|
-
|
|
2547
|
-
|
|
4393
|
+
const summary = compareEvalArtifactsV4(baseline, candidate, {
|
|
4394
|
+
allowConfigDrift: parsed.allowConfigDrift,
|
|
4395
|
+
...parsed.metric === void 0 ? {} : { metric: parsed.metric }
|
|
4396
|
+
});
|
|
4397
|
+
process.stdout.write(renderEvalComparisonReportV1(summary, parsed.format));
|
|
4398
|
+
return summary.hardGate.pass ? 0 : 1;
|
|
2548
4399
|
} catch (error) {
|
|
2549
4400
|
printError(error instanceof Error ? error.message : String(error));
|
|
2550
4401
|
return error instanceof InvalidIdError ? 2 : 1;
|
|
@@ -2554,27 +4405,46 @@ async function runEvalGateCommand(parsed) {
|
|
|
2554
4405
|
try {
|
|
2555
4406
|
const dataDir = clioDataDir();
|
|
2556
4407
|
const candidate = await loadEvalArtifactV4(dataDir, parsed.evalId ?? "");
|
|
2557
|
-
await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
|
|
2558
|
-
const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }] } : loadThresholds(parsed.thresholds);
|
|
4408
|
+
const baseline = await loadEvalArtifactV4(dataDir, parsed.baseline ?? "");
|
|
4409
|
+
const thresholds = parsed.thresholds === void 0 ? { fail: [{ metric: "result.pass", op: "eq", value: false }], informational: [] } : loadThresholds(parsed.thresholds);
|
|
2559
4410
|
const gate = evaluateGate(candidate, thresholds);
|
|
2560
|
-
|
|
4411
|
+
const comparison = compareEvalArtifactsV4(baseline, candidate);
|
|
4412
|
+
if (gate.informational.length > 0) {
|
|
4413
|
+
process.stdout.write(`informational budgets: ${gate.informational.length} notice
|
|
4414
|
+
`);
|
|
4415
|
+
for (const finding of gate.informational) process.stdout.write(renderInformationalBudget(finding));
|
|
4416
|
+
}
|
|
4417
|
+
if (gate.pass && comparison.hardGate.pass) {
|
|
2561
4418
|
process.stdout.write("gate: pass\n");
|
|
2562
4419
|
return 0;
|
|
2563
4420
|
}
|
|
2564
|
-
|
|
4421
|
+
const failureCount = gate.failures.length + comparison.hardGate.failures.length + comparison.hardGate.envelopeFailures.length;
|
|
4422
|
+
process.stdout.write(`gate: fail (${failureCount} hard failure)
|
|
2565
4423
|
`);
|
|
2566
4424
|
for (const failure of gate.failures) process.stdout.write(renderGateFailure(failure));
|
|
4425
|
+
for (const failure of comparison.hardGate.failures) {
|
|
4426
|
+
process.stdout.write(
|
|
4427
|
+
` ${failure.metric} [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: ${failure.change} (hard behavioral gate)
|
|
4428
|
+
`
|
|
4429
|
+
);
|
|
4430
|
+
}
|
|
4431
|
+
for (const failure of comparison.hardGate.envelopeFailures) {
|
|
4432
|
+
process.stdout.write(
|
|
4433
|
+
` execution envelope [${failure.scenarioId}:${failure.role}:${failure.target.id}/${failure.target.model ?? "none"}]: incomparable fields ${failure.fields.join(", ")}
|
|
4434
|
+
`
|
|
4435
|
+
);
|
|
4436
|
+
}
|
|
2567
4437
|
return 1;
|
|
2568
4438
|
} catch (error) {
|
|
2569
4439
|
printError(error instanceof Error ? error.message : String(error));
|
|
2570
4440
|
return error instanceof InvalidIdError ? 2 : 1;
|
|
2571
4441
|
}
|
|
2572
4442
|
}
|
|
2573
|
-
function renderArtifactReport(artifact,
|
|
2574
|
-
if (
|
|
2575
|
-
if (
|
|
2576
|
-
if (
|
|
2577
|
-
if (
|
|
4443
|
+
function renderArtifactReport(artifact, format2, _dataDir) {
|
|
4444
|
+
if (format2 === "json") return renderEvalJsonReportV4(artifact);
|
|
4445
|
+
if (format2 === "md") return renderEvalMarkdownReportV4(artifact);
|
|
4446
|
+
if (format2 === "swe-jsonl") return renderEvalSweJsonlReportV4(artifact);
|
|
4447
|
+
if (format2 === "junit") return renderEvalJunitReportV4(artifact);
|
|
2578
4448
|
return renderEvalTextReportV4(artifact);
|
|
2579
4449
|
}
|
|
2580
4450
|
function handleEvalLoadError(error, fallback = 2) {
|
|
@@ -2603,7 +4473,11 @@ function reportFormat(value) {
|
|
|
2603
4473
|
if (value === "text" || value === "json" || value === "md" || value === "swe-jsonl" || value === "junit") return value;
|
|
2604
4474
|
throw new Error("--format must be text, json, md, swe-jsonl, or junit");
|
|
2605
4475
|
}
|
|
4476
|
+
function comparisonFormat(value) {
|
|
4477
|
+
if (value === "text" || value === "json" || value === "md" || value === "junit") return value;
|
|
4478
|
+
throw new Error("eval compare --format must be text, json, md, or junit");
|
|
4479
|
+
}
|
|
2606
4480
|
export {
|
|
2607
4481
|
runEvalCommand
|
|
2608
4482
|
};
|
|
2609
|
-
//# sourceMappingURL=eval-
|
|
4483
|
+
//# sourceMappingURL=eval-IJ5VEZDJ.js.map
|