@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
import type { EvalCompareV4Summary } from "../compare/compare.js";
|
|
2
|
+
import { renderEvalComparisonV4 } from "../compare/compare.js";
|
|
3
|
+
|
|
4
|
+
export type EvalComparisonReportFormat = "text" | "json" | "md" | "junit";
|
|
5
|
+
|
|
6
|
+
export function renderEvalComparisonReportV1(
|
|
7
|
+
summary: EvalCompareV4Summary,
|
|
8
|
+
format: EvalComparisonReportFormat,
|
|
9
|
+
): string {
|
|
10
|
+
if (format === "json") return `${JSON.stringify(summary, null, 2)}\n`;
|
|
11
|
+
if (format === "md") return renderMarkdown(summary);
|
|
12
|
+
if (format === "junit") return renderJunit(summary);
|
|
13
|
+
return renderEvalComparisonV4(summary);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function renderMarkdown(summary: EvalCompareV4Summary): string {
|
|
17
|
+
const rows = summary.behavioralMetrics.map(
|
|
18
|
+
(row) =>
|
|
19
|
+
`| ${cell(row.scenarioId)} | ${cell(row.role)} | ${cell(`${row.target.id}/${row.target.model ?? "none"}`)} | ${row.family} | ${row.metric} | ${format(row.baseline.mean)} | ${format(row.baseline.variance)} | ${row.baseline.measured}/${row.baseline.observations} | ${format(row.candidate.mean)} | ${format(row.candidate.variance)} | ${row.candidate.measured}/${row.candidate.observations} | ${row.change} | ${row.varianceChange} | ${row.comparability.comparable ? "comparable" : cell(row.comparability.mismatchedFields.join(", "))} | ${row.hardGate ? "hard" : "informational"} |`,
|
|
20
|
+
);
|
|
21
|
+
const scenarioRows = summary.scenarioReports.map(
|
|
22
|
+
(report) => `| ${cell(report.id)} | ${changeCounts(report.metrics)} | ${changeCounts(report.variance)} |`,
|
|
23
|
+
);
|
|
24
|
+
const roleRows = summary.roleReports.map(
|
|
25
|
+
(report) => `| ${cell(report.id)} | ${changeCounts(report.metrics)} | ${changeCounts(report.variance)} |`,
|
|
26
|
+
);
|
|
27
|
+
return [
|
|
28
|
+
`# Eval comparison ${summary.baselineEvalId} → ${summary.candidateEvalId}`,
|
|
29
|
+
"",
|
|
30
|
+
`Behavioral hard gate: **${summary.hardGate.pass ? "pass" : "fail"}**`,
|
|
31
|
+
...summary.hardGate.failures.map(
|
|
32
|
+
(failure) =>
|
|
33
|
+
`- Hard failure: ${failure.scenarioId} / ${failure.role} / ${failure.target.id}/${failure.target.model ?? "none"} / ${failure.metric}: ${failure.change}`,
|
|
34
|
+
),
|
|
35
|
+
...summary.envelopeMismatches.map(
|
|
36
|
+
(mismatch) =>
|
|
37
|
+
`- Incomparable envelope: ${mismatch.scenarioId} / ${mismatch.role} / ${mismatch.target.id}/${mismatch.target.model ?? "none"}: ${mismatch.fields.join(", ")}`,
|
|
38
|
+
),
|
|
39
|
+
...summary.affectedCorpusResults.map(
|
|
40
|
+
(result) => `- Affected corpus result: ${result.scenarioId} / ${result.role}: ${result.changedFields.join(", ")}`,
|
|
41
|
+
),
|
|
42
|
+
`Pass-rate delta: ${(summary.passRateDelta * 100).toFixed(2)}%`,
|
|
43
|
+
`Token delta: ${summary.tokenDelta === null ? "unmeasured" : summary.tokenDelta}`,
|
|
44
|
+
`Wall-time delta ms: ${summary.wallTimeDelta}`,
|
|
45
|
+
"",
|
|
46
|
+
"| Scenario | Role | Target/model | Family | Metric | Baseline mean | Baseline variance | Baseline measured | Candidate mean | Candidate variance | Candidate measured | Change | Variance | Comparability | Gate |",
|
|
47
|
+
"|---|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|",
|
|
48
|
+
...rows,
|
|
49
|
+
"",
|
|
50
|
+
"## Per-scenario baseline/candidate report",
|
|
51
|
+
"",
|
|
52
|
+
"| Scenario | Metric changes | Variance changes |",
|
|
53
|
+
"|---|---|---|",
|
|
54
|
+
...scenarioRows,
|
|
55
|
+
"",
|
|
56
|
+
"## Per-role baseline/candidate report",
|
|
57
|
+
"",
|
|
58
|
+
"| Role | Metric changes | Variance changes |",
|
|
59
|
+
"|---|---|---|",
|
|
60
|
+
...roleRows,
|
|
61
|
+
"",
|
|
62
|
+
].join("\n");
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function renderJunit(summary: EvalCompareV4Summary): string {
|
|
66
|
+
const failures = new Set(
|
|
67
|
+
summary.hardGate.failures.map((failure) =>
|
|
68
|
+
JSON.stringify([failure.scenarioId, failure.role, failure.target.id, failure.target.model, failure.metric]),
|
|
69
|
+
),
|
|
70
|
+
);
|
|
71
|
+
const represented = new Set<string>();
|
|
72
|
+
const cases = summary.behavioralMetrics.map((row) => {
|
|
73
|
+
const name = `${row.scenarioId}[${row.role}:${row.target.id}:${row.target.model ?? "none"}].${row.metric}`;
|
|
74
|
+
const key = JSON.stringify([row.scenarioId, row.role, row.target.id, row.target.model, row.metric]);
|
|
75
|
+
represented.add(key);
|
|
76
|
+
const detail = `change=${row.change} variance=${row.varianceChange} baseline=${format(row.baseline.mean)} candidate=${format(row.candidate.mean)}`;
|
|
77
|
+
return failures.has(key)
|
|
78
|
+
? ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><failure message="${escapeXml(row.change)}">${escapeXml(detail)}</failure></testcase>`
|
|
79
|
+
: ` <testcase classname="eval.behavior.${escapeXml(row.family)}" name="${escapeXml(name)}"><system-out>${escapeXml(detail)}</system-out></testcase>`;
|
|
80
|
+
});
|
|
81
|
+
for (const failure of summary.hardGate.failures) {
|
|
82
|
+
const key = JSON.stringify([
|
|
83
|
+
failure.scenarioId,
|
|
84
|
+
failure.role,
|
|
85
|
+
failure.target.id,
|
|
86
|
+
failure.target.model,
|
|
87
|
+
failure.metric,
|
|
88
|
+
]);
|
|
89
|
+
if (represented.has(key)) continue;
|
|
90
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].${failure.metric}`;
|
|
91
|
+
cases.push(
|
|
92
|
+
` <testcase classname="eval.behavior.hard" name="${escapeXml(name)}"><failure message="${escapeXml(failure.change)}">hard behavioral gate</failure></testcase>`,
|
|
93
|
+
);
|
|
94
|
+
}
|
|
95
|
+
for (const failure of summary.hardGate.envelopeFailures) {
|
|
96
|
+
const name = `${failure.scenarioId}[${failure.role}:${failure.target.id}:${failure.target.model ?? "none"}].execution-envelope`;
|
|
97
|
+
cases.push(
|
|
98
|
+
` <testcase classname="eval.behavior.envelope" name="${escapeXml(name)}"><failure message="incomparable">${escapeXml(failure.fields.join(", "))}</failure></testcase>`,
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
return [
|
|
102
|
+
`<testsuite name="eval-comparison" tests="${cases.length}" failures="${summary.hardGate.failures.length + summary.hardGate.envelopeFailures.length}">`,
|
|
103
|
+
...cases,
|
|
104
|
+
"</testsuite>",
|
|
105
|
+
"",
|
|
106
|
+
].join("\n");
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function changeCounts(counts: {
|
|
110
|
+
improved: number;
|
|
111
|
+
regressed: number;
|
|
112
|
+
unchanged: number;
|
|
113
|
+
incomparable: number;
|
|
114
|
+
}): string {
|
|
115
|
+
return `improved ${counts.improved}, regressed ${counts.regressed}, unchanged ${counts.unchanged}, incomparable ${counts.incomparable}`;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function format(value: number | null): string {
|
|
119
|
+
return value === null ? "null" : Number.isInteger(value) ? String(value) : value.toFixed(4);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
function cell(value: string): string {
|
|
123
|
+
return value.replaceAll("|", "\\|");
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
function escapeXml(value: string): string {
|
|
127
|
+
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">").replaceAll('"', """);
|
|
128
|
+
}
|
|
@@ -1,17 +1,31 @@
|
|
|
1
1
|
import type { EvalArtifactV4 } from "../schema/artifact.js";
|
|
2
2
|
|
|
3
3
|
export function renderEvalJunitReportV4(artifact: EvalArtifactV4): string {
|
|
4
|
+
let failures = 0;
|
|
5
|
+
let skipped = 0;
|
|
4
6
|
const cases = artifact.results
|
|
5
7
|
.map((result) => {
|
|
6
8
|
const name = escapeXml(
|
|
7
9
|
`${result.taskId}[${result.target.id}:${result.target.model ?? "default"}:${result.repeatIndex}]`,
|
|
8
10
|
);
|
|
9
|
-
if (result.pass)
|
|
10
|
-
|
|
11
|
+
if (!result.pass) {
|
|
12
|
+
failures += 1;
|
|
13
|
+
return ` <testcase name="${name}"><failure message="${escapeXml(result.failureClass ?? "failed")}" /></testcase>`;
|
|
14
|
+
}
|
|
15
|
+
const outcome = result.behavioral?.outcome;
|
|
16
|
+
if (outcome === "behavioral_failure" || outcome === "infrastructure_failure") {
|
|
17
|
+
failures += 1;
|
|
18
|
+
return ` <testcase name="${name}"><failure message="${escapeXml(outcome)}" /></testcase>`;
|
|
19
|
+
}
|
|
20
|
+
if (outcome === "unknown" || outcome === "unmeasured") {
|
|
21
|
+
skipped += 1;
|
|
22
|
+
return ` <testcase name="${name}"><skipped message="behavioral ${escapeXml(outcome)}" /></testcase>`;
|
|
23
|
+
}
|
|
24
|
+
return ` <testcase name="${name}" />`;
|
|
11
25
|
})
|
|
12
26
|
.join("\n");
|
|
13
27
|
return [
|
|
14
|
-
`<testsuite name="${escapeXml(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${
|
|
28
|
+
`<testsuite name="${escapeXml(artifact.suite.id)}" tests="${artifact.summary.runs}" failures="${failures}" skipped="${skipped}">`,
|
|
15
29
|
cases,
|
|
16
30
|
"</testsuite>",
|
|
17
31
|
"",
|
|
@@ -8,11 +8,11 @@ export function renderEvalMarkdownReportV4(artifact: EvalArtifactV4): string {
|
|
|
8
8
|
`Target: ${artifact.matrix.target}`,
|
|
9
9
|
`Pass rate: ${(artifact.summary.passRate * 100).toFixed(2)}%`,
|
|
10
10
|
"",
|
|
11
|
-
"| Task | Target | Model | Repeat |
|
|
12
|
-
"
|
|
11
|
+
"| Task | Role | Target | Model | Repeat | Result | Behavioral | Failure |",
|
|
12
|
+
"|---|---|---|---|---:|---|---|---|",
|
|
13
13
|
...artifact.results.map(
|
|
14
14
|
(result) =>
|
|
15
|
-
`| ${result.taskId} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.failureClass ?? ""} |`,
|
|
15
|
+
`| ${result.taskId} | ${result.behavioralMetrics?.role ?? ""} | ${result.target.id} | ${result.target.model ?? ""} | ${result.repeatIndex} | ${result.pass ? "pass" : "fail"} | ${result.behavioral?.outcome ?? "unmeasured"} | ${result.failureClass ?? ""} |`,
|
|
16
16
|
),
|
|
17
17
|
"",
|
|
18
18
|
];
|
|
@@ -2,6 +2,15 @@ import type { EvalArtifactV4 } from "../schema/artifact.js";
|
|
|
2
2
|
|
|
3
3
|
export function renderEvalTextReportV4(artifact: EvalArtifactV4): string {
|
|
4
4
|
const tokens = artifact.summary.tokens;
|
|
5
|
+
const behavioral = artifact.results.flatMap((result) =>
|
|
6
|
+
result.behavioral === undefined ? [] : [result.behavioral.outcome],
|
|
7
|
+
);
|
|
8
|
+
const behavioralSummary =
|
|
9
|
+
behavioral.length === 0
|
|
10
|
+
? []
|
|
11
|
+
: [
|
|
12
|
+
`behavioral: pass=${count(behavioral, "pass")} failure=${count(behavioral, "behavioral_failure")} unknown=${count(behavioral, "unknown")} unmeasured=${count(behavioral, "unmeasured")} infrastructure=${count(behavioral, "infrastructure_failure")}`,
|
|
13
|
+
];
|
|
5
14
|
return [
|
|
6
15
|
`eval: ${artifact.evalId}`,
|
|
7
16
|
`suite: ${artifact.suite.id}`,
|
|
@@ -20,6 +29,11 @@ export function renderEvalTextReportV4(artifact: EvalArtifactV4): string {
|
|
|
20
29
|
? `tokens total: ${tokens.total}`
|
|
21
30
|
: `tokens total: ${tokens.total} (measured in ${tokens.measuredRuns} of ${tokens.runs} runs)`,
|
|
22
31
|
`wall time ms: ${artifact.summary.wallTimeMs}`,
|
|
32
|
+
...behavioralSummary,
|
|
23
33
|
"",
|
|
24
34
|
].join("\n");
|
|
25
35
|
}
|
|
36
|
+
|
|
37
|
+
function count(values: ReadonlyArray<string>, wanted: string): number {
|
|
38
|
+
return values.filter((value) => value === wanted).length;
|
|
39
|
+
}
|
|
@@ -1,8 +1,28 @@
|
|
|
1
1
|
import { subtractEvalHarnessMetrics } from "./harness-metrics.js";
|
|
2
|
+
import type { EvalMetricSource } from "./schema/verdict.js";
|
|
2
3
|
import type { EvalFailureClass, EvalHarnessMetrics, EvalRunArtifact, EvalRunRecord, EvalSummary } from "./types.js";
|
|
3
4
|
|
|
4
5
|
export const EVAL_COMPARE_MATCHING_RULE = "taskId+repeatIndex";
|
|
5
6
|
|
|
7
|
+
export class EvalTrackedMetricSourceMismatchError extends Error {
|
|
8
|
+
constructor(metric: string, baseline: ReadonlyArray<EvalMetricSource>, candidate: ReadonlyArray<EvalMetricSource>) {
|
|
9
|
+
super(
|
|
10
|
+
`tracked metric ${metric} cannot compare estimated and measured values (baseline=${baseline.join(",") || "none"}, candidate=${candidate.join(",") || "none"})`,
|
|
11
|
+
);
|
|
12
|
+
this.name = "EvalTrackedMetricSourceMismatchError";
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/** Refuse a delta when exactly one side contains estimated observations. */
|
|
17
|
+
export function assertComparableTrackedMetricSources(
|
|
18
|
+
metric: string,
|
|
19
|
+
baseline: ReadonlyArray<EvalMetricSource>,
|
|
20
|
+
candidate: ReadonlyArray<EvalMetricSource>,
|
|
21
|
+
): void {
|
|
22
|
+
if (baseline.includes("estimated") === candidate.includes("estimated")) return;
|
|
23
|
+
throw new EvalTrackedMetricSourceMismatchError(metric, baseline, candidate);
|
|
24
|
+
}
|
|
25
|
+
|
|
6
26
|
export interface EvalCompareTotals {
|
|
7
27
|
passed: number;
|
|
8
28
|
failed: number;
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { isAbsolute, relative, resolve, sep } from "node:path";
|
|
1
2
|
import { shellQuote } from "../../../core/shell-quote.js";
|
|
2
3
|
import { clioStateDir } from "../../../core/xdg.js";
|
|
3
4
|
import {
|
|
@@ -17,6 +18,7 @@ export async function runClioRunRunner(
|
|
|
17
18
|
timeoutMs: number,
|
|
18
19
|
target: EvalSuiteTargetV2,
|
|
19
20
|
env?: NodeJS.ProcessEnv,
|
|
21
|
+
readObservation?: { allowedPaths: string[]; decoyPaths: string[] },
|
|
20
22
|
): Promise<EvalRunnerOutput> {
|
|
21
23
|
const prompt = runner.prompt ?? "";
|
|
22
24
|
const args = [
|
|
@@ -28,6 +30,7 @@ export async function runClioRunRunner(
|
|
|
28
30
|
shellQuote(target.id),
|
|
29
31
|
...(target.model === undefined ? [] : ["--model", shellQuote(target.model)]),
|
|
30
32
|
...(target.thinking === undefined ? [] : ["--thinking", shellQuote(target.thinking)]),
|
|
33
|
+
...(runner.autonomy === undefined ? [] : ["--autonomy", runner.autonomy]),
|
|
31
34
|
shellQuote(prompt),
|
|
32
35
|
];
|
|
33
36
|
const result = await runShellCommand(`${process.execPath} ${args.join(" ")}`, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
@@ -36,6 +39,7 @@ export async function runClioRunRunner(
|
|
|
36
39
|
const tokens = result.usage;
|
|
37
40
|
const toolMetricStream = result.metricJsonl.length > 0 ? result.metricJsonl : result.stdout;
|
|
38
41
|
const tools = toolCallMetricsFromJsonl(toolMetricStream);
|
|
42
|
+
const behavioralTools = toolBehaviorMetricEntriesFromJsonl(toolMetricStream, cwd, readObservation);
|
|
39
43
|
// Evidence metrics resolve only from the sealed receipt the --agent path
|
|
40
44
|
// prints; a runner without a receipt leaves them absent so any gate on
|
|
41
45
|
// them fails closed instead of reading prose labels.
|
|
@@ -61,6 +65,7 @@ export async function runClioRunRunner(
|
|
|
61
65
|
"tools.totalCalls": tools.totalCalls,
|
|
62
66
|
"tools.failed": tools.failed,
|
|
63
67
|
"tools.blocked": tools.blocked,
|
|
68
|
+
...behavioralTools,
|
|
64
69
|
"verifier.exitCode": result.exitCode,
|
|
65
70
|
...(receipt === null ? {} : evidenceMetricsFromReceipt(receipt, { envelope })),
|
|
66
71
|
...(receipt === null
|
|
@@ -70,8 +75,11 @@ export async function runClioRunRunner(
|
|
|
70
75
|
artifacts: {
|
|
71
76
|
stdout: result.stdout,
|
|
72
77
|
stderr: result.stderr,
|
|
78
|
+
callLedger: JSON.stringify(result.ledgerEntries),
|
|
73
79
|
...(receipt === null ? {} : { receipt: JSON.stringify(receipt) }),
|
|
74
80
|
},
|
|
81
|
+
receipt,
|
|
82
|
+
ledgerEntries: result.ledgerEntries,
|
|
75
83
|
};
|
|
76
84
|
}
|
|
77
85
|
|
|
@@ -128,6 +136,125 @@ export function toolCallMetricsFromJsonl(stdout: string): ToolCallMetrics {
|
|
|
128
136
|
return canonicalFinishes.totalCalls > 0 ? canonicalFinishes : executionEnds;
|
|
129
137
|
}
|
|
130
138
|
|
|
139
|
+
interface BehavioralToolTerminal {
|
|
140
|
+
callId: string | null;
|
|
141
|
+
tool: string;
|
|
142
|
+
outcome: ToolOutcome;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Reduce the same terminal stream to bounded behavioral facts. Tool names are
|
|
147
|
+
* a dynamic metric suffix because extensions may add tools. Read paths become
|
|
148
|
+
* bounded counters against the suite's explicit public allowlist and decoy
|
|
149
|
+
* list; they never become behavioral fact values or evidence excerpts.
|
|
150
|
+
*/
|
|
151
|
+
export function toolBehaviorMetricEntriesFromJsonl(
|
|
152
|
+
stdout: string,
|
|
153
|
+
cwd: string,
|
|
154
|
+
readObservation?: { allowedPaths: string[]; decoyPaths: string[] },
|
|
155
|
+
): Record<string, number> {
|
|
156
|
+
const starts = new Map<string, { tool: string; path: string | null }>();
|
|
157
|
+
const readPaths = new Set<string>();
|
|
158
|
+
const executionEnds: BehavioralToolTerminal[] = [];
|
|
159
|
+
const canonicalFinishes: BehavioralToolTerminal[] = [];
|
|
160
|
+
const seenExecution = new Set<string>();
|
|
161
|
+
const seenCanonical = new Set<string>();
|
|
162
|
+
|
|
163
|
+
for (const line of stdout.split(/\r?\n/)) {
|
|
164
|
+
if (line.trim().length === 0) continue;
|
|
165
|
+
let event: Record<string, unknown>;
|
|
166
|
+
try {
|
|
167
|
+
const parsed: unknown = JSON.parse(line);
|
|
168
|
+
if (!isRecord(parsed)) continue;
|
|
169
|
+
event = parsed;
|
|
170
|
+
} catch {
|
|
171
|
+
continue;
|
|
172
|
+
}
|
|
173
|
+
if (event.type === "tool_execution_start") {
|
|
174
|
+
const callId = stringField(event, "toolCallId");
|
|
175
|
+
const tool = stringField(event, "toolName");
|
|
176
|
+
if (callId === undefined || tool === undefined) continue;
|
|
177
|
+
const path = tool === "read" && isRecord(event.args) ? toolPath(event.args) : null;
|
|
178
|
+
starts.set(callId, { tool, path });
|
|
179
|
+
if (path !== null) readPaths.add(normalizeObservedPath(cwd, path));
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
if (event.type === "tool_execution_end") {
|
|
183
|
+
const callId = stringField(event, "toolCallId") ?? null;
|
|
184
|
+
if (callId !== null && seenExecution.has(callId)) continue;
|
|
185
|
+
if (callId !== null) seenExecution.add(callId);
|
|
186
|
+
const tool = stringField(event, "toolName") ?? (callId === null ? undefined : starts.get(callId)?.tool);
|
|
187
|
+
if (tool === undefined) continue;
|
|
188
|
+
executionEnds.push({ callId, tool, outcome: toolOutcome(event) ?? (event.isError === true ? "error" : "ok") });
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
if (event.type !== "clio_tool_finish" || !isRecord(event.payload)) continue;
|
|
192
|
+
const outcome = toolOutcome(event.payload);
|
|
193
|
+
const tool = stringField(event.payload, "tool");
|
|
194
|
+
if (outcome === undefined || tool === undefined) continue;
|
|
195
|
+
const callId = stringField(event.payload, "toolCallId") ?? stringField(event, "toolCallId") ?? null;
|
|
196
|
+
if (callId !== null && seenCanonical.has(callId)) continue;
|
|
197
|
+
if (callId !== null) seenCanonical.add(callId);
|
|
198
|
+
canonicalFinishes.push({ callId, tool, outcome });
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
const terminals = canonicalFinishes.length > 0 ? canonicalFinishes : executionEnds;
|
|
202
|
+
const calls = new Map<string, number>();
|
|
203
|
+
const blocked = new Map<string, number>();
|
|
204
|
+
for (const terminal of terminals) {
|
|
205
|
+
const tool = metricToolName(terminal.tool);
|
|
206
|
+
calls.set(tool, (calls.get(tool) ?? 0) + 1);
|
|
207
|
+
if (terminal.outcome === "blocked") blocked.set(tool, (blocked.get(tool) ?? 0) + 1);
|
|
208
|
+
}
|
|
209
|
+
const namedTools = new Set(["bash", "dispatch", "read", ...calls.keys(), ...blocked.keys()]);
|
|
210
|
+
const entries: Record<string, number> = { "tools.read.distinctPaths": readPaths.size };
|
|
211
|
+
for (const tool of [...namedTools].sort()) {
|
|
212
|
+
entries[`tools.calls.${tool}`] = calls.get(tool) ?? 0;
|
|
213
|
+
entries[`tools.blocked.${tool}`] = blocked.get(tool) ?? 0;
|
|
214
|
+
}
|
|
215
|
+
if (readObservation !== undefined) {
|
|
216
|
+
const allowed = readObservation.allowedPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
217
|
+
const decoys = readObservation.decoyPaths.map((path) => normalizeObservedPath(cwd, path));
|
|
218
|
+
entries["tools.read.outsideAllowed"] = [...readPaths].filter(
|
|
219
|
+
(path) => !allowed.some((root) => pathWithin(path, root)),
|
|
220
|
+
).length;
|
|
221
|
+
entries["tools.read.decoyHits"] = [...readPaths].filter((path) =>
|
|
222
|
+
decoys.some((root) => pathWithin(path, root)),
|
|
223
|
+
).length;
|
|
224
|
+
}
|
|
225
|
+
return entries;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
function toolPath(args: Record<string, unknown>): string | null {
|
|
229
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
230
|
+
const value = args[field];
|
|
231
|
+
if (typeof value === "string" && value.length > 0 && value.length <= 4_096) return value;
|
|
232
|
+
}
|
|
233
|
+
return null;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
function normalizeObservedPath(cwd: string, path: string): string {
|
|
237
|
+
const absolute = resolve(cwd, path);
|
|
238
|
+
const local = relative(cwd, absolute);
|
|
239
|
+
return (isAbsolute(path) && (local.startsWith("..") || isAbsolute(local)) ? absolute : local || ".")
|
|
240
|
+
.split(sep)
|
|
241
|
+
.join("/");
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
function pathWithin(path: string, root: string): boolean {
|
|
245
|
+
if (root === ".") return !isAbsolute(path) && path !== ".." && !path.startsWith("../");
|
|
246
|
+
return path === root || path.startsWith(`${root}/`);
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
function metricToolName(tool: string): string {
|
|
250
|
+
return (
|
|
251
|
+
tool
|
|
252
|
+
.toLowerCase()
|
|
253
|
+
.replaceAll(/[^a-z0-9_-]/gu, "_")
|
|
254
|
+
.slice(0, 64) || "unknown"
|
|
255
|
+
);
|
|
256
|
+
}
|
|
257
|
+
|
|
131
258
|
function recordToolOutcome(metrics: ToolCallMetrics, outcome: ToolOutcome): void {
|
|
132
259
|
metrics.totalCalls += 1;
|
|
133
260
|
if (outcome === "error") metrics.failed += 1;
|
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import { performance } from "node:perf_hooks";
|
|
3
|
+
import type { RunReceipt } from "../../dispatch/types.js";
|
|
4
|
+
import type { SessionEntry } from "../../session/entries.js";
|
|
5
|
+
import { createEvalCallLedgerFold } from "../metrics/call-ledger-stream.js";
|
|
3
6
|
import {
|
|
4
7
|
addFleetLoopObservations,
|
|
5
8
|
createFleetLoopFold,
|
|
@@ -33,6 +36,10 @@ export interface EvalRunnerOutput {
|
|
|
33
36
|
wallTimeMs: number;
|
|
34
37
|
metrics: Record<string, number | string | boolean | null>;
|
|
35
38
|
artifacts: Record<string, string | string[] | null>;
|
|
39
|
+
/** Parsed sealed receipt when this runner exposes one. */
|
|
40
|
+
receipt?: RunReceipt | null;
|
|
41
|
+
/** Structured per-call facts folded before bounded stdout truncation. */
|
|
42
|
+
ledgerEntries?: SessionEntry[];
|
|
36
43
|
}
|
|
37
44
|
|
|
38
45
|
export interface ShellCommandResult {
|
|
@@ -46,6 +53,7 @@ export interface ShellCommandResult {
|
|
|
46
53
|
/** Wire-stream structural invariants folded live, for the same reason. */
|
|
47
54
|
streamInvariants: EvalStreamInvariants;
|
|
48
55
|
fleetLoops: EvalFleetLoopObservation;
|
|
56
|
+
ledgerEntries: SessionEntry[];
|
|
49
57
|
stderr: string;
|
|
50
58
|
wallTimeMs: number;
|
|
51
59
|
timedOut: boolean;
|
|
@@ -75,6 +83,7 @@ export async function runExternalCommandRunner(
|
|
|
75
83
|
let usage = UNMEASURED_TOKEN_USAGE;
|
|
76
84
|
let streamInvariants = EMPTY_STREAM_INVARIANTS;
|
|
77
85
|
let fleetLoops = EMPTY_FLEET_LOOP_OBSERVATION;
|
|
86
|
+
const ledgerEntries: SessionEntry[] = [];
|
|
78
87
|
for (const command of commands) {
|
|
79
88
|
const result = await runShellCommand(command, cwd, runner.timeoutMs ?? timeoutMs, env);
|
|
80
89
|
stdout = appendLimited(stdout, result.stdout);
|
|
@@ -83,6 +92,7 @@ export async function runExternalCommandRunner(
|
|
|
83
92
|
usage = addTokenStreamUsage(usage, result.usage);
|
|
84
93
|
streamInvariants = addStreamInvariants(streamInvariants, result.streamInvariants);
|
|
85
94
|
fleetLoops = addFleetLoopObservations(fleetLoops, result.fleetLoops);
|
|
95
|
+
ledgerEntries.push(...result.ledgerEntries);
|
|
86
96
|
if (result.exitCode !== 0) {
|
|
87
97
|
return {
|
|
88
98
|
assignmentId: null,
|
|
@@ -99,6 +109,7 @@ export async function runExternalCommandRunner(
|
|
|
99
109
|
"verifier.exitCode": result.exitCode,
|
|
100
110
|
},
|
|
101
111
|
artifacts: {},
|
|
112
|
+
ledgerEntries,
|
|
102
113
|
};
|
|
103
114
|
}
|
|
104
115
|
}
|
|
@@ -117,6 +128,7 @@ export async function runExternalCommandRunner(
|
|
|
117
128
|
"verifier.exitCode": 0,
|
|
118
129
|
},
|
|
119
130
|
artifacts: {},
|
|
131
|
+
ledgerEntries,
|
|
120
132
|
};
|
|
121
133
|
}
|
|
122
134
|
|
|
@@ -137,6 +149,7 @@ export function runShellCommand(
|
|
|
137
149
|
const usageFold = createTokenUsageFold();
|
|
138
150
|
const streamFold = createStreamInvariantFold();
|
|
139
151
|
const fleetLoopFold = createFleetLoopFold();
|
|
152
|
+
const callLedgerFold = createEvalCallLedgerFold();
|
|
140
153
|
let timedOut = false;
|
|
141
154
|
let settled = false;
|
|
142
155
|
const child = spawn(command, {
|
|
@@ -155,6 +168,7 @@ export function runShellCommand(
|
|
|
155
168
|
usageFold.push(chunk);
|
|
156
169
|
streamFold.push(chunk);
|
|
157
170
|
fleetLoopFold.push(chunk);
|
|
171
|
+
callLedgerFold.push(chunk);
|
|
158
172
|
});
|
|
159
173
|
child.stderr.on("data", (chunk: string) => {
|
|
160
174
|
stderr = appendLimited(stderr, chunk);
|
|
@@ -176,6 +190,7 @@ export function runShellCommand(
|
|
|
176
190
|
usage: usageFold.usage(),
|
|
177
191
|
streamInvariants: streamFold.invariants(),
|
|
178
192
|
fleetLoops: fleetLoopFold.observation(),
|
|
193
|
+
ledgerEntries: callLedgerFold.entries(),
|
|
179
194
|
stderr,
|
|
180
195
|
wallTimeMs: Math.round(performance.now() - started),
|
|
181
196
|
timedOut,
|
|
@@ -284,9 +299,11 @@ function compactMetricEvent(event: Record<string, unknown>): Record<string, unkn
|
|
|
284
299
|
toolName,
|
|
285
300
|
...(toolName === "dispatch" && isRecord(event.args)
|
|
286
301
|
? { args: event.args }
|
|
287
|
-
: toolName === "
|
|
288
|
-
? { args:
|
|
289
|
-
:
|
|
302
|
+
: toolName === "read" && isRecord(event.args)
|
|
303
|
+
? { args: boundedReadArgs(event.args) }
|
|
304
|
+
: toolName === "code_nav" && isRecord(event.args)
|
|
305
|
+
? { args: { mode: event.args.mode } }
|
|
306
|
+
: {}),
|
|
290
307
|
};
|
|
291
308
|
}
|
|
292
309
|
if (type === "tool_execution_end") {
|
|
@@ -313,6 +330,14 @@ function compactMetricEvent(event: Record<string, unknown>): Record<string, unkn
|
|
|
313
330
|
};
|
|
314
331
|
}
|
|
315
332
|
|
|
333
|
+
function boundedReadArgs(args: Record<string, unknown>): Record<string, string> {
|
|
334
|
+
for (const field of ["path", "filePath", "file_path"]) {
|
|
335
|
+
const value = args[field];
|
|
336
|
+
if (typeof value === "string" && value.length > 0) return { [field]: value.slice(0, 4_096) };
|
|
337
|
+
}
|
|
338
|
+
return {};
|
|
339
|
+
}
|
|
340
|
+
|
|
316
341
|
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
317
342
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
318
343
|
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import type { EvalArtifactResultV4 } from "./artifact.js";
|
|
3
|
+
import {
|
|
4
|
+
assertEvalBehaviorReferencesVerdictV1,
|
|
5
|
+
type EvalBehaviorFactSourceV1,
|
|
6
|
+
type EvalBehaviorJudgeFactV1,
|
|
7
|
+
type EvalBehaviorScenarioV1,
|
|
8
|
+
type EvalBehaviorVerdictV1,
|
|
9
|
+
judgeEvalBehaviorV1,
|
|
10
|
+
} from "./behavioral.js";
|
|
11
|
+
import {
|
|
12
|
+
EVAL_VERDICT_SCHEMA_V1,
|
|
13
|
+
type EvalTrackedMetricsV1,
|
|
14
|
+
type EvalVerdictEnvelopeV1,
|
|
15
|
+
parseEvalVerdictEnvelopeV1,
|
|
16
|
+
} from "./verdict.js";
|
|
17
|
+
|
|
18
|
+
type AdaptableSuiteResult = Pick<
|
|
19
|
+
EvalArtifactResultV4,
|
|
20
|
+
"assignmentId" | "terminalReceiptDigest" | "taskId" | "repeatIndex" | "pass" | "failureClass" | "metrics"
|
|
21
|
+
>;
|
|
22
|
+
|
|
23
|
+
/** Adapt one Suite v2 matrix result into the versioned verdict carried by Artifact v4. */
|
|
24
|
+
export function adaptSuiteV2ResultToVerdictV1(
|
|
25
|
+
result: AdaptableSuiteResult,
|
|
26
|
+
trackedMetrics: EvalTrackedMetricsV1,
|
|
27
|
+
): EvalVerdictEnvelopeV1 {
|
|
28
|
+
// The suite runner owns the one final pass decision. In particular, a
|
|
29
|
+
// declared grader failure changes `pass` without pretending the runner or
|
|
30
|
+
// its invariants broke; every other failed result is a machinery failure.
|
|
31
|
+
const machinery = result.pass || result.failureClass === "grader_failed" ? "ok" : "infrastructure_failure";
|
|
32
|
+
const outcome = result.pass ? "pass" : "fail";
|
|
33
|
+
const graderExitCode = result.metrics["task.exitCode"];
|
|
34
|
+
return parseEvalVerdictEnvelopeV1({
|
|
35
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
36
|
+
scenarioId: result.taskId,
|
|
37
|
+
trialIndex: result.repeatIndex,
|
|
38
|
+
outcome,
|
|
39
|
+
machinery,
|
|
40
|
+
reason: result.pass ? null : (result.failureClass ?? "result_failed"),
|
|
41
|
+
trackedMetrics,
|
|
42
|
+
behavioral: null,
|
|
43
|
+
evidence: {
|
|
44
|
+
assignmentId: result.assignmentId,
|
|
45
|
+
terminalReceiptDigest: result.terminalReceiptDigest,
|
|
46
|
+
graderExitCode: typeof graderExitCode === "number" && Number.isInteger(graderExitCode) ? graderExitCode : null,
|
|
47
|
+
},
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Adapt Suite v2 metrics into deterministic, bounded facts for a behavioral scenario. */
|
|
52
|
+
export function adaptSuiteV2ResultToBehaviorV1(
|
|
53
|
+
result: AdaptableSuiteResult,
|
|
54
|
+
verdict: EvalVerdictEnvelopeV1,
|
|
55
|
+
scenario: EvalBehaviorScenarioV1,
|
|
56
|
+
): EvalBehaviorVerdictV1 {
|
|
57
|
+
const requestedFacts = new Set(
|
|
58
|
+
[...scenario.expectedBehavior, ...scenario.forbiddenBehavior].map(
|
|
59
|
+
(rule) => `${rule.fact.source}\u0000${rule.fact.key}`,
|
|
60
|
+
),
|
|
61
|
+
);
|
|
62
|
+
const observedSources = new Set<EvalBehaviorFactSourceV1>();
|
|
63
|
+
const facts = Object.entries(result.metrics).flatMap(([key, value]) => {
|
|
64
|
+
if (value === null) return [];
|
|
65
|
+
const source = metricFactSource(key);
|
|
66
|
+
observedSources.add(source);
|
|
67
|
+
if (!requestedFacts.has(`${source}\u0000${key}`)) return [];
|
|
68
|
+
const serialized = JSON.stringify({ source, key, value });
|
|
69
|
+
const digest = createHash("sha256").update(serialized, "utf8").digest("hex");
|
|
70
|
+
const fact: EvalBehaviorJudgeFactV1 = {
|
|
71
|
+
id: `metric-${digest.slice(0, 16)}`,
|
|
72
|
+
source,
|
|
73
|
+
key,
|
|
74
|
+
value,
|
|
75
|
+
evidence: { locator: `artifact.metrics.${key}`, digest, excerpt: serialized.slice(0, 1_000) },
|
|
76
|
+
};
|
|
77
|
+
return [fact];
|
|
78
|
+
});
|
|
79
|
+
const allSources: EvalBehaviorFactSourceV1[] = ["transcript", "tool", "receipt", "grader"];
|
|
80
|
+
const unavailableSources = allSources.filter(
|
|
81
|
+
(source) => !observedSources.has(source) || (source === "tool" && scenario.execution.toolTarget === "none"),
|
|
82
|
+
);
|
|
83
|
+
const behavior = judgeEvalBehaviorV1(scenario, verdict, {
|
|
84
|
+
facts,
|
|
85
|
+
unavailableSources,
|
|
86
|
+
infrastructureFailure: verdict.machinery === "infrastructure_failure",
|
|
87
|
+
});
|
|
88
|
+
assertEvalBehaviorReferencesVerdictV1(behavior, verdict);
|
|
89
|
+
return behavior;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function metricFactSource(key: string): EvalBehaviorFactSourceV1 {
|
|
93
|
+
if (key.startsWith("tools.")) return "tool";
|
|
94
|
+
if (
|
|
95
|
+
key.startsWith("task.") ||
|
|
96
|
+
key.startsWith("claims.") ||
|
|
97
|
+
key.startsWith("completion.") ||
|
|
98
|
+
key === "result.pass" ||
|
|
99
|
+
key === "verifier.exitCode"
|
|
100
|
+
)
|
|
101
|
+
return "grader";
|
|
102
|
+
if (
|
|
103
|
+
key.startsWith("receipt.") ||
|
|
104
|
+
key.startsWith("evidence.") ||
|
|
105
|
+
key.startsWith("boundary.") ||
|
|
106
|
+
key.startsWith("loop.") ||
|
|
107
|
+
key.startsWith("cost.")
|
|
108
|
+
)
|
|
109
|
+
return "receipt";
|
|
110
|
+
return "transcript";
|
|
111
|
+
}
|
|
@@ -1,4 +1,10 @@
|
|
|
1
|
+
import type { EvalScenarioAggregateV1 } from "../metrics/aggregate.js";
|
|
1
2
|
import type { EvalClioProvenance, EvalEnvironmentProvenance } from "../types.js";
|
|
3
|
+
import type { EvalBehaviorVerdictV1 } from "./behavioral.js";
|
|
4
|
+
import type { EvalBehaviorMetricsV1 } from "./behavioral-metrics.js";
|
|
5
|
+
import type { EvalExecutionEnvelopeV1, EvalExecutionMatrixDimensionV1 } from "./execution-envelope.js";
|
|
6
|
+
import type { EvalServingConfigurationV1 } from "./serving.js";
|
|
7
|
+
import type { EvalVerdictEnvelopeV1 } from "./verdict.js";
|
|
2
8
|
|
|
3
9
|
export interface EvalTokenMetricsV4 {
|
|
4
10
|
input: number;
|
|
@@ -45,6 +51,14 @@ export interface EvalArtifactResultV4 extends EvalArtifactAssignmentReference {
|
|
|
45
51
|
failureClass: string | null;
|
|
46
52
|
metrics: Record<string, number | string | boolean | null>;
|
|
47
53
|
artifacts: Record<string, string | string[] | null>;
|
|
54
|
+
/** Additive Suite v2 adapter output. Current runners always populate it. */
|
|
55
|
+
verdict?: EvalVerdictEnvelopeV1;
|
|
56
|
+
/** Optional sibling document that references the unchanged verdict v1 identity. */
|
|
57
|
+
behavioral?: EvalBehaviorVerdictV1;
|
|
58
|
+
/** Additive, typed multi-metric projection for behavioral comparison. */
|
|
59
|
+
behavioralMetrics?: EvalBehaviorMetricsV1;
|
|
60
|
+
/** Exact prompt, route, policy, and project-context identity for a behavioral result. */
|
|
61
|
+
executionEnvelope?: EvalExecutionEnvelopeV1;
|
|
48
62
|
}
|
|
49
63
|
|
|
50
64
|
/** The only current eval artifact format. Routing accepts this version only. */
|
|
@@ -61,8 +75,14 @@ export interface EvalArtifactV4 {
|
|
|
61
75
|
target: string;
|
|
62
76
|
model: string | null;
|
|
63
77
|
thinking: string | null;
|
|
78
|
+
/** Envelope fields intentionally varied by this matrix. */
|
|
79
|
+
dimensions?: EvalExecutionMatrixDimensionV1[];
|
|
64
80
|
};
|
|
81
|
+
/** Exact serving facts used to decide whether two eval runs are comparable. */
|
|
82
|
+
servingConfiguration?: EvalServingConfigurationV1;
|
|
65
83
|
summary: EvalArtifactSummaryV4;
|
|
84
|
+
/** Scenario reductions over the versioned per-trial verdicts. */
|
|
85
|
+
aggregates?: EvalScenarioAggregateV1[];
|
|
66
86
|
results: EvalArtifactResultV4[];
|
|
67
87
|
}
|
|
68
88
|
|