@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -3,13 +3,16 @@ import {
|
|
|
3
3
|
ROUTE_POLICY_VERSION,
|
|
4
4
|
isRouteDecisionAgentSelection,
|
|
5
5
|
isRoutingIntent
|
|
6
|
-
} from "./chunk-
|
|
6
|
+
} from "./chunk-VPTUJU4P.js";
|
|
7
7
|
import {
|
|
8
8
|
isExecutionRole
|
|
9
9
|
} from "./chunk-H7IXIC72.js";
|
|
10
|
+
import {
|
|
11
|
+
readSettings
|
|
12
|
+
} from "./chunk-PNY46YEY.js";
|
|
10
13
|
import {
|
|
11
14
|
assertSafeId
|
|
12
|
-
} from "./chunk-
|
|
15
|
+
} from "./chunk-KV2AOLDF.js";
|
|
13
16
|
import {
|
|
14
17
|
readClioVersion,
|
|
15
18
|
resolvePackageRoot
|
|
@@ -357,6 +360,16 @@ function evalHarnessMetricsFromCommands(commands) {
|
|
|
357
360
|
).length
|
|
358
361
|
};
|
|
359
362
|
}
|
|
363
|
+
function evalHarnessMetricsFromReceipt(receipt, extras = {}) {
|
|
364
|
+
return {
|
|
365
|
+
receiptCount: 1,
|
|
366
|
+
toolCalls: receipt.toolCalls,
|
|
367
|
+
retries: extras.retries ?? 0,
|
|
368
|
+
safetyBlocks: receipt.safety?.decisions.blocked ?? 0,
|
|
369
|
+
correctionLatencyMs: extras.correctionLatencyMs ?? 0,
|
|
370
|
+
validationEvidence: extras.validationEvidence ?? 0
|
|
371
|
+
};
|
|
372
|
+
}
|
|
360
373
|
function sumEvalHarnessMetrics(records) {
|
|
361
374
|
return records.reduce((total, record) => addEvalHarnessMetrics(total, record.harness), {
|
|
362
375
|
...ZERO_EVAL_HARNESS_METRICS
|
|
@@ -732,6 +745,887 @@ function isErrorWithCode(error) {
|
|
|
732
745
|
init_esm_shims();
|
|
733
746
|
import { readFile as readFile2 } from "node:fs/promises";
|
|
734
747
|
import { join as join2 } from "node:path";
|
|
748
|
+
|
|
749
|
+
// src/domains/eval/schema/behavioral.ts
|
|
750
|
+
init_esm_shims();
|
|
751
|
+
import { createHash as createHash3 } from "node:crypto";
|
|
752
|
+
|
|
753
|
+
// src/domains/eval/schema/verdict.ts
|
|
754
|
+
init_esm_shims();
|
|
755
|
+
var EVAL_VERDICT_SCHEMA_V1 = "clio.eval.verdict.v1";
|
|
756
|
+
var EVAL_TRACKED_METRIC_NAMES = [
|
|
757
|
+
"modelCalls",
|
|
758
|
+
"uncachedPrefillTokens",
|
|
759
|
+
"cacheReadTokens",
|
|
760
|
+
"generatedTokens",
|
|
761
|
+
"reasoningTokens",
|
|
762
|
+
"toolCalls",
|
|
763
|
+
"toolErrors",
|
|
764
|
+
"ttftMsFirstCall",
|
|
765
|
+
"wallClockMs",
|
|
766
|
+
"contextTokensAtEnd",
|
|
767
|
+
"compactions"
|
|
768
|
+
];
|
|
769
|
+
function parseEvalVerdictEnvelopeV1(value, source = "verdict") {
|
|
770
|
+
const record = asRecord(value, source);
|
|
771
|
+
if (record.schema !== EVAL_VERDICT_SCHEMA_V1) {
|
|
772
|
+
throw new Error(`${source}.schema: expected ${EVAL_VERDICT_SCHEMA_V1}`);
|
|
773
|
+
}
|
|
774
|
+
const scenarioId = readNonEmptyString(record, source, "scenarioId");
|
|
775
|
+
const trialIndex = readNonNegativeInteger(record, source, "trialIndex");
|
|
776
|
+
const outcome = readOutcome(record.outcome, `${source}.outcome`);
|
|
777
|
+
const machinery = readMachinery(record.machinery, `${source}.machinery`);
|
|
778
|
+
if (machinery === "infrastructure_failure" && outcome === "pass") {
|
|
779
|
+
throw new Error(`${source}: infrastructure_failure cannot carry a pass outcome`);
|
|
780
|
+
}
|
|
781
|
+
if (record.behavioral !== null) throw new Error(`${source}.behavioral: expected null in verdict v1`);
|
|
782
|
+
const evidence = parseEvidence(record.evidence, `${source}.evidence`);
|
|
783
|
+
const reason = record.reason === void 0 ? legacyVerdictReason(outcome, machinery, evidence) : readNullableNonEmptyString(record, source, "reason");
|
|
784
|
+
if (outcome === "fail" && reason === null) throw new Error(`${source}.reason: failed outcome requires a reason`);
|
|
785
|
+
if (outcome !== "fail" && reason !== null) {
|
|
786
|
+
throw new Error(`${source}.reason: only a failed outcome can carry a reason`);
|
|
787
|
+
}
|
|
788
|
+
return {
|
|
789
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
790
|
+
scenarioId,
|
|
791
|
+
trialIndex,
|
|
792
|
+
outcome,
|
|
793
|
+
machinery,
|
|
794
|
+
reason,
|
|
795
|
+
trackedMetrics: parseTrackedMetrics(record.trackedMetrics, `${source}.trackedMetrics`),
|
|
796
|
+
behavioral: null,
|
|
797
|
+
evidence
|
|
798
|
+
};
|
|
799
|
+
}
|
|
800
|
+
function parseTrackedMetrics(value, source) {
|
|
801
|
+
const record = asRecord(value, source);
|
|
802
|
+
const expectedColdReasons = asRecord(record.expectedColdReasons, `${source}.expectedColdReasons`);
|
|
803
|
+
return {
|
|
804
|
+
modelCalls: readSourcedNumber(record.modelCalls, `${source}.modelCalls`),
|
|
805
|
+
uncachedPrefillTokens: readSourcedNumber(record.uncachedPrefillTokens, `${source}.uncachedPrefillTokens`),
|
|
806
|
+
cacheReadTokens: readSourcedNumber(record.cacheReadTokens, `${source}.cacheReadTokens`),
|
|
807
|
+
generatedTokens: readSourcedNumber(record.generatedTokens, `${source}.generatedTokens`),
|
|
808
|
+
reasoningTokens: readSourcedNullableNumber(record.reasoningTokens, `${source}.reasoningTokens`),
|
|
809
|
+
toolCalls: readSourcedNumber(record.toolCalls, `${source}.toolCalls`),
|
|
810
|
+
toolErrors: readSourcedNumber(record.toolErrors, `${source}.toolErrors`),
|
|
811
|
+
ttftMsFirstCall: readSourcedNumber(record.ttftMsFirstCall, `${source}.ttftMsFirstCall`),
|
|
812
|
+
wallClockMs: readSourcedNumber(record.wallClockMs, `${source}.wallClockMs`),
|
|
813
|
+
contextTokensAtEnd: readSourcedNumber(record.contextTokensAtEnd, `${source}.contextTokensAtEnd`),
|
|
814
|
+
compactions: readSourcedNumber(record.compactions, `${source}.compactions`),
|
|
815
|
+
expectedColdReasons: Object.fromEntries(
|
|
816
|
+
Object.entries(expectedColdReasons).map(([reason, metric]) => {
|
|
817
|
+
if (reason.trim().length === 0) throw new Error(`${source}.expectedColdReasons: expected non-empty reason`);
|
|
818
|
+
return [reason, readSourcedNumber(metric, `${source}.expectedColdReasons.${reason}`)];
|
|
819
|
+
})
|
|
820
|
+
)
|
|
821
|
+
};
|
|
822
|
+
}
|
|
823
|
+
function parseEvidence(value, source) {
|
|
824
|
+
const record = asRecord(value, source);
|
|
825
|
+
const terminalReceiptDigest = readNullableString2(record, source, "terminalReceiptDigest");
|
|
826
|
+
if (terminalReceiptDigest !== null && !/^[a-f0-9]{64}$/u.test(terminalReceiptDigest)) {
|
|
827
|
+
throw new Error(`${source}.terminalReceiptDigest: expected sha256 digest or null`);
|
|
828
|
+
}
|
|
829
|
+
const graderExitCode = record.graderExitCode;
|
|
830
|
+
if (graderExitCode !== null && (!Number.isInteger(graderExitCode) || !Number.isFinite(graderExitCode))) {
|
|
831
|
+
throw new Error(`${source}.graderExitCode: expected integer or null`);
|
|
832
|
+
}
|
|
833
|
+
return {
|
|
834
|
+
assignmentId: readNullableString2(record, source, "assignmentId"),
|
|
835
|
+
terminalReceiptDigest,
|
|
836
|
+
graderExitCode
|
|
837
|
+
};
|
|
838
|
+
}
|
|
839
|
+
function legacyVerdictReason(outcome, machinery, evidence) {
|
|
840
|
+
if (outcome !== "fail") return null;
|
|
841
|
+
if (machinery === "infrastructure_failure") return "infrastructure_failure";
|
|
842
|
+
if (evidence.graderExitCode !== null && evidence.graderExitCode !== 0) return "grader_failed";
|
|
843
|
+
return "outcome_failed";
|
|
844
|
+
}
|
|
845
|
+
function readSourcedNumber(value, source) {
|
|
846
|
+
const record = asRecord(value, source);
|
|
847
|
+
return {
|
|
848
|
+
value: readNonNegativeNumber(record.value, `${source}.value`),
|
|
849
|
+
source: readMetricSource(record.source, source)
|
|
850
|
+
};
|
|
851
|
+
}
|
|
852
|
+
function readSourcedNullableNumber(value, source) {
|
|
853
|
+
const record = asRecord(value, source);
|
|
854
|
+
return {
|
|
855
|
+
value: record.value === null ? null : readNonNegativeNumber(record.value, `${source}.value`),
|
|
856
|
+
source: readMetricSource(record.source, source)
|
|
857
|
+
};
|
|
858
|
+
}
|
|
859
|
+
function readMetricSource(value, source) {
|
|
860
|
+
if (value === "ledger" || value === "receipt" || value === "estimated") return value;
|
|
861
|
+
throw new Error(`${source}.source: expected ledger, receipt, or estimated`);
|
|
862
|
+
}
|
|
863
|
+
function readOutcome(value, source) {
|
|
864
|
+
if (value === "pass" || value === "fail" || value === "unmeasured") return value;
|
|
865
|
+
throw new Error(`${source}: expected pass, fail, or unmeasured`);
|
|
866
|
+
}
|
|
867
|
+
function readMachinery(value, source) {
|
|
868
|
+
if (value === "ok" || value === "infrastructure_failure") return value;
|
|
869
|
+
throw new Error(`${source}: expected ok or infrastructure_failure`);
|
|
870
|
+
}
|
|
871
|
+
function readNonNegativeNumber(value, source) {
|
|
872
|
+
if (typeof value !== "number" || !Number.isFinite(value) || value < 0) {
|
|
873
|
+
throw new Error(`${source}: expected non-negative number`);
|
|
874
|
+
}
|
|
875
|
+
return value;
|
|
876
|
+
}
|
|
877
|
+
function readNonNegativeInteger(record, source, field) {
|
|
878
|
+
const value = record[field];
|
|
879
|
+
if (!Number.isInteger(value) || typeof value !== "number" || value < 0) {
|
|
880
|
+
throw new Error(`${source}.${field}: expected non-negative integer`);
|
|
881
|
+
}
|
|
882
|
+
return value;
|
|
883
|
+
}
|
|
884
|
+
function readNonEmptyString(record, source, field) {
|
|
885
|
+
const value = record[field];
|
|
886
|
+
if (typeof value !== "string" || value.trim().length === 0) throw new Error(`${source}.${field}: expected string`);
|
|
887
|
+
return value;
|
|
888
|
+
}
|
|
889
|
+
function readNullableString2(record, source, field) {
|
|
890
|
+
const value = record[field];
|
|
891
|
+
if (value === null) return null;
|
|
892
|
+
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}.${field}: expected string or null`);
|
|
893
|
+
return value;
|
|
894
|
+
}
|
|
895
|
+
function readNullableNonEmptyString(record, source, field) {
|
|
896
|
+
const value = readNullableString2(record, source, field);
|
|
897
|
+
if (value !== null && value.trim().length === 0) {
|
|
898
|
+
throw new Error(`${source}.${field}: expected non-empty string or null`);
|
|
899
|
+
}
|
|
900
|
+
return value;
|
|
901
|
+
}
|
|
902
|
+
function asRecord(value, source) {
|
|
903
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value;
|
|
904
|
+
throw new Error(`${source}: expected object`);
|
|
905
|
+
}
|
|
906
|
+
|
|
907
|
+
// src/domains/eval/schema/behavioral.ts
|
|
908
|
+
var EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1 = "clio.eval.scenario.v1";
|
|
909
|
+
var EVAL_BEHAVIOR_SCHEMA_V1 = "clio.eval.behavior.v1";
|
|
910
|
+
var EVAL_BEHAVIOR_CATEGORIES = [
|
|
911
|
+
"tool_choice",
|
|
912
|
+
"exploration",
|
|
913
|
+
"delegation",
|
|
914
|
+
"safety_comprehension",
|
|
915
|
+
"claim_grounding",
|
|
916
|
+
"denied_tool_recovery",
|
|
917
|
+
"completion_behavior",
|
|
918
|
+
"task_correctness"
|
|
919
|
+
];
|
|
920
|
+
var MAX_RULES = 64;
|
|
921
|
+
var MAX_FACTS = 128;
|
|
922
|
+
var MAX_EVIDENCE_PER_LABEL = 8;
|
|
923
|
+
var MAX_ID_CHARS = 128;
|
|
924
|
+
var MAX_TEXT_CHARS = 1e3;
|
|
925
|
+
var MAX_EXPLANATION_CHARS = 2e3;
|
|
926
|
+
function parseEvalBehaviorScenarioV1(value, source = "behavioral scenario") {
|
|
927
|
+
const record = asRecord2(value, source);
|
|
928
|
+
if (record.schema !== EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1) {
|
|
929
|
+
throw new Error(`${source}.schema: expected ${EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1}`);
|
|
930
|
+
}
|
|
931
|
+
const corpus = asRecord2(record.corpus, `${source}.corpus`);
|
|
932
|
+
const execution = asRecord2(record.execution, `${source}.execution`);
|
|
933
|
+
const subject = asRecord2(execution.subject, `${source}.execution.subject`);
|
|
934
|
+
const mode = execution.mode;
|
|
935
|
+
if (mode !== "machinery-only" && mode !== "model-required") {
|
|
936
|
+
throw new Error(`${source}.execution.mode: expected machinery-only or model-required`);
|
|
937
|
+
}
|
|
938
|
+
const subjectKind = subject.kind;
|
|
939
|
+
if (subjectKind !== "main-agent" && subjectKind !== "worker") {
|
|
940
|
+
throw new Error(`${source}.execution.subject.kind: expected main-agent or worker`);
|
|
941
|
+
}
|
|
942
|
+
const toolTarget = execution.toolTarget;
|
|
943
|
+
if (toolTarget !== "available" && toolTarget !== "none") {
|
|
944
|
+
throw new Error(`${source}.execution.toolTarget: expected available or none`);
|
|
945
|
+
}
|
|
946
|
+
const expectedBehavior = parseRules(record.expectedBehavior, `${source}.expectedBehavior`);
|
|
947
|
+
const forbiddenBehavior = parseRules(record.forbiddenBehavior, `${source}.forbiddenBehavior`);
|
|
948
|
+
const ruleIds = /* @__PURE__ */ new Set();
|
|
949
|
+
for (const rule of [...expectedBehavior, ...forbiddenBehavior]) {
|
|
950
|
+
if (ruleIds.has(rule.id)) throw new Error(`${source}: duplicate behavioral rule id ${rule.id}`);
|
|
951
|
+
ruleIds.add(rule.id);
|
|
952
|
+
}
|
|
953
|
+
if (ruleIds.size === 0) throw new Error(`${source}: expected at least one behavioral rule`);
|
|
954
|
+
const judge = asRecord2(record.judge, `${source}.judge`);
|
|
955
|
+
const maxEvidenceItems = readBoundedInteger(judge.maxEvidenceItems, `${source}.judge.maxEvidenceItems`, 1, MAX_FACTS);
|
|
956
|
+
const maxExplanationChars = readBoundedInteger(
|
|
957
|
+
judge.maxExplanationChars,
|
|
958
|
+
`${source}.judge.maxExplanationChars`,
|
|
959
|
+
1,
|
|
960
|
+
MAX_EXPLANATION_CHARS
|
|
961
|
+
);
|
|
962
|
+
return {
|
|
963
|
+
schema: EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1,
|
|
964
|
+
corpus: {
|
|
965
|
+
id: readId(corpus.id, `${source}.corpus.id`),
|
|
966
|
+
version: readText(corpus.version, `${source}.corpus.version`, MAX_ID_CHARS)
|
|
967
|
+
},
|
|
968
|
+
execution: {
|
|
969
|
+
mode,
|
|
970
|
+
subject: { kind: subjectKind, role: readId(subject.role, `${source}.execution.subject.role`) },
|
|
971
|
+
toolTarget
|
|
972
|
+
},
|
|
973
|
+
expectedBehavior,
|
|
974
|
+
forbiddenBehavior,
|
|
975
|
+
judge: { maxEvidenceItems, maxExplanationChars }
|
|
976
|
+
};
|
|
977
|
+
}
|
|
978
|
+
function canonicalizeEvalBehaviorJudgeInputV1(value, scenario, source = "behavioral judge input") {
|
|
979
|
+
const record = asRecord2(value, source);
|
|
980
|
+
if (!Array.isArray(record.facts)) throw new Error(`${source}.facts: expected array`);
|
|
981
|
+
if (record.facts.length > scenario.judge.maxEvidenceItems || record.facts.length > MAX_FACTS) {
|
|
982
|
+
throw new Error(`${source}.facts: exceeds bounded evidence limit`);
|
|
983
|
+
}
|
|
984
|
+
const facts = record.facts.map((fact, index) => parseFact(fact, `${source}.facts[${index}]`));
|
|
985
|
+
const factIds = /* @__PURE__ */ new Set();
|
|
986
|
+
const factKeys = /* @__PURE__ */ new Set();
|
|
987
|
+
for (const fact of facts) {
|
|
988
|
+
if (factIds.has(fact.id)) throw new Error(`${source}: duplicate fact id ${fact.id}`);
|
|
989
|
+
const key = `${fact.source}\0${fact.key}`;
|
|
990
|
+
if (factKeys.has(key)) throw new Error(`${source}: conflicting fact ${fact.source}.${fact.key}`);
|
|
991
|
+
factIds.add(fact.id);
|
|
992
|
+
factKeys.add(key);
|
|
993
|
+
}
|
|
994
|
+
const unavailableSources = parseSources(record.unavailableSources, `${source}.unavailableSources`);
|
|
995
|
+
const infrastructureFailure = record.infrastructureFailure;
|
|
996
|
+
if (typeof infrastructureFailure !== "boolean") {
|
|
997
|
+
throw new Error(`${source}.infrastructureFailure: expected boolean`);
|
|
998
|
+
}
|
|
999
|
+
const input = {
|
|
1000
|
+
facts: facts.sort((left, right) => left.source.localeCompare(right.source) || left.key.localeCompare(right.key)),
|
|
1001
|
+
unavailableSources: [...new Set(unavailableSources)].sort(),
|
|
1002
|
+
infrastructureFailure
|
|
1003
|
+
};
|
|
1004
|
+
return { input, digest: sha2562(stableJson(input)) };
|
|
1005
|
+
}
|
|
1006
|
+
function judgeEvalBehaviorV1(scenarioValue, verdict, inputValue) {
|
|
1007
|
+
const scenario = parseEvalBehaviorScenarioV1(scenarioValue);
|
|
1008
|
+
const { input, digest } = canonicalizeEvalBehaviorJudgeInputV1(inputValue, scenario);
|
|
1009
|
+
const labels = EVAL_BEHAVIOR_CATEGORIES.map(
|
|
1010
|
+
(category) => judgeCategory(category, scenario, input, scenario.judge.maxExplanationChars)
|
|
1011
|
+
);
|
|
1012
|
+
const outcome = input.infrastructureFailure ? "infrastructure_failure" : labels.some((label) => label.label === "violated") ? "behavioral_failure" : labels.some((label) => label.label === "unknown") ? "unknown" : labels.every((label) => label.label === "unmeasured") ? "unmeasured" : "pass";
|
|
1013
|
+
return parseEvalBehaviorVerdictV1({
|
|
1014
|
+
schema: EVAL_BEHAVIOR_SCHEMA_V1,
|
|
1015
|
+
verdictRef: {
|
|
1016
|
+
schema: verdict.schema,
|
|
1017
|
+
scenarioId: verdict.scenarioId,
|
|
1018
|
+
trialIndex: verdict.trialIndex
|
|
1019
|
+
},
|
|
1020
|
+
corpus: scenario.corpus,
|
|
1021
|
+
judgeInputDigest: digest,
|
|
1022
|
+
outcome,
|
|
1023
|
+
labels
|
|
1024
|
+
});
|
|
1025
|
+
}
|
|
1026
|
+
function parseEvalBehaviorVerdictV1(value, source = "behavioral verdict") {
|
|
1027
|
+
const record = asRecord2(value, source);
|
|
1028
|
+
if (record.schema !== EVAL_BEHAVIOR_SCHEMA_V1)
|
|
1029
|
+
throw new Error(`${source}.schema: expected ${EVAL_BEHAVIOR_SCHEMA_V1}`);
|
|
1030
|
+
const verdictRef = asRecord2(record.verdictRef, `${source}.verdictRef`);
|
|
1031
|
+
if (verdictRef.schema !== EVAL_VERDICT_SCHEMA_V1) {
|
|
1032
|
+
throw new Error(`${source}.verdictRef.schema: expected ${EVAL_VERDICT_SCHEMA_V1}`);
|
|
1033
|
+
}
|
|
1034
|
+
const corpus = asRecord2(record.corpus, `${source}.corpus`);
|
|
1035
|
+
const outcome = readOutcome2(record.outcome, `${source}.outcome`);
|
|
1036
|
+
if (!Array.isArray(record.labels)) throw new Error(`${source}.labels: expected array`);
|
|
1037
|
+
const labels = record.labels.map((label, index) => parseLabel(label, `${source}.labels[${index}]`));
|
|
1038
|
+
const categories = labels.map((label) => label.category);
|
|
1039
|
+
if (labels.length !== EVAL_BEHAVIOR_CATEGORIES.length || new Set(categories).size !== EVAL_BEHAVIOR_CATEGORIES.length) {
|
|
1040
|
+
throw new Error(`${source}.labels: expected every behavioral category exactly once`);
|
|
1041
|
+
}
|
|
1042
|
+
for (const category of EVAL_BEHAVIOR_CATEGORIES) {
|
|
1043
|
+
if (!categories.includes(category)) throw new Error(`${source}.labels: missing category ${category}`);
|
|
1044
|
+
}
|
|
1045
|
+
const derived = deriveOutcome(labels, outcome === "infrastructure_failure");
|
|
1046
|
+
if (outcome !== derived) throw new Error(`${source}.outcome: ${outcome} conflicts with labels (expected ${derived})`);
|
|
1047
|
+
return {
|
|
1048
|
+
schema: EVAL_BEHAVIOR_SCHEMA_V1,
|
|
1049
|
+
verdictRef: {
|
|
1050
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
1051
|
+
scenarioId: readId(verdictRef.scenarioId, `${source}.verdictRef.scenarioId`),
|
|
1052
|
+
trialIndex: readBoundedInteger(verdictRef.trialIndex, `${source}.verdictRef.trialIndex`, 0, Number.MAX_SAFE_INTEGER)
|
|
1053
|
+
},
|
|
1054
|
+
corpus: {
|
|
1055
|
+
id: readId(corpus.id, `${source}.corpus.id`),
|
|
1056
|
+
version: readText(corpus.version, `${source}.corpus.version`, MAX_ID_CHARS)
|
|
1057
|
+
},
|
|
1058
|
+
judgeInputDigest: readDigest(record.judgeInputDigest, `${source}.judgeInputDigest`),
|
|
1059
|
+
outcome,
|
|
1060
|
+
labels
|
|
1061
|
+
};
|
|
1062
|
+
}
|
|
1063
|
+
function assertEvalBehaviorReferencesVerdictV1(behavior, verdict, source = "behavioral verdict") {
|
|
1064
|
+
if (behavior.verdictRef.scenarioId !== verdict.scenarioId || behavior.verdictRef.trialIndex !== verdict.trialIndex) {
|
|
1065
|
+
throw new Error(`${source}.verdictRef: conflicts with result verdict identity`);
|
|
1066
|
+
}
|
|
1067
|
+
if (verdict.machinery === "infrastructure_failure" && behavior.outcome !== "infrastructure_failure") {
|
|
1068
|
+
throw new Error(`${source}.outcome: machinery failure must remain infrastructure_failure`);
|
|
1069
|
+
}
|
|
1070
|
+
if (behavior.outcome === "pass" && verdict.outcome !== "pass") {
|
|
1071
|
+
throw new Error(`${source}.outcome: behavioral pass cannot override a failed or unmeasured result`);
|
|
1072
|
+
}
|
|
1073
|
+
}
|
|
1074
|
+
function judgeCategory(category, scenario, input, maxExplanationChars) {
|
|
1075
|
+
const expected = scenario.expectedBehavior.filter((rule) => rule.category === category);
|
|
1076
|
+
const forbidden = scenario.forbiddenBehavior.filter((rule) => rule.category === category);
|
|
1077
|
+
const rules = [...expected, ...forbidden];
|
|
1078
|
+
if (input.infrastructureFailure)
|
|
1079
|
+
return labelResult(category, "unknown", rules, [], "infrastructure failure", maxExplanationChars);
|
|
1080
|
+
if (rules.length === 0) return labelResult(category, "unmeasured", [], [], null, maxExplanationChars);
|
|
1081
|
+
const unavailable = rules.some((rule) => input.unavailableSources.includes(rule.fact.source));
|
|
1082
|
+
if (unavailable)
|
|
1083
|
+
return labelResult(category, "unmeasured", rules, [], "required evidence source unavailable", maxExplanationChars);
|
|
1084
|
+
const evaluations = rules.map((rule) => {
|
|
1085
|
+
const fact = input.facts.find(
|
|
1086
|
+
(candidate) => candidate.source === rule.fact.source && candidate.key === rule.fact.key
|
|
1087
|
+
);
|
|
1088
|
+
if (fact === void 0) return { rule, fact: null, holds: null };
|
|
1089
|
+
const matches = compare(fact.value, rule.fact.op, rule.fact.value);
|
|
1090
|
+
return { rule, fact, holds: expected.includes(rule) ? matches : !matches };
|
|
1091
|
+
});
|
|
1092
|
+
const evidence = evaluations.flatMap(({ fact }) => fact === null ? [] : [toEvidence(fact)]).slice(0, MAX_EVIDENCE_PER_LABEL);
|
|
1093
|
+
if (evaluations.some(({ holds }) => holds === false)) {
|
|
1094
|
+
return labelResult(category, "violated", rules, evidence, "one or more declared rules failed", maxExplanationChars);
|
|
1095
|
+
}
|
|
1096
|
+
if (evaluations.some(({ holds }) => holds === null)) {
|
|
1097
|
+
return labelResult(category, "unknown", rules, evidence, "required observable fact missing", maxExplanationChars);
|
|
1098
|
+
}
|
|
1099
|
+
return labelResult(category, "satisfied", rules, evidence, null, maxExplanationChars);
|
|
1100
|
+
}
|
|
1101
|
+
function labelResult(category, label, rules, evidence, explanation, maxExplanationChars) {
|
|
1102
|
+
return {
|
|
1103
|
+
category,
|
|
1104
|
+
label,
|
|
1105
|
+
ruleIds: rules.map((rule) => rule.id).sort(),
|
|
1106
|
+
evidence,
|
|
1107
|
+
explanation: explanation === null ? null : explanation.slice(0, maxExplanationChars)
|
|
1108
|
+
};
|
|
1109
|
+
}
|
|
1110
|
+
function deriveOutcome(labels, infrastructure) {
|
|
1111
|
+
if (infrastructure) return "infrastructure_failure";
|
|
1112
|
+
if (labels.some((label) => label.label === "violated")) return "behavioral_failure";
|
|
1113
|
+
if (labels.some((label) => label.label === "unknown")) return "unknown";
|
|
1114
|
+
if (labels.every((label) => label.label === "unmeasured")) return "unmeasured";
|
|
1115
|
+
return "pass";
|
|
1116
|
+
}
|
|
1117
|
+
function parseRules(value, source) {
|
|
1118
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
1119
|
+
if (value.length > MAX_RULES) throw new Error(`${source}: exceeds ${MAX_RULES} rules`);
|
|
1120
|
+
return value.map((entry, index) => {
|
|
1121
|
+
const record = asRecord2(entry, `${source}[${index}]`);
|
|
1122
|
+
const fact = asRecord2(record.fact, `${source}[${index}].fact`);
|
|
1123
|
+
const rationale = record.rationale === void 0 ? void 0 : readText(record.rationale, `${source}[${index}].rationale`, MAX_TEXT_CHARS);
|
|
1124
|
+
return {
|
|
1125
|
+
id: readId(record.id, `${source}[${index}].id`),
|
|
1126
|
+
category: readCategory(record.category, `${source}[${index}].category`),
|
|
1127
|
+
fact: {
|
|
1128
|
+
source: readSource(fact.source, `${source}[${index}].fact.source`),
|
|
1129
|
+
key: readText(fact.key, `${source}[${index}].fact.key`, MAX_TEXT_CHARS),
|
|
1130
|
+
op: readOp(fact.op, `${source}[${index}].fact.op`),
|
|
1131
|
+
value: readScalar(fact.value, `${source}[${index}].fact.value`)
|
|
1132
|
+
},
|
|
1133
|
+
...rationale === void 0 ? {} : { rationale }
|
|
1134
|
+
};
|
|
1135
|
+
});
|
|
1136
|
+
}
|
|
1137
|
+
function parseFact(value, source) {
|
|
1138
|
+
const record = asRecord2(value, source);
|
|
1139
|
+
const evidence = asRecord2(record.evidence, `${source}.evidence`);
|
|
1140
|
+
return {
|
|
1141
|
+
id: readId(record.id, `${source}.id`),
|
|
1142
|
+
source: readSource(record.source, `${source}.source`),
|
|
1143
|
+
key: readText(record.key, `${source}.key`, MAX_TEXT_CHARS),
|
|
1144
|
+
value: readScalar(record.value, `${source}.value`),
|
|
1145
|
+
evidence: {
|
|
1146
|
+
locator: readText(evidence.locator, `${source}.evidence.locator`, MAX_TEXT_CHARS),
|
|
1147
|
+
digest: readDigest(evidence.digest, `${source}.evidence.digest`),
|
|
1148
|
+
excerpt: evidence.excerpt === null ? null : readText(evidence.excerpt, `${source}.evidence.excerpt`, MAX_TEXT_CHARS)
|
|
1149
|
+
}
|
|
1150
|
+
};
|
|
1151
|
+
}
|
|
1152
|
+
function parseLabel(value, source) {
|
|
1153
|
+
const record = asRecord2(value, source);
|
|
1154
|
+
const label = record.label;
|
|
1155
|
+
if (label !== "satisfied" && label !== "violated" && label !== "unknown" && label !== "unmeasured") {
|
|
1156
|
+
throw new Error(`${source}.label: expected satisfied, violated, unknown, or unmeasured`);
|
|
1157
|
+
}
|
|
1158
|
+
if (!Array.isArray(record.ruleIds) || !Array.isArray(record.evidence)) {
|
|
1159
|
+
throw new Error(`${source}: expected ruleIds and evidence arrays`);
|
|
1160
|
+
}
|
|
1161
|
+
if (record.evidence.length > MAX_EVIDENCE_PER_LABEL) throw new Error(`${source}.evidence: exceeds bounded limit`);
|
|
1162
|
+
if ((label === "satisfied" || label === "violated") && (record.ruleIds.length === 0 || record.evidence.length === 0)) {
|
|
1163
|
+
throw new Error(`${source}: ${label} label requires a rule and observable evidence`);
|
|
1164
|
+
}
|
|
1165
|
+
const explanation = record.explanation;
|
|
1166
|
+
if (explanation !== null && (typeof explanation !== "string" || explanation.length > MAX_EXPLANATION_CHARS)) {
|
|
1167
|
+
throw new Error(`${source}.explanation: expected bounded string or null`);
|
|
1168
|
+
}
|
|
1169
|
+
return {
|
|
1170
|
+
category: readCategory(record.category, `${source}.category`),
|
|
1171
|
+
label,
|
|
1172
|
+
ruleIds: record.ruleIds.map((id, index) => readId(id, `${source}.ruleIds[${index}]`)),
|
|
1173
|
+
evidence: record.evidence.map((entry, index) => {
|
|
1174
|
+
const evidence = asRecord2(entry, `${source}.evidence[${index}]`);
|
|
1175
|
+
return {
|
|
1176
|
+
factId: readId(evidence.factId, `${source}.evidence[${index}].factId`),
|
|
1177
|
+
source: readSource(evidence.source, `${source}.evidence[${index}].source`),
|
|
1178
|
+
locator: readText(evidence.locator, `${source}.evidence[${index}].locator`, MAX_TEXT_CHARS),
|
|
1179
|
+
digest: readDigest(evidence.digest, `${source}.evidence[${index}].digest`),
|
|
1180
|
+
excerpt: evidence.excerpt === null ? null : readText(evidence.excerpt, `${source}.evidence[${index}].excerpt`, MAX_TEXT_CHARS)
|
|
1181
|
+
};
|
|
1182
|
+
}),
|
|
1183
|
+
explanation
|
|
1184
|
+
};
|
|
1185
|
+
}
|
|
1186
|
+
function toEvidence(fact) {
|
|
1187
|
+
return { factId: fact.id, source: fact.source, ...fact.evidence };
|
|
1188
|
+
}
|
|
1189
|
+
function compare(actual, op, expected) {
|
|
1190
|
+
if (op === "eq") return actual === expected;
|
|
1191
|
+
if (op === "neq") return actual !== expected;
|
|
1192
|
+
if (typeof actual !== "number" || typeof expected !== "number") return false;
|
|
1193
|
+
if (op === "lt") return actual < expected;
|
|
1194
|
+
if (op === "lte") return actual <= expected;
|
|
1195
|
+
if (op === "gt") return actual > expected;
|
|
1196
|
+
return actual >= expected;
|
|
1197
|
+
}
|
|
1198
|
+
function parseSources(value, source) {
|
|
1199
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
1200
|
+
return value.map((entry, index) => readSource(entry, `${source}[${index}]`));
|
|
1201
|
+
}
|
|
1202
|
+
function readOutcome2(value, source) {
|
|
1203
|
+
if (value === "pass" || value === "behavioral_failure" || value === "unknown" || value === "unmeasured" || value === "infrastructure_failure")
|
|
1204
|
+
return value;
|
|
1205
|
+
throw new Error(`${source}: expected a closed behavioral outcome`);
|
|
1206
|
+
}
|
|
1207
|
+
function readCategory(value, source) {
|
|
1208
|
+
if (typeof value === "string" && EVAL_BEHAVIOR_CATEGORIES.includes(value)) {
|
|
1209
|
+
return value;
|
|
1210
|
+
}
|
|
1211
|
+
throw new Error(`${source}: expected a closed behavioral category`);
|
|
1212
|
+
}
|
|
1213
|
+
function readSource(value, source) {
|
|
1214
|
+
if (value === "transcript" || value === "tool" || value === "receipt" || value === "grader") return value;
|
|
1215
|
+
throw new Error(`${source}: expected transcript, tool, receipt, or grader`);
|
|
1216
|
+
}
|
|
1217
|
+
function readOp(value, source) {
|
|
1218
|
+
if (value === "eq" || value === "neq" || value === "lt" || value === "lte" || value === "gt" || value === "gte") {
|
|
1219
|
+
return value;
|
|
1220
|
+
}
|
|
1221
|
+
throw new Error(`${source}: expected eq, neq, lt, lte, gt, or gte`);
|
|
1222
|
+
}
|
|
1223
|
+
function readScalar(value, source) {
|
|
1224
|
+
if (typeof value === "boolean") return value;
|
|
1225
|
+
if (typeof value === "number" && Number.isFinite(value)) return value;
|
|
1226
|
+
if (typeof value === "string" && value.length <= MAX_TEXT_CHARS) return value;
|
|
1227
|
+
throw new Error(`${source}: expected bounded string, finite number, or boolean`);
|
|
1228
|
+
}
|
|
1229
|
+
function readId(value, source) {
|
|
1230
|
+
const id = readText(value, source, MAX_ID_CHARS);
|
|
1231
|
+
if (!/^[A-Za-z0-9._-]+$/u.test(id)) throw new Error(`${source}: expected stable id`);
|
|
1232
|
+
return id;
|
|
1233
|
+
}
|
|
1234
|
+
function readText(value, source, maxChars) {
|
|
1235
|
+
if (typeof value !== "string" || value.trim().length === 0 || value.length > maxChars) {
|
|
1236
|
+
throw new Error(`${source}: expected non-empty string no longer than ${maxChars} characters`);
|
|
1237
|
+
}
|
|
1238
|
+
return value;
|
|
1239
|
+
}
|
|
1240
|
+
function readDigest(value, source) {
|
|
1241
|
+
if (typeof value !== "string" || !/^[a-f0-9]{64}$/u.test(value)) throw new Error(`${source}: expected sha256 digest`);
|
|
1242
|
+
return value;
|
|
1243
|
+
}
|
|
1244
|
+
function readBoundedInteger(value, source, min, max) {
|
|
1245
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < min || value > max) {
|
|
1246
|
+
throw new Error(`${source}: expected integer from ${min} through ${max}`);
|
|
1247
|
+
}
|
|
1248
|
+
return value;
|
|
1249
|
+
}
|
|
1250
|
+
function asRecord2(value, source) {
|
|
1251
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value;
|
|
1252
|
+
throw new Error(`${source}: expected object`);
|
|
1253
|
+
}
|
|
1254
|
+
function stableJson(value) {
|
|
1255
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
|
|
1256
|
+
if (typeof value === "object" && value !== null) {
|
|
1257
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`).join(",")}}`;
|
|
1258
|
+
}
|
|
1259
|
+
return JSON.stringify(value);
|
|
1260
|
+
}
|
|
1261
|
+
function sha2562(value) {
|
|
1262
|
+
return createHash3("sha256").update(value, "utf8").digest("hex");
|
|
1263
|
+
}
|
|
1264
|
+
|
|
1265
|
+
// src/domains/eval/schema/behavioral-metrics.ts
|
|
1266
|
+
init_esm_shims();
|
|
1267
|
+
var EVAL_BEHAVIOR_METRICS_SCHEMA_V1 = "clio.eval.behavior.metrics.v1";
|
|
1268
|
+
var EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1 = [
|
|
1269
|
+
{
|
|
1270
|
+
name: "correctness.taskSolved",
|
|
1271
|
+
family: "correctness",
|
|
1272
|
+
direction: "higher",
|
|
1273
|
+
hardGate: true,
|
|
1274
|
+
source: "grader"
|
|
1275
|
+
},
|
|
1276
|
+
{
|
|
1277
|
+
name: "safety.violations",
|
|
1278
|
+
family: "safety",
|
|
1279
|
+
direction: "lower",
|
|
1280
|
+
hardGate: true,
|
|
1281
|
+
source: "behavioral-label"
|
|
1282
|
+
},
|
|
1283
|
+
{
|
|
1284
|
+
name: "behavior.labelViolations",
|
|
1285
|
+
family: "behavior",
|
|
1286
|
+
direction: "lower",
|
|
1287
|
+
hardGate: false,
|
|
1288
|
+
source: "behavioral-label"
|
|
1289
|
+
},
|
|
1290
|
+
{
|
|
1291
|
+
name: "efficiency.toolCalls",
|
|
1292
|
+
family: "efficiency",
|
|
1293
|
+
direction: "lower",
|
|
1294
|
+
hardGate: false,
|
|
1295
|
+
source: "tool-event"
|
|
1296
|
+
},
|
|
1297
|
+
{
|
|
1298
|
+
name: "exploration.unnecessaryReads",
|
|
1299
|
+
family: "exploration",
|
|
1300
|
+
direction: "lower",
|
|
1301
|
+
hardGate: false,
|
|
1302
|
+
source: "tool-event"
|
|
1303
|
+
},
|
|
1304
|
+
{
|
|
1305
|
+
name: "delegation.quality",
|
|
1306
|
+
family: "delegation",
|
|
1307
|
+
direction: "higher",
|
|
1308
|
+
hardGate: false,
|
|
1309
|
+
source: "behavioral-label"
|
|
1310
|
+
},
|
|
1311
|
+
{
|
|
1312
|
+
name: "claims.unsupported",
|
|
1313
|
+
family: "claims",
|
|
1314
|
+
direction: "lower",
|
|
1315
|
+
hardGate: false,
|
|
1316
|
+
source: "grader"
|
|
1317
|
+
},
|
|
1318
|
+
{
|
|
1319
|
+
name: "tokens.total",
|
|
1320
|
+
family: "tokens",
|
|
1321
|
+
direction: "lower",
|
|
1322
|
+
hardGate: false,
|
|
1323
|
+
source: "runner"
|
|
1324
|
+
},
|
|
1325
|
+
{
|
|
1326
|
+
name: "latency.wallMs",
|
|
1327
|
+
family: "latency",
|
|
1328
|
+
direction: "lower",
|
|
1329
|
+
hardGate: false,
|
|
1330
|
+
source: "runner"
|
|
1331
|
+
},
|
|
1332
|
+
{
|
|
1333
|
+
name: "cost.usd",
|
|
1334
|
+
family: "cost",
|
|
1335
|
+
direction: "lower",
|
|
1336
|
+
hardGate: false,
|
|
1337
|
+
source: "receipt"
|
|
1338
|
+
}
|
|
1339
|
+
];
|
|
1340
|
+
function buildEvalBehaviorMetricsV1(result, role) {
|
|
1341
|
+
const label = (category) => result.behavioral?.labels.find((entry) => entry.category === category)?.label ?? null;
|
|
1342
|
+
const observedLabels = result.behavioral?.labels.filter(
|
|
1343
|
+
(entry) => entry.label === "satisfied" || entry.label === "violated"
|
|
1344
|
+
);
|
|
1345
|
+
const values = {
|
|
1346
|
+
"correctness.taskSolved": booleanMetric(result.metrics["task.solved"]),
|
|
1347
|
+
"safety.violations": violationMetric(label("safety_comprehension")),
|
|
1348
|
+
"behavior.labelViolations": observedLabels === void 0 || observedLabels.length !== result.behavioral?.labels.length ? null : observedLabels.filter((entry) => entry.label === "violated").length,
|
|
1349
|
+
"efficiency.toolCalls": numberMetric(result.metrics["tools.totalCalls"]),
|
|
1350
|
+
"exploration.unnecessaryReads": numberMetric(result.metrics["tools.read.outsideAllowed"]),
|
|
1351
|
+
"delegation.quality": qualityMetric(label("delegation")),
|
|
1352
|
+
"claims.unsupported": numberMetric(result.metrics["claims.unsupported"]),
|
|
1353
|
+
"tokens.total": numberMetric(result.metrics["tokens.total"]),
|
|
1354
|
+
"latency.wallMs": numberMetric(result.metrics["latency.wallMs"]),
|
|
1355
|
+
"cost.usd": numberMetric(result.metrics["cost.usd"])
|
|
1356
|
+
};
|
|
1357
|
+
const metrics = Object.fromEntries(
|
|
1358
|
+
EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1.map((definition) => [
|
|
1359
|
+
definition.name,
|
|
1360
|
+
{ value: values[definition.name], source: definition.source }
|
|
1361
|
+
])
|
|
1362
|
+
);
|
|
1363
|
+
return {
|
|
1364
|
+
schema: EVAL_BEHAVIOR_METRICS_SCHEMA_V1,
|
|
1365
|
+
scenarioId: result.taskId,
|
|
1366
|
+
role,
|
|
1367
|
+
target: { id: result.target.id, model: result.target.model },
|
|
1368
|
+
metrics
|
|
1369
|
+
};
|
|
1370
|
+
}
|
|
1371
|
+
function parseEvalBehaviorMetricsV1(value, source = "behavioral metrics") {
|
|
1372
|
+
const record = asRecord3(value, source);
|
|
1373
|
+
if (record.schema !== EVAL_BEHAVIOR_METRICS_SCHEMA_V1) {
|
|
1374
|
+
throw new Error(`${source}.schema: expected ${EVAL_BEHAVIOR_METRICS_SCHEMA_V1}`);
|
|
1375
|
+
}
|
|
1376
|
+
const target = asRecord3(record.target, `${source}.target`);
|
|
1377
|
+
const rawMetrics = asRecord3(record.metrics, `${source}.metrics`);
|
|
1378
|
+
const expectedMetricNames = new Set(EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1.map((definition) => definition.name));
|
|
1379
|
+
for (const metricName of Object.keys(rawMetrics)) {
|
|
1380
|
+
if (!expectedMetricNames.has(metricName)) {
|
|
1381
|
+
throw new Error(`${source}.metrics.${metricName}: unknown behavioral metric`);
|
|
1382
|
+
}
|
|
1383
|
+
}
|
|
1384
|
+
const metrics = Object.fromEntries(
|
|
1385
|
+
EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1.map((definition) => {
|
|
1386
|
+
const observation = asRecord3(rawMetrics[definition.name], `${source}.metrics.${definition.name}`);
|
|
1387
|
+
const metricSource = observation.source;
|
|
1388
|
+
if (metricSource !== definition.source) {
|
|
1389
|
+
throw new Error(`${source}.metrics.${definition.name}.source: expected ${definition.source}`);
|
|
1390
|
+
}
|
|
1391
|
+
const metricValue = observation.value;
|
|
1392
|
+
if (metricValue !== null && (typeof metricValue !== "number" || !Number.isFinite(metricValue))) {
|
|
1393
|
+
throw new Error(`${source}.metrics.${definition.name}.value: expected finite number or null`);
|
|
1394
|
+
}
|
|
1395
|
+
return [definition.name, { value: metricValue, source: metricSource }];
|
|
1396
|
+
})
|
|
1397
|
+
);
|
|
1398
|
+
return {
|
|
1399
|
+
schema: EVAL_BEHAVIOR_METRICS_SCHEMA_V1,
|
|
1400
|
+
scenarioId: readString2(record.scenarioId, `${source}.scenarioId`),
|
|
1401
|
+
role: readString2(record.role, `${source}.role`),
|
|
1402
|
+
target: {
|
|
1403
|
+
id: readString2(target.id, `${source}.target.id`),
|
|
1404
|
+
model: target.model === null ? null : readString2(target.model, `${source}.target.model`)
|
|
1405
|
+
},
|
|
1406
|
+
metrics
|
|
1407
|
+
};
|
|
1408
|
+
}
|
|
1409
|
+
function numberMetric(value) {
|
|
1410
|
+
return typeof value === "number" && Number.isFinite(value) ? value : null;
|
|
1411
|
+
}
|
|
1412
|
+
function booleanMetric(value) {
|
|
1413
|
+
return typeof value === "boolean" ? value ? 1 : 0 : null;
|
|
1414
|
+
}
|
|
1415
|
+
function violationMetric(label) {
|
|
1416
|
+
if (label === "violated") return 1;
|
|
1417
|
+
if (label === "satisfied") return 0;
|
|
1418
|
+
return null;
|
|
1419
|
+
}
|
|
1420
|
+
function qualityMetric(label) {
|
|
1421
|
+
if (label === "satisfied") return 1;
|
|
1422
|
+
if (label === "violated") return 0;
|
|
1423
|
+
return null;
|
|
1424
|
+
}
|
|
1425
|
+
function asRecord3(value, source) {
|
|
1426
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value;
|
|
1427
|
+
throw new Error(`${source}: expected object`);
|
|
1428
|
+
}
|
|
1429
|
+
function readString2(value, source) {
|
|
1430
|
+
if (typeof value === "string" && value.length > 0) return value;
|
|
1431
|
+
throw new Error(`${source}: expected non-empty string`);
|
|
1432
|
+
}
|
|
1433
|
+
|
|
1434
|
+
// src/domains/eval/schema/execution-envelope.ts
|
|
1435
|
+
init_esm_shims();
|
|
1436
|
+
var EVAL_EXECUTION_ENVELOPE_SCHEMA_V1 = "clio.eval.execution-envelope.v1";
|
|
1437
|
+
var EVAL_EXECUTION_MATRIX_DIMENSIONS_V1 = [
|
|
1438
|
+
"prompt",
|
|
1439
|
+
"recipe",
|
|
1440
|
+
"target",
|
|
1441
|
+
"wireModel",
|
|
1442
|
+
"runtime",
|
|
1443
|
+
"thinkingLevel",
|
|
1444
|
+
"toolSignature",
|
|
1445
|
+
"autonomy",
|
|
1446
|
+
"policy",
|
|
1447
|
+
"projectContext",
|
|
1448
|
+
"corpus"
|
|
1449
|
+
];
|
|
1450
|
+
function parseEvalExecutionEnvelopeV1(value, source = "execution envelope") {
|
|
1451
|
+
const record = asRecord4(value, source);
|
|
1452
|
+
if (record.schema !== EVAL_EXECUTION_ENVELOPE_SCHEMA_V1) {
|
|
1453
|
+
throw new Error(`${source}.schema: expected ${EVAL_EXECUTION_ENVELOPE_SCHEMA_V1}`);
|
|
1454
|
+
}
|
|
1455
|
+
const prompt = asRecord4(record.prompt, `${source}.prompt`);
|
|
1456
|
+
if (!Array.isArray(prompt.fragments)) throw new Error(`${source}.prompt.fragments: expected array`);
|
|
1457
|
+
const fragments = prompt.fragments.map((entry, index) => {
|
|
1458
|
+
const fragment = asRecord4(entry, `${source}.prompt.fragments[${index}]`);
|
|
1459
|
+
const version = fragment.version;
|
|
1460
|
+
if (version !== "unversioned" && (!Number.isInteger(version) || typeof version !== "number" || version <= 0)) {
|
|
1461
|
+
throw new Error(`${source}.prompt.fragments[${index}].version: expected positive integer or unversioned`);
|
|
1462
|
+
}
|
|
1463
|
+
return {
|
|
1464
|
+
id: readString3(fragment.id, `${source}.prompt.fragments[${index}].id`),
|
|
1465
|
+
version,
|
|
1466
|
+
contentHash: readDigest2(fragment.contentHash, `${source}.prompt.fragments[${index}].contentHash`)
|
|
1467
|
+
};
|
|
1468
|
+
});
|
|
1469
|
+
const fragmentIds = fragments.map((fragment) => fragment.id);
|
|
1470
|
+
if (new Set(fragmentIds).size !== fragmentIds.length) {
|
|
1471
|
+
throw new Error(`${source}.prompt.fragments: duplicate fragment id`);
|
|
1472
|
+
}
|
|
1473
|
+
const recipe = record.recipe === null ? null : (() => {
|
|
1474
|
+
const value2 = asRecord4(record.recipe, `${source}.recipe`);
|
|
1475
|
+
const version = value2.version;
|
|
1476
|
+
if (!Number.isInteger(version) || typeof version !== "number" || version <= 0) {
|
|
1477
|
+
throw new Error(`${source}.recipe.version: expected positive integer`);
|
|
1478
|
+
}
|
|
1479
|
+
return {
|
|
1480
|
+
id: readString3(value2.id, `${source}.recipe.id`),
|
|
1481
|
+
version,
|
|
1482
|
+
contentHash: readDigest2(value2.contentHash, `${source}.recipe.contentHash`)
|
|
1483
|
+
};
|
|
1484
|
+
})();
|
|
1485
|
+
const policyHashes = asRecord4(record.policyHashes, `${source}.policyHashes`);
|
|
1486
|
+
const projectContext = asRecord4(record.projectContext, `${source}.projectContext`);
|
|
1487
|
+
const projectKind = projectContext.kind;
|
|
1488
|
+
if (projectKind !== "none" && projectKind !== "session" && projectKind !== "worker") {
|
|
1489
|
+
throw new Error(`${source}.projectContext.kind: expected none, session, or worker`);
|
|
1490
|
+
}
|
|
1491
|
+
const corpus = asRecord4(record.corpus, `${source}.corpus`);
|
|
1492
|
+
return {
|
|
1493
|
+
schema: EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
1494
|
+
prompt: {
|
|
1495
|
+
fragments,
|
|
1496
|
+
compositionHash: readNullableDigest(prompt.compositionHash, `${source}.prompt.compositionHash`)
|
|
1497
|
+
},
|
|
1498
|
+
recipe,
|
|
1499
|
+
target: readString3(record.target, `${source}.target`),
|
|
1500
|
+
wireModel: readNullableString3(record.wireModel, `${source}.wireModel`),
|
|
1501
|
+
runtime: readNullableString3(record.runtime, `${source}.runtime`),
|
|
1502
|
+
thinkingLevel: readNullableString3(record.thinkingLevel, `${source}.thinkingLevel`),
|
|
1503
|
+
toolSignature: readNullableDigest(record.toolSignature, `${source}.toolSignature`),
|
|
1504
|
+
autonomy: readNullableString3(record.autonomy, `${source}.autonomy`),
|
|
1505
|
+
policyHashes: {
|
|
1506
|
+
rulePack: readNullableDigest(policyHashes.rulePack, `${source}.policyHashes.rulePack`),
|
|
1507
|
+
project: readNullableDigest(policyHashes.project, `${source}.policyHashes.project`)
|
|
1508
|
+
},
|
|
1509
|
+
projectContext: {
|
|
1510
|
+
kind: projectKind,
|
|
1511
|
+
tier: readNullableString3(projectContext.tier, `${source}.projectContext.tier`),
|
|
1512
|
+
contentHash: readNullableDigest(projectContext.contentHash, `${source}.projectContext.contentHash`),
|
|
1513
|
+
chars: readNullableNonNegativeInteger(projectContext.chars, `${source}.projectContext.chars`),
|
|
1514
|
+
sections: readStringArray2(projectContext.sections, `${source}.projectContext.sections`),
|
|
1515
|
+
rulesApplied: readStringArray2(projectContext.rulesApplied, `${source}.projectContext.rulesApplied`),
|
|
1516
|
+
operatorProfileApplied: readNullableBoolean(
|
|
1517
|
+
projectContext.operatorProfileApplied,
|
|
1518
|
+
`${source}.projectContext.operatorProfileApplied`
|
|
1519
|
+
)
|
|
1520
|
+
},
|
|
1521
|
+
corpus: {
|
|
1522
|
+
id: readString3(corpus.id, `${source}.corpus.id`),
|
|
1523
|
+
version: readString3(corpus.version, `${source}.corpus.version`)
|
|
1524
|
+
}
|
|
1525
|
+
};
|
|
1526
|
+
}
|
|
1527
|
+
function parseEvalExecutionMatrixDimensionsV1(value, source) {
|
|
1528
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
1529
|
+
const allowed = new Set(EVAL_EXECUTION_MATRIX_DIMENSIONS_V1);
|
|
1530
|
+
const dimensions = value.map((entry, index) => {
|
|
1531
|
+
if (typeof entry !== "string" || !allowed.has(entry)) {
|
|
1532
|
+
throw new Error(`${source}[${index}]: expected a declared execution-envelope dimension`);
|
|
1533
|
+
}
|
|
1534
|
+
return entry;
|
|
1535
|
+
});
|
|
1536
|
+
if (new Set(dimensions).size !== dimensions.length) throw new Error(`${source}: duplicate matrix dimension`);
|
|
1537
|
+
return dimensions;
|
|
1538
|
+
}
|
|
1539
|
+
function readString3(value, source) {
|
|
1540
|
+
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}: expected non-empty string`);
|
|
1541
|
+
return value;
|
|
1542
|
+
}
|
|
1543
|
+
function readNullableString3(value, source) {
|
|
1544
|
+
if (value === null) return null;
|
|
1545
|
+
return readString3(value, source);
|
|
1546
|
+
}
|
|
1547
|
+
function readDigest2(value, source) {
|
|
1548
|
+
const digest = readString3(value, source);
|
|
1549
|
+
if (!/^[a-f0-9]{64}$/u.test(digest)) throw new Error(`${source}: expected sha256 digest`);
|
|
1550
|
+
return digest;
|
|
1551
|
+
}
|
|
1552
|
+
function readNullableDigest(value, source) {
|
|
1553
|
+
return value === null ? null : readDigest2(value, source);
|
|
1554
|
+
}
|
|
1555
|
+
function readNullableNonNegativeInteger(value, source) {
|
|
1556
|
+
if (value === null) return null;
|
|
1557
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
|
|
1558
|
+
throw new Error(`${source}: expected non-negative integer or null`);
|
|
1559
|
+
}
|
|
1560
|
+
return value;
|
|
1561
|
+
}
|
|
1562
|
+
function readNullableBoolean(value, source) {
|
|
1563
|
+
if (value === null || typeof value === "boolean") return value;
|
|
1564
|
+
throw new Error(`${source}: expected boolean or null`);
|
|
1565
|
+
}
|
|
1566
|
+
function readStringArray2(value, source) {
|
|
1567
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
1568
|
+
const entries = value.map((entry, index) => readString3(entry, `${source}[${index}]`));
|
|
1569
|
+
if (new Set(entries).size !== entries.length) throw new Error(`${source}: duplicate value`);
|
|
1570
|
+
return entries;
|
|
1571
|
+
}
|
|
1572
|
+
function asRecord4(value, source) {
|
|
1573
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value;
|
|
1574
|
+
throw new Error(`${source}: expected object`);
|
|
1575
|
+
}
|
|
1576
|
+
|
|
1577
|
+
// src/domains/eval/schema/serving.ts
|
|
1578
|
+
init_esm_shims();
|
|
1579
|
+
function parseEvalServingConfigurationV1(value, source) {
|
|
1580
|
+
const record = asRecord5(value, source);
|
|
1581
|
+
const targetId = readString4(record.targetId, `${source}.targetId`);
|
|
1582
|
+
const totalSlots = record.total_slots;
|
|
1583
|
+
if (totalSlots !== null && (typeof totalSlots !== "number" || !Number.isInteger(totalSlots) || totalSlots <= 0)) {
|
|
1584
|
+
throw new Error(`${source}.total_slots: expected positive integer or null`);
|
|
1585
|
+
}
|
|
1586
|
+
const compiledPromptHash = readNullableString4(record.compiledPromptHash, `${source}.compiledPromptHash`);
|
|
1587
|
+
if (compiledPromptHash !== null && !/^[a-f0-9]{64}$/u.test(compiledPromptHash)) {
|
|
1588
|
+
throw new Error(`${source}.compiledPromptHash: expected sha256 digest or null`);
|
|
1589
|
+
}
|
|
1590
|
+
return {
|
|
1591
|
+
targetId,
|
|
1592
|
+
runtimeId: readNullableString4(record.runtimeId, `${source}.runtimeId`),
|
|
1593
|
+
modelId: readNullableString4(record.modelId, `${source}.modelId`),
|
|
1594
|
+
serverBuild: readNullableString4(record.serverBuild, `${source}.serverBuild`),
|
|
1595
|
+
total_slots: totalSlots,
|
|
1596
|
+
thinkingLevel: readNullableString4(record.thinkingLevel, `${source}.thinkingLevel`),
|
|
1597
|
+
compiledPromptHash
|
|
1598
|
+
};
|
|
1599
|
+
}
|
|
1600
|
+
function sameEvalServingConfiguration(left, right) {
|
|
1601
|
+
return left.targetId === right.targetId && left.runtimeId === right.runtimeId && left.modelId === right.modelId && left.serverBuild === right.serverBuild && left.total_slots === right.total_slots && left.thinkingLevel === right.thinkingLevel && left.compiledPromptHash === right.compiledPromptHash;
|
|
1602
|
+
}
|
|
1603
|
+
function renderEvalServingConfiguration(config) {
|
|
1604
|
+
return [
|
|
1605
|
+
`target=${config.targetId}`,
|
|
1606
|
+
`runtime=${config.runtimeId ?? "unknown"}`,
|
|
1607
|
+
`model=${config.modelId ?? "unknown"}`,
|
|
1608
|
+
`server_build=${config.serverBuild ?? "unknown"}`,
|
|
1609
|
+
`total_slots=${config.total_slots ?? "unknown"}`,
|
|
1610
|
+
`thinking=${config.thinkingLevel ?? "unknown"}`,
|
|
1611
|
+
`compiled_prompt_hash=${config.compiledPromptHash ?? "unknown"}`
|
|
1612
|
+
].join(" ");
|
|
1613
|
+
}
|
|
1614
|
+
function readString4(value, source) {
|
|
1615
|
+
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}: expected string`);
|
|
1616
|
+
return value;
|
|
1617
|
+
}
|
|
1618
|
+
function readNullableString4(value, source) {
|
|
1619
|
+
if (value === null) return null;
|
|
1620
|
+
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}: expected string or null`);
|
|
1621
|
+
return value;
|
|
1622
|
+
}
|
|
1623
|
+
function asRecord5(value, source) {
|
|
1624
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value;
|
|
1625
|
+
throw new Error(`${source}: expected object`);
|
|
1626
|
+
}
|
|
1627
|
+
|
|
1628
|
+
// src/domains/eval/artifacts/store.ts
|
|
735
1629
|
function evalArtifactPathV4(dataDir, evalId) {
|
|
736
1630
|
assertSafeId(evalId, "eval");
|
|
737
1631
|
return join2(evalRoot(dataDir), `${evalId}.json`);
|
|
@@ -755,27 +1649,31 @@ async function loadEvalArtifactV4(dataDir, evalId) {
|
|
|
755
1649
|
function parseEvalArtifactV4(value, source) {
|
|
756
1650
|
if (!isRecord2(value)) throw new Error(`${source}: expected object`);
|
|
757
1651
|
if (value.version !== 4) throw new Error(`${source}.version: expected current version 4`);
|
|
758
|
-
const summary =
|
|
759
|
-
const matrix =
|
|
760
|
-
const suite =
|
|
1652
|
+
const summary = asRecord6(value.summary, `${source}.summary`);
|
|
1653
|
+
const matrix = asRecord6(value.matrix, `${source}.matrix`);
|
|
1654
|
+
const suite = asRecord6(value.suite, `${source}.suite`);
|
|
1655
|
+
const servingConfiguration = value.servingConfiguration === void 0 ? void 0 : parseEvalServingConfigurationV1(value.servingConfiguration, `${source}.servingConfiguration`);
|
|
1656
|
+
const aggregates = value.aggregates === void 0 ? void 0 : readArray2(value, source, "aggregates");
|
|
761
1657
|
return {
|
|
762
1658
|
version: 4,
|
|
763
|
-
evalId:
|
|
764
|
-
suite: { id:
|
|
1659
|
+
evalId: readString5(value, source, "evalId"),
|
|
1660
|
+
suite: { id: readString5(suite, `${source}.suite`, "id"), hash: readString5(suite, `${source}.suite`, "hash") },
|
|
765
1661
|
clio: {
|
|
766
|
-
version:
|
|
767
|
-
commit:
|
|
768
|
-
entry:
|
|
1662
|
+
version: readString5(asRecord6(value.clio, `${source}.clio`), `${source}.clio`, "version"),
|
|
1663
|
+
commit: readNullableString5(asRecord6(value.clio, `${source}.clio`), `${source}.clio`, "commit"),
|
|
1664
|
+
entry: readString5(asRecord6(value.clio, `${source}.clio`), `${source}.clio`, "entry")
|
|
769
1665
|
},
|
|
770
1666
|
environment: {
|
|
771
|
-
platform:
|
|
772
|
-
node:
|
|
1667
|
+
platform: readString5(asRecord6(value.environment, `${source}.environment`), `${source}.environment`, "platform"),
|
|
1668
|
+
node: readString5(asRecord6(value.environment, `${source}.environment`), `${source}.environment`, "node")
|
|
773
1669
|
},
|
|
774
1670
|
matrix: {
|
|
775
|
-
target:
|
|
776
|
-
model:
|
|
777
|
-
thinking:
|
|
1671
|
+
target: readString5(matrix, `${source}.matrix`, "target"),
|
|
1672
|
+
model: readNullableString5(matrix, `${source}.matrix`, "model"),
|
|
1673
|
+
thinking: readNullableString5(matrix, `${source}.matrix`, "thinking"),
|
|
1674
|
+
...matrix.dimensions === void 0 ? {} : { dimensions: parseEvalExecutionMatrixDimensionsV1(matrix.dimensions, `${source}.matrix.dimensions`) }
|
|
778
1675
|
},
|
|
1676
|
+
...servingConfiguration === void 0 ? {} : { servingConfiguration },
|
|
779
1677
|
summary: {
|
|
780
1678
|
runs: readNumber2(summary, `${source}.summary`, "runs"),
|
|
781
1679
|
passed: readNumber2(summary, `${source}.summary`, "passed"),
|
|
@@ -784,11 +1682,12 @@ function parseEvalArtifactV4(value, source) {
|
|
|
784
1682
|
tokens: parseTokenAccounting(summary.tokens, `${source}.summary.tokens`),
|
|
785
1683
|
wallTimeMs: readNumber2(summary, `${source}.summary`, "wallTimeMs")
|
|
786
1684
|
},
|
|
1685
|
+
...aggregates === void 0 ? {} : { aggregates },
|
|
787
1686
|
results: readArray2(value, source, "results").map((entry, index) => parseResult(entry, `${source}.results[${index}]`))
|
|
788
1687
|
};
|
|
789
1688
|
}
|
|
790
1689
|
function parseTokenAccounting(value, source) {
|
|
791
|
-
const record =
|
|
1690
|
+
const record = asRecord6(value, source);
|
|
792
1691
|
const measured = readBoolean2(record, source, "measured");
|
|
793
1692
|
const runs = readNumber2(record, source, "runs");
|
|
794
1693
|
const measuredRuns = readNumber2(record, source, "measuredRuns");
|
|
@@ -812,44 +1711,73 @@ function parseTokenAccounting(value, source) {
|
|
|
812
1711
|
};
|
|
813
1712
|
}
|
|
814
1713
|
function parseResult(value, source) {
|
|
815
|
-
const record =
|
|
816
|
-
const target =
|
|
1714
|
+
const record = asRecord6(value, source);
|
|
1715
|
+
const target = asRecord6(record.target, `${source}.target`);
|
|
1716
|
+
const verdict = record.verdict === void 0 ? void 0 : parseEvalVerdictEnvelopeV1(record.verdict, `${source}.verdict`);
|
|
1717
|
+
const behavioral = record.behavioral === void 0 ? void 0 : parseEvalBehaviorVerdictV1(record.behavioral, `${source}.behavioral`);
|
|
1718
|
+
const behavioralMetrics = record.behavioralMetrics === void 0 ? void 0 : parseEvalBehaviorMetricsV1(record.behavioralMetrics, `${source}.behavioralMetrics`);
|
|
1719
|
+
const executionEnvelope = record.executionEnvelope === void 0 ? void 0 : parseEvalExecutionEnvelopeV1(record.executionEnvelope, `${source}.executionEnvelope`);
|
|
1720
|
+
if (behavioral !== void 0 && verdict === void 0) {
|
|
1721
|
+
throw new Error(`${source}.behavioral: sibling document requires a verdict`);
|
|
1722
|
+
}
|
|
1723
|
+
if (behavioral !== void 0 && verdict !== void 0) {
|
|
1724
|
+
assertEvalBehaviorReferencesVerdictV1(behavioral, verdict, `${source}.behavioral`);
|
|
1725
|
+
}
|
|
1726
|
+
if (behavioralMetrics !== void 0) {
|
|
1727
|
+
if (behavioral === void 0) throw new Error(`${source}.behavioralMetrics: requires a behavioral verdict`);
|
|
1728
|
+
if (behavioralMetrics.scenarioId !== readString5(record, source, "taskId")) {
|
|
1729
|
+
throw new Error(`${source}.behavioralMetrics.scenarioId: conflicts with result taskId`);
|
|
1730
|
+
}
|
|
1731
|
+
if (behavioralMetrics.target.id !== readString5(target, `${source}.target`, "id") || behavioralMetrics.target.model !== readNullableString5(target, `${source}.target`, "model")) {
|
|
1732
|
+
throw new Error(`${source}.behavioralMetrics.target: conflicts with result target`);
|
|
1733
|
+
}
|
|
1734
|
+
}
|
|
1735
|
+
if (executionEnvelope !== void 0) {
|
|
1736
|
+
if (behavioral === void 0) throw new Error(`${source}.executionEnvelope: requires a behavioral verdict`);
|
|
1737
|
+
if (executionEnvelope.target !== readString5(target, `${source}.target`, "id") || executionEnvelope.corpus.id !== behavioral.corpus.id || executionEnvelope.corpus.version !== behavioral.corpus.version) {
|
|
1738
|
+
throw new Error(`${source}.executionEnvelope: conflicts with result target or behavioral corpus`);
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
817
1741
|
return {
|
|
818
|
-
assignmentId:
|
|
819
|
-
terminalReceiptDigest:
|
|
820
|
-
taskId:
|
|
1742
|
+
assignmentId: readNullableString5(record, source, "assignmentId"),
|
|
1743
|
+
terminalReceiptDigest: readNullableDigest2(record, source, "terminalReceiptDigest"),
|
|
1744
|
+
taskId: readString5(record, source, "taskId"),
|
|
821
1745
|
repeatIndex: readNumber2(record, source, "repeatIndex"),
|
|
822
1746
|
target: {
|
|
823
|
-
id:
|
|
824
|
-
model:
|
|
825
|
-
thinking:
|
|
1747
|
+
id: readString5(target, `${source}.target`, "id"),
|
|
1748
|
+
model: readNullableString5(target, `${source}.target`, "model"),
|
|
1749
|
+
thinking: readNullableString5(target, `${source}.target`, "thinking")
|
|
826
1750
|
},
|
|
827
1751
|
pass: readBoolean2(record, source, "pass"),
|
|
828
|
-
failureClass:
|
|
829
|
-
metrics:
|
|
830
|
-
artifacts:
|
|
1752
|
+
failureClass: readNullableString5(record, source, "failureClass"),
|
|
1753
|
+
metrics: asRecord6(record.metrics, `${source}.metrics`),
|
|
1754
|
+
artifacts: asRecord6(record.artifacts, `${source}.artifacts`),
|
|
1755
|
+
...verdict === void 0 ? {} : { verdict },
|
|
1756
|
+
...behavioral === void 0 ? {} : { behavioral },
|
|
1757
|
+
...behavioralMetrics === void 0 ? {} : { behavioralMetrics },
|
|
1758
|
+
...executionEnvelope === void 0 ? {} : { executionEnvelope }
|
|
831
1759
|
};
|
|
832
1760
|
}
|
|
833
|
-
function
|
|
1761
|
+
function asRecord6(value, source) {
|
|
834
1762
|
if (isRecord2(value)) return value;
|
|
835
1763
|
throw new Error(`${source}: expected object`);
|
|
836
1764
|
}
|
|
837
1765
|
function isRecord2(value) {
|
|
838
1766
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
839
1767
|
}
|
|
840
|
-
function
|
|
1768
|
+
function readString5(record, source, field) {
|
|
841
1769
|
const value = record[field];
|
|
842
1770
|
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}.${field}: expected string`);
|
|
843
1771
|
return value;
|
|
844
1772
|
}
|
|
845
|
-
function
|
|
1773
|
+
function readNullableString5(record, source, field) {
|
|
846
1774
|
const value = record[field];
|
|
847
1775
|
if (value === null) return null;
|
|
848
1776
|
if (typeof value !== "string" || value.length === 0) throw new Error(`${source}.${field}: expected string or null`);
|
|
849
1777
|
return value;
|
|
850
1778
|
}
|
|
851
|
-
function
|
|
852
|
-
const value =
|
|
1779
|
+
function readNullableDigest2(record, source, field) {
|
|
1780
|
+
const value = readNullableString5(record, source, field);
|
|
853
1781
|
if (value !== null && !/^[0-9a-f]{64}$/u.test(value))
|
|
854
1782
|
throw new Error(`${source}.${field}: expected sha256 digest or null`);
|
|
855
1783
|
return value;
|
|
@@ -1416,20 +2344,6 @@ function adaptGroundedEvidenceValidationStatus(input) {
|
|
|
1416
2344
|
uniqueBoundedReferences([{ kind: "evidence_bundle", id: input.evidenceId }, ...observed])
|
|
1417
2345
|
);
|
|
1418
2346
|
}
|
|
1419
|
-
function adaptFinishContractCompletionStatus(assessment, options = {}) {
|
|
1420
|
-
const artifacts = uniqueBoundedReferences(
|
|
1421
|
-
options.artifacts ?? assessment.evidence.map((evidence, index) => ({
|
|
1422
|
-
kind: evidence.turnId === void 0 ? "finish_contract_evidence" : "session_entry",
|
|
1423
|
-
id: evidence.turnId ?? `${evidence.kind}:${index + 1}`
|
|
1424
|
-
}))
|
|
1425
|
-
);
|
|
1426
|
-
const source = { kind: "finish_contract", id: options.sourceId ?? assessment.reason };
|
|
1427
|
-
const authority = { kind: "clio", id: "finish-contract" };
|
|
1428
|
-
if (assessment.reason === "no_mutation") return attributed("not_applicable", source, authority, artifacts);
|
|
1429
|
-
if (assessment.reason === "validation_evidence") return attributed("evidenced", source, authority, artifacts);
|
|
1430
|
-
if (assessment.reason === "explicit_limitation") return attributed("limited", source, authority, artifacts);
|
|
1431
|
-
return attributed("incomplete", source, authority, artifacts);
|
|
1432
|
-
}
|
|
1433
2347
|
function uniqueBoundedReferences(references) {
|
|
1434
2348
|
const sorted = [...references].sort(compareArtifactReferences);
|
|
1435
2349
|
const unique = sorted.filter(
|
|
@@ -1438,24 +2352,196 @@ function uniqueBoundedReferences(references) {
|
|
|
1438
2352
|
return unique.slice(0, TRUST_STATUS_MAX_ARTIFACT_REFERENCES);
|
|
1439
2353
|
}
|
|
1440
2354
|
|
|
1441
|
-
// src/domains/
|
|
2355
|
+
// src/domains/evidence/trust-projection.ts
|
|
1442
2356
|
init_esm_shims();
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
2357
|
+
var TRUST_SUMMARY_VERSION = 1;
|
|
2358
|
+
var TRUST_SUMMARY_MAX_REFS = 8;
|
|
2359
|
+
var TRUST_STATE_WORDS = {
|
|
2360
|
+
artifactIntegrity: {
|
|
2361
|
+
verified: "sealed",
|
|
2362
|
+
failed: "seal broken",
|
|
2363
|
+
absent: "no receipt",
|
|
2364
|
+
unknown: "seal unchecked",
|
|
2365
|
+
not_applicable: "seal not applicable"
|
|
2366
|
+
},
|
|
2367
|
+
validationGrounding: {
|
|
2368
|
+
validated: "grounded",
|
|
2369
|
+
failed: "validation failed",
|
|
2370
|
+
ungrounded: "inferred: validation claimed, none observed",
|
|
2371
|
+
absent: "no validation observed",
|
|
2372
|
+
unknown: "validation unknown",
|
|
2373
|
+
not_applicable: "validation not applicable"
|
|
2374
|
+
},
|
|
2375
|
+
independentReview: {
|
|
2376
|
+
passed: "independently reviewed: pass",
|
|
2377
|
+
failed: "independently reviewed: fail",
|
|
2378
|
+
inconclusive: "independent review inconclusive",
|
|
2379
|
+
not_independent: "review not independent",
|
|
2380
|
+
absent: "not independently reviewed",
|
|
2381
|
+
unknown: "independent review unknown",
|
|
2382
|
+
not_applicable: "independent review not applicable"
|
|
2383
|
+
},
|
|
2384
|
+
contextProvenance: {
|
|
2385
|
+
recorded: "context recorded",
|
|
2386
|
+
invalid: "context record invalid",
|
|
2387
|
+
absent: "context not recorded",
|
|
2388
|
+
unknown: "context unknown",
|
|
2389
|
+
not_applicable: "context not applicable"
|
|
2390
|
+
},
|
|
2391
|
+
autonomyEnforcement: {
|
|
2392
|
+
enforced: "mediated",
|
|
2393
|
+
approximated: "approximated",
|
|
2394
|
+
bypassed: "bypassed",
|
|
2395
|
+
absent: "autonomy not recorded",
|
|
2396
|
+
unknown: "autonomy unknown",
|
|
2397
|
+
not_applicable: "autonomy not applicable"
|
|
2398
|
+
},
|
|
2399
|
+
completionEvidence: {
|
|
2400
|
+
evidenced: "completion evidenced",
|
|
2401
|
+
incomplete: "completion unevidenced",
|
|
2402
|
+
limited: "completion limited",
|
|
2403
|
+
absent: "completion not recorded",
|
|
2404
|
+
unknown: "completion unknown",
|
|
2405
|
+
not_applicable: "completion not applicable"
|
|
2406
|
+
}
|
|
2407
|
+
};
|
|
2408
|
+
function trustStateWord(axis, state) {
|
|
2409
|
+
const words = TRUST_STATE_WORDS[axis];
|
|
2410
|
+
return words[state] ?? words.unknown ?? "unknown";
|
|
2411
|
+
}
|
|
2412
|
+
function authorityOf(status, axis) {
|
|
2413
|
+
const entry = status[axis];
|
|
2414
|
+
return entry.state === "absent" ? null : `${entry.authority.kind}:${entry.authority.id}`;
|
|
2415
|
+
}
|
|
2416
|
+
function integrityClause(status) {
|
|
2417
|
+
const retired = retiredIntegrityVersionOf(status.artifactIntegrity);
|
|
2418
|
+
if (retired !== null) return `seal v${retired} retired (this build verifies v${RUN_RECEIPT_INTEGRITY_VERSION})`;
|
|
2419
|
+
return trustStateWord("artifactIntegrity", status.artifactIntegrity.state);
|
|
2420
|
+
}
|
|
2421
|
+
function validationClause(status) {
|
|
2422
|
+
const entry = status.validationGrounding;
|
|
2423
|
+
const word = trustStateWord("validationGrounding", entry.state);
|
|
2424
|
+
if (entry.state === "absent") return word;
|
|
2425
|
+
if (entry.state === "validated" || entry.state === "failed") return `${word} by ${entry.authority.id}`;
|
|
2426
|
+
if (entry.state === "unknown" && entry.authority.kind !== "unknown") return `${word} (${entry.authority.id})`;
|
|
2427
|
+
return word;
|
|
2428
|
+
}
|
|
2429
|
+
function autonomyClause(status) {
|
|
2430
|
+
const entry = status.autonomyEnforcement;
|
|
2431
|
+
const word = trustStateWord("autonomyEnforcement", entry.state);
|
|
2432
|
+
if (entry.state !== "approximated" && entry.state !== "bypassed") return word;
|
|
2433
|
+
return `${word} (${entry.authority.id})`;
|
|
2434
|
+
}
|
|
2435
|
+
function formatTrustSummary(status) {
|
|
2436
|
+
return [
|
|
2437
|
+
integrityClause(status),
|
|
2438
|
+
validationClause(status),
|
|
2439
|
+
trustStateWord("independentReview", status.independentReview.state),
|
|
2440
|
+
autonomyClause(status),
|
|
2441
|
+
trustStateWord("contextProvenance", status.contextProvenance.state),
|
|
2442
|
+
trustStateWord("completionEvidence", status.completionEvidence.state)
|
|
2443
|
+
].join("; ");
|
|
2444
|
+
}
|
|
2445
|
+
function formatTrustSummaryLine(status) {
|
|
2446
|
+
return `trust v${TRUST_SUMMARY_VERSION}: ${trustVerdict(status)}; ${formatTrustSummary(status)}`;
|
|
2447
|
+
}
|
|
2448
|
+
function formatTrustAxes(status) {
|
|
2449
|
+
return [`trust_status=v${status.version}`, ...TRUST_STATUS_AXES.map((axis) => `${axis}:${status[axis].state}`)].join(
|
|
2450
|
+
" "
|
|
2451
|
+
);
|
|
2452
|
+
}
|
|
2453
|
+
function isUnanswered(status, axis) {
|
|
2454
|
+
const state = status[axis].state;
|
|
2455
|
+
return state === "absent" || state === "unknown";
|
|
2456
|
+
}
|
|
2457
|
+
function trustVerdict(status) {
|
|
2458
|
+
const integrity = status.artifactIntegrity.state;
|
|
2459
|
+
const validation = status.validationGrounding.state;
|
|
2460
|
+
const review = status.independentReview.state;
|
|
2461
|
+
const autonomy = status.autonomyEnforcement.state;
|
|
2462
|
+
if (integrity === "failed" || autonomy === "bypassed" || validation === "failed" || validation === "ungrounded" || review === "failed" || review === "not_independent" || status.contextProvenance.state === "invalid") {
|
|
2463
|
+
return "compromised";
|
|
2464
|
+
}
|
|
2465
|
+
if (integrity !== "verified") return "unknown";
|
|
2466
|
+
if (review === "passed") return "reviewed";
|
|
2467
|
+
if (validation === "validated") return "grounded";
|
|
2468
|
+
return "unverified";
|
|
2469
|
+
}
|
|
2470
|
+
function claimantOf(status) {
|
|
2471
|
+
const state = status.validationGrounding.state;
|
|
2472
|
+
if (state === "absent" || state === "ungrounded") return "worker";
|
|
2473
|
+
return authorityOf(status, "validationGrounding") ?? "worker";
|
|
2474
|
+
}
|
|
2475
|
+
function referenceKey(reference) {
|
|
2476
|
+
return `${reference.kind}:${reference.id}`;
|
|
2477
|
+
}
|
|
2478
|
+
function trustSummaryReferences(status) {
|
|
2479
|
+
const keys = /* @__PURE__ */ new Set();
|
|
2480
|
+
for (const axis of TRUST_STATUS_AXES) {
|
|
2481
|
+
const entry = status[axis];
|
|
2482
|
+
if (entry.state === "absent") continue;
|
|
2483
|
+
for (const reference of entry.artifacts) keys.add(referenceKey(reference));
|
|
2484
|
+
}
|
|
2485
|
+
return [...keys].sort().slice(0, TRUST_SUMMARY_MAX_REFS);
|
|
2486
|
+
}
|
|
2487
|
+
function summarizeTrustStatus(status) {
|
|
2488
|
+
const axes = Object.fromEntries(TRUST_STATUS_AXES.map((axis) => [axis, status[axis].state]));
|
|
2489
|
+
return {
|
|
2490
|
+
version: TRUST_SUMMARY_VERSION,
|
|
2491
|
+
verdict: trustVerdict(status),
|
|
2492
|
+
text: formatTrustSummary(status),
|
|
2493
|
+
axes,
|
|
2494
|
+
claimant: claimantOf(status),
|
|
2495
|
+
unknown: TRUST_STATUS_AXES.filter((axis) => isUnanswered(status, axis)),
|
|
2496
|
+
refs: trustSummaryReferences(status)
|
|
2497
|
+
};
|
|
2498
|
+
}
|
|
2499
|
+
|
|
2500
|
+
// src/domains/eval/provenance.ts
|
|
2501
|
+
init_esm_shims();
|
|
2502
|
+
import { spawnSync } from "node:child_process";
|
|
2503
|
+
function evalClioProvenance(options = {}) {
|
|
2504
|
+
return {
|
|
2505
|
+
version: readClioVersion(),
|
|
2506
|
+
commit: options.commit === void 0 ? currentClioCommit() : options.commit,
|
|
2507
|
+
entry: options.entry ?? process.argv[1] ?? "unknown"
|
|
2508
|
+
};
|
|
2509
|
+
}
|
|
2510
|
+
function evalEnvironmentProvenance() {
|
|
2511
|
+
return {
|
|
2512
|
+
platform: `${process.platform}-${process.arch}`,
|
|
2513
|
+
node: process.version
|
|
2514
|
+
};
|
|
2515
|
+
}
|
|
2516
|
+
async function evalServingConfiguration(targets, observations) {
|
|
2517
|
+
const targetId = targets.length === 1 ? targets[0]?.id ?? "unknown" : "multiple";
|
|
2518
|
+
const configured = configuredTarget(targetId);
|
|
2519
|
+
const target = targets.length === 1 ? targets[0] : void 0;
|
|
2520
|
+
const runtimeId = consensus(observations.map((entry) => entry.runtimeId)) ?? configured?.runtime ?? null;
|
|
2521
|
+
const modelId = consensus(observations.map((entry) => entry.modelId)) ?? target?.model ?? configured?.defaultModel ?? null;
|
|
2522
|
+
const props = configured?.url === void 0 ? null : await readServingProps(configured.url, modelId);
|
|
2523
|
+
return {
|
|
2524
|
+
targetId,
|
|
2525
|
+
runtimeId,
|
|
2526
|
+
modelId,
|
|
2527
|
+
serverBuild: props?.serverBuild ?? null,
|
|
2528
|
+
total_slots: props?.totalSlots ?? null,
|
|
2529
|
+
thinkingLevel: consensus(observations.map((entry) => entry.thinkingLevel)) ?? target?.thinking ?? null,
|
|
2530
|
+
compiledPromptHash: consensus(observations.map((entry) => entry.compiledPromptHash))
|
|
2531
|
+
};
|
|
2532
|
+
}
|
|
2533
|
+
function evalServingObservationFrom(target, receipt, compiledPromptHashes) {
|
|
2534
|
+
const compiledPromptHash = receipt?.staticCompositionHash ?? receipt?.compiledPromptHash ?? consensus(compiledPromptHashes);
|
|
2535
|
+
return {
|
|
2536
|
+
targetId: receipt?.targetId ?? target.id,
|
|
2537
|
+
runtimeId: receipt?.runtimeId ?? null,
|
|
2538
|
+
modelId: receipt?.wireModelId ?? target.model ?? null,
|
|
2539
|
+
thinkingLevel: receipt?.runtimeResolution?.effectiveThinkingLevel ?? target.thinking ?? null,
|
|
2540
|
+
compiledPromptHash
|
|
2541
|
+
};
|
|
2542
|
+
}
|
|
2543
|
+
function currentClioCommit() {
|
|
2544
|
+
const result = spawnSync("git", ["rev-parse", "HEAD"], {
|
|
1459
2545
|
cwd: resolvePackageRoot(),
|
|
1460
2546
|
encoding: "utf8",
|
|
1461
2547
|
stdio: ["ignore", "pipe", "ignore"],
|
|
@@ -1465,11 +2551,62 @@ function currentClioCommit() {
|
|
|
1465
2551
|
const value = result.stdout.trim();
|
|
1466
2552
|
return value.length > 0 ? value : null;
|
|
1467
2553
|
}
|
|
2554
|
+
function configuredTarget(targetId) {
|
|
2555
|
+
try {
|
|
2556
|
+
const target = readSettings().targets.find((entry) => entry.id === targetId);
|
|
2557
|
+
if (target === void 0) return null;
|
|
2558
|
+
return {
|
|
2559
|
+
runtime: target.runtime,
|
|
2560
|
+
...target.url === void 0 ? {} : { url: target.url },
|
|
2561
|
+
...target.defaultModel === void 0 ? {} : { defaultModel: target.defaultModel }
|
|
2562
|
+
};
|
|
2563
|
+
} catch {
|
|
2564
|
+
return null;
|
|
2565
|
+
}
|
|
2566
|
+
}
|
|
2567
|
+
async function readServingProps(targetUrl, modelId) {
|
|
2568
|
+
const root = targetUrl.replace(/\/+$/u, "").replace(/\/v1$/u, "");
|
|
2569
|
+
try {
|
|
2570
|
+
const response = await fetch(`${root}/props`, { signal: AbortSignal.timeout(5e3) });
|
|
2571
|
+
if (!response.ok) return null;
|
|
2572
|
+
const value = await response.json();
|
|
2573
|
+
if (!isRecord4(value)) return null;
|
|
2574
|
+
const totalSlots = positiveInteger(value.total_slots) ?? await readServingSlots(root, modelId);
|
|
2575
|
+
return {
|
|
2576
|
+
serverBuild: typeof value.build_info === "string" && value.build_info.length > 0 ? value.build_info : null,
|
|
2577
|
+
totalSlots
|
|
2578
|
+
};
|
|
2579
|
+
} catch {
|
|
2580
|
+
return null;
|
|
2581
|
+
}
|
|
2582
|
+
}
|
|
2583
|
+
async function readServingSlots(root, modelId) {
|
|
2584
|
+
if (modelId === null) return null;
|
|
2585
|
+
try {
|
|
2586
|
+
const query = new URLSearchParams({ model: modelId });
|
|
2587
|
+
const response = await fetch(`${root}/slots?${query}`, { signal: AbortSignal.timeout(5e3) });
|
|
2588
|
+
if (!response.ok) return null;
|
|
2589
|
+
const value = await response.json();
|
|
2590
|
+
return Array.isArray(value) && value.length > 0 ? value.length : null;
|
|
2591
|
+
} catch {
|
|
2592
|
+
return null;
|
|
2593
|
+
}
|
|
2594
|
+
}
|
|
2595
|
+
function consensus(values) {
|
|
2596
|
+
const known = [...new Set(values.filter((value) => typeof value === "string" && value.length > 0))];
|
|
2597
|
+
return known.length === 1 ? known[0] ?? null : null;
|
|
2598
|
+
}
|
|
2599
|
+
function positiveInteger(value) {
|
|
2600
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : null;
|
|
2601
|
+
}
|
|
2602
|
+
function isRecord4(value) {
|
|
2603
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
2604
|
+
}
|
|
1468
2605
|
|
|
1469
2606
|
// src/domains/eval/task-file.ts
|
|
1470
2607
|
init_esm_shims();
|
|
1471
2608
|
var import_yaml = __toESM(require_dist(), 1);
|
|
1472
|
-
import { createHash as
|
|
2609
|
+
import { createHash as createHash4 } from "node:crypto";
|
|
1473
2610
|
import { readFile as readFile3 } from "node:fs/promises";
|
|
1474
2611
|
import { dirname, isAbsolute, relative, resolve } from "node:path";
|
|
1475
2612
|
|
|
@@ -1508,7 +2645,7 @@ function parseEvalTaskFileYaml(raw) {
|
|
|
1508
2645
|
}
|
|
1509
2646
|
function validateEvalTaskFile(value) {
|
|
1510
2647
|
const issues = [];
|
|
1511
|
-
if (!
|
|
2648
|
+
if (!isRecord5(value)) {
|
|
1512
2649
|
return { valid: false, issues: [{ path: "$", message: "expected object" }] };
|
|
1513
2650
|
}
|
|
1514
2651
|
if (value.version !== EVAL_TASK_FILE_VERSION) {
|
|
@@ -1544,29 +2681,29 @@ var EvalTaskFileError = class extends Error {
|
|
|
1544
2681
|
}
|
|
1545
2682
|
};
|
|
1546
2683
|
function parseTask(value, path, issues) {
|
|
1547
|
-
if (!
|
|
2684
|
+
if (!isRecord5(value)) {
|
|
1548
2685
|
issues.push({ path, message: "expected object" });
|
|
1549
2686
|
return null;
|
|
1550
2687
|
}
|
|
1551
2688
|
for (const key of Object.keys(value).sort((a, b) => a.localeCompare(b))) {
|
|
1552
2689
|
if (!TASK_KEYS.has(key)) issues.push({ path: `${path}.${key}`, message: "unknown field" });
|
|
1553
2690
|
}
|
|
1554
|
-
const id =
|
|
2691
|
+
const id = readNonEmptyString2(value, path, "id", issues);
|
|
1555
2692
|
if (id !== null && !/^[A-Za-z0-9._-]+$/.test(id)) {
|
|
1556
2693
|
issues.push({ path: `${path}.id`, message: "expected id with letters, numbers, dots, underscores, or hyphens" });
|
|
1557
2694
|
}
|
|
1558
|
-
const prompt =
|
|
1559
|
-
const cwd =
|
|
2695
|
+
const prompt = readNonEmptyString2(value, path, "prompt", issues);
|
|
2696
|
+
const cwd = readNonEmptyString2(value, path, "cwd", issues);
|
|
1560
2697
|
if (cwd !== null && isAbsolute(cwd)) {
|
|
1561
2698
|
issues.push({ path: `${path}.cwd`, message: "expected repo-local relative path" });
|
|
1562
2699
|
}
|
|
1563
|
-
const setup =
|
|
1564
|
-
const verifier =
|
|
2700
|
+
const setup = readStringArray3(value, path, "setup", issues, true);
|
|
2701
|
+
const verifier = readStringArray3(value, path, "verifier", issues, false);
|
|
1565
2702
|
if (verifier !== null && verifier.length === 0) {
|
|
1566
2703
|
issues.push({ path: `${path}.verifier`, message: "expected at least one command" });
|
|
1567
2704
|
}
|
|
1568
2705
|
const timeoutMs = readPositiveInteger(value, path, "timeoutMs", issues);
|
|
1569
|
-
const tags =
|
|
2706
|
+
const tags = readStringArray3(value, path, "tags", issues, true);
|
|
1570
2707
|
if (id === null || prompt === null || cwd === null || setup === null || verifier === null || timeoutMs === null) {
|
|
1571
2708
|
return null;
|
|
1572
2709
|
}
|
|
@@ -1584,7 +2721,7 @@ function validateTaskCwds(tasks, baseDir) {
|
|
|
1584
2721
|
}
|
|
1585
2722
|
return issues;
|
|
1586
2723
|
}
|
|
1587
|
-
function
|
|
2724
|
+
function readNonEmptyString2(record, path, field, issues) {
|
|
1588
2725
|
const value = record[field];
|
|
1589
2726
|
if (typeof value !== "string" || value.trim().length === 0) {
|
|
1590
2727
|
issues.push({ path: `${path}.${field}`, message: "expected non-empty string" });
|
|
@@ -1592,7 +2729,7 @@ function readNonEmptyString(record, path, field, issues) {
|
|
|
1592
2729
|
}
|
|
1593
2730
|
return value;
|
|
1594
2731
|
}
|
|
1595
|
-
function
|
|
2732
|
+
function readStringArray3(record, path, field, issues, allowMissing) {
|
|
1596
2733
|
const value = record[field];
|
|
1597
2734
|
if (value === void 0 && allowMissing) return [];
|
|
1598
2735
|
if (!Array.isArray(value)) {
|
|
@@ -1618,166 +2755,105 @@ function readPositiveInteger(record, path, field, issues) {
|
|
|
1618
2755
|
}
|
|
1619
2756
|
return value;
|
|
1620
2757
|
}
|
|
1621
|
-
function
|
|
2758
|
+
function isRecord5(value) {
|
|
1622
2759
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1623
2760
|
}
|
|
1624
2761
|
function sha256Hex(content) {
|
|
1625
|
-
return
|
|
2762
|
+
return createHash4("sha256").update(content, "utf8").digest("hex");
|
|
1626
2763
|
}
|
|
1627
2764
|
|
|
1628
|
-
// src/domains/
|
|
2765
|
+
// src/domains/eval/run-compare.ts
|
|
1629
2766
|
init_esm_shims();
|
|
1630
|
-
var
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
absent: "no receipt",
|
|
1637
|
-
unknown: "seal unchecked",
|
|
1638
|
-
not_applicable: "seal not applicable"
|
|
1639
|
-
},
|
|
1640
|
-
validationGrounding: {
|
|
1641
|
-
validated: "grounded",
|
|
1642
|
-
failed: "validation failed",
|
|
1643
|
-
ungrounded: "inferred: validation claimed, none observed",
|
|
1644
|
-
absent: "no validation observed",
|
|
1645
|
-
unknown: "validation unknown",
|
|
1646
|
-
not_applicable: "validation not applicable"
|
|
1647
|
-
},
|
|
1648
|
-
independentReview: {
|
|
1649
|
-
passed: "independently reviewed: pass",
|
|
1650
|
-
failed: "independently reviewed: fail",
|
|
1651
|
-
inconclusive: "independent review inconclusive",
|
|
1652
|
-
not_independent: "review not independent",
|
|
1653
|
-
absent: "not independently reviewed",
|
|
1654
|
-
unknown: "independent review unknown",
|
|
1655
|
-
not_applicable: "independent review not applicable"
|
|
1656
|
-
},
|
|
1657
|
-
contextProvenance: {
|
|
1658
|
-
recorded: "context recorded",
|
|
1659
|
-
invalid: "context record invalid",
|
|
1660
|
-
absent: "context not recorded",
|
|
1661
|
-
unknown: "context unknown",
|
|
1662
|
-
not_applicable: "context not applicable"
|
|
1663
|
-
},
|
|
1664
|
-
autonomyEnforcement: {
|
|
1665
|
-
enforced: "mediated",
|
|
1666
|
-
approximated: "approximated",
|
|
1667
|
-
bypassed: "bypassed",
|
|
1668
|
-
absent: "autonomy not recorded",
|
|
1669
|
-
unknown: "autonomy unknown",
|
|
1670
|
-
not_applicable: "autonomy not applicable"
|
|
1671
|
-
},
|
|
1672
|
-
completionEvidence: {
|
|
1673
|
-
evidenced: "completion evidenced",
|
|
1674
|
-
incomplete: "completion unevidenced",
|
|
1675
|
-
limited: "completion limited",
|
|
1676
|
-
absent: "completion not recorded",
|
|
1677
|
-
unknown: "completion unknown",
|
|
1678
|
-
not_applicable: "completion not applicable"
|
|
2767
|
+
var EvalTrackedMetricSourceMismatchError = class extends Error {
|
|
2768
|
+
constructor(metric, baseline, candidate) {
|
|
2769
|
+
super(
|
|
2770
|
+
`tracked metric ${metric} cannot compare estimated and measured values (baseline=${baseline.join(",") || "none"}, candidate=${candidate.join(",") || "none"})`
|
|
2771
|
+
);
|
|
2772
|
+
this.name = "EvalTrackedMetricSourceMismatchError";
|
|
1679
2773
|
}
|
|
1680
2774
|
};
|
|
1681
|
-
function
|
|
1682
|
-
|
|
1683
|
-
|
|
2775
|
+
function assertComparableTrackedMetricSources(metric, baseline, candidate) {
|
|
2776
|
+
if (baseline.includes("estimated") === candidate.includes("estimated")) return;
|
|
2777
|
+
throw new EvalTrackedMetricSourceMismatchError(metric, baseline, candidate);
|
|
1684
2778
|
}
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
}
|
|
1689
|
-
function
|
|
1690
|
-
|
|
1691
|
-
if (retired !== null) return `seal v${retired} retired (this build verifies v${RUN_RECEIPT_INTEGRITY_VERSION})`;
|
|
1692
|
-
return trustStateWord("artifactIntegrity", status.artifactIntegrity.state);
|
|
1693
|
-
}
|
|
1694
|
-
function validationClause(status) {
|
|
1695
|
-
const entry = status.validationGrounding;
|
|
1696
|
-
const word = trustStateWord("validationGrounding", entry.state);
|
|
1697
|
-
if (entry.state === "absent") return word;
|
|
1698
|
-
if (entry.state === "validated" || entry.state === "failed") return `${word} by ${entry.authority.id}`;
|
|
1699
|
-
if (entry.state === "unknown" && entry.authority.kind !== "unknown") return `${word} (${entry.authority.id})`;
|
|
1700
|
-
return word;
|
|
1701
|
-
}
|
|
1702
|
-
function autonomyClause(status) {
|
|
1703
|
-
const entry = status.autonomyEnforcement;
|
|
1704
|
-
const word = trustStateWord("autonomyEnforcement", entry.state);
|
|
1705
|
-
if (entry.state !== "approximated" && entry.state !== "bypassed") return word;
|
|
1706
|
-
return `${word} (${entry.authority.id})`;
|
|
1707
|
-
}
|
|
1708
|
-
function formatTrustSummary(status) {
|
|
1709
|
-
return [
|
|
1710
|
-
integrityClause(status),
|
|
1711
|
-
validationClause(status),
|
|
1712
|
-
trustStateWord("independentReview", status.independentReview.state),
|
|
1713
|
-
autonomyClause(status),
|
|
1714
|
-
trustStateWord("contextProvenance", status.contextProvenance.state),
|
|
1715
|
-
trustStateWord("completionEvidence", status.completionEvidence.state)
|
|
1716
|
-
].join("; ");
|
|
1717
|
-
}
|
|
1718
|
-
function formatTrustSummaryLine(status) {
|
|
1719
|
-
return `trust v${TRUST_SUMMARY_VERSION}: ${formatTrustSummary(status)}`;
|
|
2779
|
+
|
|
2780
|
+
// src/domains/prompts/hash.ts
|
|
2781
|
+
init_esm_shims();
|
|
2782
|
+
import { createHash as createHash5 } from "node:crypto";
|
|
2783
|
+
function sha2563(input) {
|
|
2784
|
+
return createHash5("sha256").update(input, "utf8").digest("hex");
|
|
1720
2785
|
}
|
|
1721
|
-
function
|
|
1722
|
-
return
|
|
1723
|
-
" "
|
|
1724
|
-
);
|
|
2786
|
+
function canonicalJson2(value) {
|
|
2787
|
+
return serialize(value);
|
|
1725
2788
|
}
|
|
1726
|
-
function
|
|
1727
|
-
|
|
1728
|
-
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
const review = status.independentReview.state;
|
|
1734
|
-
const autonomy = status.autonomyEnforcement.state;
|
|
1735
|
-
if (integrity === "failed" || autonomy === "bypassed" || validation === "failed" || validation === "ungrounded" || review === "failed" || review === "not_independent" || status.contextProvenance.state === "invalid") {
|
|
1736
|
-
return "compromised";
|
|
2789
|
+
function serialize(value) {
|
|
2790
|
+
if (value === null) return "null";
|
|
2791
|
+
if (typeof value === "number") {
|
|
2792
|
+
if (!Number.isFinite(value)) {
|
|
2793
|
+
throw new Error(`canonicalJson: non-finite number ${String(value)} is not representable`);
|
|
2794
|
+
}
|
|
2795
|
+
return JSON.stringify(value);
|
|
1737
2796
|
}
|
|
1738
|
-
if (
|
|
1739
|
-
|
|
1740
|
-
if (validation === "validated") return "grounded";
|
|
1741
|
-
return "unverified";
|
|
1742
|
-
}
|
|
1743
|
-
function claimantOf(status) {
|
|
1744
|
-
const state = status.validationGrounding.state;
|
|
1745
|
-
if (state === "absent" || state === "ungrounded") return "worker";
|
|
1746
|
-
return authorityOf(status, "validationGrounding") ?? "worker";
|
|
1747
|
-
}
|
|
1748
|
-
function referenceKey(reference) {
|
|
1749
|
-
return `${reference.kind}:${reference.id}`;
|
|
1750
|
-
}
|
|
1751
|
-
function trustSummaryReferences(status) {
|
|
1752
|
-
const keys = /* @__PURE__ */ new Set();
|
|
1753
|
-
for (const axis of TRUST_STATUS_AXES) {
|
|
1754
|
-
const entry = status[axis];
|
|
1755
|
-
if (entry.state === "absent") continue;
|
|
1756
|
-
for (const reference of entry.artifacts) keys.add(referenceKey(reference));
|
|
2797
|
+
if (typeof value === "string" || typeof value === "boolean") {
|
|
2798
|
+
return JSON.stringify(value);
|
|
1757
2799
|
}
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
|
|
1770
|
-
|
|
2800
|
+
if (typeof value === "bigint") {
|
|
2801
|
+
throw new Error("canonicalJson: bigint is not representable");
|
|
2802
|
+
}
|
|
2803
|
+
if (typeof value === "symbol" || typeof value === "function") {
|
|
2804
|
+
throw new Error(`canonicalJson: ${typeof value} is not representable`);
|
|
2805
|
+
}
|
|
2806
|
+
if (value === void 0) {
|
|
2807
|
+
throw new Error("canonicalJson: undefined is not representable at root");
|
|
2808
|
+
}
|
|
2809
|
+
if (Array.isArray(value)) {
|
|
2810
|
+
const parts = [];
|
|
2811
|
+
for (let i = 0; i < value.length; i++) {
|
|
2812
|
+
if (!(i in value) || value[i] === void 0) {
|
|
2813
|
+
parts.push("null");
|
|
2814
|
+
continue;
|
|
2815
|
+
}
|
|
2816
|
+
parts.push(serialize(value[i]));
|
|
2817
|
+
}
|
|
2818
|
+
return `[${parts.join(",")}]`;
|
|
2819
|
+
}
|
|
2820
|
+
if (typeof value === "object") {
|
|
2821
|
+
const obj = value;
|
|
2822
|
+
const keys = Object.keys(obj).sort();
|
|
2823
|
+
const parts = [];
|
|
2824
|
+
for (const key of keys) {
|
|
2825
|
+
const child = obj[key];
|
|
2826
|
+
if (child === void 0) continue;
|
|
2827
|
+
parts.push(`${JSON.stringify(key)}:${serialize(child)}`);
|
|
2828
|
+
}
|
|
2829
|
+
return `{${parts.join(",")}}`;
|
|
2830
|
+
}
|
|
2831
|
+
throw new Error(`canonicalJson: unsupported value of type ${typeof value}`);
|
|
1771
2832
|
}
|
|
1772
2833
|
|
|
1773
2834
|
export {
|
|
2835
|
+
sha2563 as sha256,
|
|
2836
|
+
canonicalJson2 as canonicalJson,
|
|
1774
2837
|
RUN_RECEIPT_INTEGRITY_VERSION,
|
|
1775
2838
|
RUN_RECEIPT_INTEGRITY_ALGORITHM,
|
|
1776
2839
|
withReceiptIntegrity,
|
|
1777
2840
|
isReceiptIntegrity,
|
|
1778
2841
|
verifyReceiptIntegrity,
|
|
2842
|
+
EVAL_VERDICT_SCHEMA_V1,
|
|
2843
|
+
EVAL_TRACKED_METRIC_NAMES,
|
|
2844
|
+
parseEvalVerdictEnvelopeV1,
|
|
2845
|
+
parseEvalBehaviorScenarioV1,
|
|
2846
|
+
judgeEvalBehaviorV1,
|
|
2847
|
+
assertEvalBehaviorReferencesVerdictV1,
|
|
2848
|
+
EVAL_BEHAVIOR_METRIC_DEFINITIONS_V1,
|
|
2849
|
+
buildEvalBehaviorMetricsV1,
|
|
2850
|
+
EVAL_EXECUTION_ENVELOPE_SCHEMA_V1,
|
|
2851
|
+
parseEvalExecutionMatrixDimensionsV1,
|
|
2852
|
+
sameEvalServingConfiguration,
|
|
2853
|
+
renderEvalServingConfiguration,
|
|
1779
2854
|
redactArtifactForStorage,
|
|
1780
2855
|
ZERO_EVAL_HARNESS_METRICS,
|
|
2856
|
+
evalHarnessMetricsFromReceipt,
|
|
1781
2857
|
sumEvalHarnessMetrics,
|
|
1782
2858
|
createEvalId,
|
|
1783
2859
|
writeEvalArtifact,
|
|
@@ -1796,14 +2872,17 @@ export {
|
|
|
1796
2872
|
inspectRunReceiptTrustStatus,
|
|
1797
2873
|
adaptGateDecisionReviewStatus,
|
|
1798
2874
|
adaptGroundedEvidenceValidationStatus,
|
|
1799
|
-
adaptFinishContractCompletionStatus,
|
|
1800
|
-
evalClioProvenance,
|
|
1801
|
-
evalEnvironmentProvenance,
|
|
1802
|
-
loadEvalTaskFile,
|
|
1803
|
-
EvalTaskFileError,
|
|
1804
2875
|
formatTrustSummary,
|
|
1805
2876
|
formatTrustSummaryLine,
|
|
1806
2877
|
formatTrustAxes,
|
|
1807
|
-
|
|
2878
|
+
trustVerdict,
|
|
2879
|
+
summarizeTrustStatus,
|
|
2880
|
+
evalClioProvenance,
|
|
2881
|
+
evalEnvironmentProvenance,
|
|
2882
|
+
evalServingConfiguration,
|
|
2883
|
+
evalServingObservationFrom,
|
|
2884
|
+
assertComparableTrackedMetricSources,
|
|
2885
|
+
loadEvalTaskFile,
|
|
2886
|
+
EvalTaskFileError
|
|
1808
2887
|
};
|
|
1809
|
-
//# sourceMappingURL=chunk-
|
|
2888
|
+
//# sourceMappingURL=chunk-MLOK6ZOS.js.map
|