@iowarp/clio-coder 0.3.8 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +7 -3
- package/dist/{acp-U67UHUK2.js → acp-7LOELQFP.js} +6 -6
- package/dist/{agents-YU6SGALZ.js → agents-FIBG2SHA.js} +27 -25
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-5ZPJOIVG.js → auth-OI4LIH2I.js} +11 -12
- package/dist/{builtins-C6JMZVV6.js → builtins-AD25UL3C.js} +5 -5
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-4SPRNWDE.js → chunk-3DUR4WUA.js} +15 -15
- package/dist/{chunk-VHN4MY6O.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-5DHKRSMQ.js → chunk-3UUY7R3Z.js} +11 -7
- package/dist/{chunk-IGWKHNIQ.js → chunk-3V5AYSEQ.js} +8 -8
- package/dist/{chunk-FHJEP5SW.js → chunk-465CC7FK.js} +8 -5
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-A3WNZD3P.js → chunk-4H6ULJ3H.js} +67 -21
- package/dist/{chunk-TB5666IT.js → chunk-4LJX2PUC.js} +3 -3
- package/dist/{chunk-XWSF374K.js → chunk-56KB5IJP.js} +2 -2
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-2HEJ2F35.js → chunk-5HFBWUMU.js} +20 -8
- package/dist/{chunk-WNIJTQQK.js → chunk-5PVQ4SRS.js} +78 -6
- package/dist/{chunk-DYIM5TJT.js → chunk-5QKCQQ3E.js} +262 -6
- package/dist/{chunk-TYPGUK6W.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/{chunk-IIZWH4XA.js → chunk-774ILSRL.js} +2 -2
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-TANS5ZJS.js → chunk-AD7Y7STJ.js} +3 -3
- package/dist/{chunk-RWSI4YD7.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-DGSYXYMX.js → chunk-AMKHQW3C.js} +2 -2
- package/dist/{chunk-VWZOAB7K.js → chunk-B5XRQOLB.js} +7 -7
- package/dist/{chunk-WXY7KU3G.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-PMDBGQSJ.js → chunk-CA42X6KT.js} +2 -2
- package/dist/{chunk-HLE42MG7.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-5Q2VVUKB.js → chunk-DG4M6ZUE.js} +3 -3
- package/dist/{chunk-ME6CCNFO.js → chunk-EBOC7MT3.js} +6 -6
- package/dist/{chunk-MXKJU4JB.js → chunk-ECUO3KDP.js} +48 -7
- package/dist/{chunk-26LEYJZH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-7RGZWPB6.js → chunk-GAYUJ7LE.js} +67 -13
- package/dist/{chunk-VCBR6CU7.js → chunk-HAY4ZE2P.js} +2 -2
- package/dist/{chunk-FBVTI2TJ.js → chunk-HCBCAYZU.js} +11 -130
- package/dist/{chunk-WSB3FPX7.js → chunk-HJB5IUKP.js} +32 -136
- package/dist/{chunk-J3YUBZWY.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-E77JEWSD.js → chunk-HPCTNZM2.js} +6 -36
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/{chunk-NMPKI6XL.js → chunk-JEQQR47K.js} +37 -18
- package/dist/{chunk-3BINW3FP.js → chunk-KV2AOLDF.js} +24 -4
- package/dist/{chunk-YS5VLNH5.js → chunk-LXPJXFM5.js} +7 -7
- package/dist/{chunk-7RFXX52T.js → chunk-MIX5N5AC.js} +271 -41
- package/dist/{chunk-K4XHGFR5.js → chunk-MLOK6ZOS.js} +1297 -218
- package/dist/{chunk-ZNLWCMVZ.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-5H3GB5BO.js → chunk-N3PBVRTZ.js} +4 -382
- package/dist/{chunk-2HFZQUHL.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-GU2UIAFZ.js → chunk-NQ6UCCOD.js} +3 -3
- package/dist/{chunk-N22QMJKY.js → chunk-NZU6YDNV.js} +4 -4
- package/dist/{chunk-ZVJ5BLO2.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-VAWNZU7Z.js → chunk-P3JGPQFL.js} +2 -2
- package/dist/{chunk-IJ7RPIYJ.js → chunk-PNY46YEY.js} +20 -3
- package/dist/{chunk-U6MBIEMB.js → chunk-PZ4I4JE2.js} +56 -35
- package/dist/{chunk-GPIEI3LY.js → chunk-QQ7EKM72.js} +2 -2
- package/dist/{chunk-WLFILSD5.js → chunk-R7LNVMCS.js} +66 -28
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-TT36MB5S.js → chunk-RKRLDWD3.js} +3 -1
- package/dist/{chunk-TTHACPOM.js → chunk-S4COXYBG.js} +456 -18
- package/dist/{chunk-WWCZ5F23.js → chunk-T3Z6VAAF.js} +69 -10
- package/dist/{chunk-GN57SG4G.js → chunk-TD7UE2L5.js} +9 -7
- package/dist/{chunk-TLQJPP24.js → chunk-TEO2TLVN.js} +523 -322
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-KTYTFRMB.js → chunk-VKBMFOYV.js} +17 -15
- package/dist/{chunk-PT7HYKEM.js → chunk-VO2LKSTM.js} +2 -2
- package/dist/{chunk-P43ETTHK.js → chunk-VPTUJU4P.js} +2 -2
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-EMYUUSFG.js → chunk-WXCJ7VME.js} +5 -5
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-JOZYP4GM.js → chunk-YKOFT37S.js} +5 -5
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +27 -27
- package/dist/{clio-QVTYJ57A.js → clio-LT5V7SSZ.js} +6 -6
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +4 -4
- package/dist/{config-LW5IJFQN.js → config-RXS5T3JT.js} +71 -43
- package/dist/{configure-7XIZCOU4.js → configure-2WYWSCSD.js} +14 -15
- package/dist/{context-Y6Y7QPR6.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-L3WL3X7K.js → context-MVOORGMF.js} +34 -33
- package/dist/{context-N52ZA626.js → context-PALKKQYL.js} +20 -20
- package/dist/{context-clear-MBQRLSDQ.js → context-clear-N2WOYZ2K.js} +34 -33
- package/dist/{context-index-HVMFQHK3.js → context-index-HNG3MOME.js} +2 -2
- package/dist/{context-working-set-GS6DSO7F.js → context-working-set-MIEVECVZ.js} +10 -11
- package/dist/{dispatch-runner-22ZCNOM3.js → dispatch-runner-VVA4SRRH.js} +34 -33
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/{eval-BEC2WHDA.js → eval-IJ5VEZDJ.js} +2016 -142
- package/dist/{evidence-REJUMSKM.js → evidence-L5APPXNV.js} +29 -28
- package/dist/{evolve-PY5ZBA5K.js → evolve-RGNKFJ52.js} +29 -28
- package/dist/{extensions-HVKU65YU.js → extensions-7WYWUX5A.js} +9 -3
- package/dist/{fleet-7WZEWRFA.js → fleet-6CNVBZZP.js} +87 -55
- package/dist/{fleet-commands-UVHWM76J.js → fleet-commands-L2SXSYEI.js} +6 -6
- package/dist/{fleet-graph-6ULH7PES.js → fleet-graph-2J3OOIPO.js} +14 -12
- package/dist/{fleet-preflight-J53T6CCE.js → fleet-preflight-CZRJ4JP5.js} +3 -4
- package/dist/{fleet-validate-72PC4SLA.js → fleet-validate-C5RI6DP7.js} +16 -15
- package/dist/{init-OG3TPGQG.js → init-VBN2ACVA.js} +50 -48
- package/dist/{library-CNTMPLRF.js → library-JHGUMLY2.js} +13 -11
- package/dist/{memory-6IS7F275.js → memory-K4OQIYWG.js} +31 -30
- package/dist/{models-ENRJDA5W.js → models-2NCZUWDD.js} +23 -22
- package/dist/{monitor-XLDVO7TN.js → monitor-MMVTJABD.js} +35 -34
- package/dist/{orchestrator-6KSPYRHA.js → orchestrator-ZKBPCHW6.js} +1627 -294
- package/dist/{reset-RZ4ER727.js → reset-DD5JGOY3.js} +3 -3
- package/dist/{run-Y2CNK5RU.js → run-QEGNX7FL.js} +56 -55
- package/dist/{share-A55GYP6Z.js → share-JKD3BQMW.js} +13 -11
- package/dist/{skills-ALC5J6AT.js → skills-LMQIKDOZ.js} +14 -12
- package/dist/{skills-eval-JPBEBYQU.js → skills-eval-I7X2774U.js} +34 -32
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/{support-MIETYA5E.js → support-I7LOJLIF.js} +4 -4
- package/dist/{targets-VGNXIR3S.js → targets-RUSR6B5Z.js} +58 -32
- package/dist/{terminal-lease-WOBR64YA.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{upgrade-FUSUAGHR.js → upgrade-XANW3FXB.js} +17 -16
- package/dist/{usage-N4MKVHKD.js → usage-4H7ZRXQT.js} +88 -44
- package/dist/{verifiers-YAWOJ3H2.js → verifiers-UZXNBZEB.js} +6 -6
- package/dist/{verify-LTDHYBGY.js → verify-BVKWTNDL.js} +5 -5
- package/dist/{wiki-generate-6M7GHTBJ.js → wiki-generate-MY7WV2QI.js} +49 -47
- package/dist/worker/entry.js +29 -33
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +6 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +1 -1
- package/docs/configuration-and-targets.md +30 -3
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +11 -10
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +4 -4
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +53 -2
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +19 -1
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +19 -3
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -1
- package/src/cli/agents.ts +1 -1
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +3 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet.ts +32 -3
- package/src/cli/targets.ts +44 -13
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/core/bus-events.ts +29 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +11 -0
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/extension.ts +166 -42
- package/src/domains/dispatch/fleet-run.ts +23 -3
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +3 -0
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +20 -3
- package/src/domains/dispatch/write-boundary.ts +62 -1
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +127 -0
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +2 -13
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +2 -2
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +38 -3
- package/src/domains/extensions/resources.ts +1 -1
- package/src/domains/extensions/state.ts +12 -3
- package/src/domains/extensions/types.ts +2 -0
- package/src/domains/lifecycle/doctor.ts +69 -1
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/prompts/loader.ts +95 -33
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/run-effects.ts +35 -4
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +84 -12
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +35 -1
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/view-overlay.ts +28 -3
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-plan.ts +17 -9
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/registry.ts +16 -0
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/doctor-M7YEDGAE.js +0 -91
|
@@ -0,0 +1,520 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { EVAL_VERDICT_SCHEMA_V1, type EvalVerdictEnvelopeV1 } from "./verdict.js";
|
|
3
|
+
|
|
4
|
+
export const EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1 = "clio.eval.scenario.v1" as const;
|
|
5
|
+
export const EVAL_BEHAVIOR_SCHEMA_V1 = "clio.eval.behavior.v1" as const;
|
|
6
|
+
|
|
7
|
+
export const EVAL_BEHAVIOR_CATEGORIES = [
|
|
8
|
+
"tool_choice",
|
|
9
|
+
"exploration",
|
|
10
|
+
"delegation",
|
|
11
|
+
"safety_comprehension",
|
|
12
|
+
"claim_grounding",
|
|
13
|
+
"denied_tool_recovery",
|
|
14
|
+
"completion_behavior",
|
|
15
|
+
"task_correctness",
|
|
16
|
+
] as const;
|
|
17
|
+
|
|
18
|
+
export type EvalBehaviorCategoryV1 = (typeof EVAL_BEHAVIOR_CATEGORIES)[number];
|
|
19
|
+
export type EvalBehaviorExecutionModeV1 = "machinery-only" | "model-required";
|
|
20
|
+
export type EvalBehaviorFactSourceV1 = "transcript" | "tool" | "receipt" | "grader";
|
|
21
|
+
export type EvalBehaviorRuleOpV1 = "eq" | "neq" | "lt" | "lte" | "gt" | "gte";
|
|
22
|
+
export type EvalBehaviorScalarV1 = string | number | boolean;
|
|
23
|
+
export type EvalBehaviorLabelV1 = "satisfied" | "violated" | "unknown" | "unmeasured";
|
|
24
|
+
export type EvalBehaviorOutcomeV1 = "pass" | "behavioral_failure" | "unknown" | "unmeasured" | "infrastructure_failure";
|
|
25
|
+
|
|
26
|
+
const MAX_RULES = 64;
|
|
27
|
+
const MAX_FACTS = 128;
|
|
28
|
+
const MAX_EVIDENCE_PER_LABEL = 8;
|
|
29
|
+
const MAX_ID_CHARS = 128;
|
|
30
|
+
const MAX_TEXT_CHARS = 1_000;
|
|
31
|
+
const MAX_EXPLANATION_CHARS = 2_000;
|
|
32
|
+
|
|
33
|
+
export interface EvalBehaviorPredicateV1 {
|
|
34
|
+
source: EvalBehaviorFactSourceV1;
|
|
35
|
+
key: string;
|
|
36
|
+
op: EvalBehaviorRuleOpV1;
|
|
37
|
+
value: EvalBehaviorScalarV1;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export interface EvalBehaviorRuleV1 {
|
|
41
|
+
id: string;
|
|
42
|
+
category: EvalBehaviorCategoryV1;
|
|
43
|
+
fact: EvalBehaviorPredicateV1;
|
|
44
|
+
rationale?: string;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface EvalBehaviorScenarioV1 {
|
|
48
|
+
schema: typeof EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1;
|
|
49
|
+
corpus: { id: string; version: string };
|
|
50
|
+
execution: {
|
|
51
|
+
mode: EvalBehaviorExecutionModeV1;
|
|
52
|
+
subject: { kind: "main-agent" | "worker"; role: string };
|
|
53
|
+
toolTarget: "available" | "none";
|
|
54
|
+
};
|
|
55
|
+
expectedBehavior: EvalBehaviorRuleV1[];
|
|
56
|
+
forbiddenBehavior: EvalBehaviorRuleV1[];
|
|
57
|
+
judge: { maxEvidenceItems: number; maxExplanationChars: number };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface EvalBehaviorEvidenceV1 {
|
|
61
|
+
factId: string;
|
|
62
|
+
source: EvalBehaviorFactSourceV1;
|
|
63
|
+
locator: string;
|
|
64
|
+
digest: string;
|
|
65
|
+
excerpt: string | null;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export interface EvalBehaviorJudgeFactV1 {
|
|
69
|
+
id: string;
|
|
70
|
+
source: EvalBehaviorFactSourceV1;
|
|
71
|
+
key: string;
|
|
72
|
+
value: EvalBehaviorScalarV1;
|
|
73
|
+
evidence: Omit<EvalBehaviorEvidenceV1, "factId" | "source">;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export interface EvalBehaviorJudgeInputV1 {
|
|
77
|
+
facts: EvalBehaviorJudgeFactV1[];
|
|
78
|
+
unavailableSources: EvalBehaviorFactSourceV1[];
|
|
79
|
+
infrastructureFailure: boolean;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
export interface EvalBehaviorLabelResultV1 {
|
|
83
|
+
category: EvalBehaviorCategoryV1;
|
|
84
|
+
label: EvalBehaviorLabelV1;
|
|
85
|
+
ruleIds: string[];
|
|
86
|
+
evidence: EvalBehaviorEvidenceV1[];
|
|
87
|
+
explanation: string | null;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export interface EvalBehaviorVerdictV1 {
|
|
91
|
+
schema: typeof EVAL_BEHAVIOR_SCHEMA_V1;
|
|
92
|
+
verdictRef: {
|
|
93
|
+
schema: typeof EVAL_VERDICT_SCHEMA_V1;
|
|
94
|
+
scenarioId: string;
|
|
95
|
+
trialIndex: number;
|
|
96
|
+
};
|
|
97
|
+
corpus: { id: string; version: string };
|
|
98
|
+
judgeInputDigest: string;
|
|
99
|
+
outcome: EvalBehaviorOutcomeV1;
|
|
100
|
+
labels: EvalBehaviorLabelResultV1[];
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export function parseEvalBehaviorScenarioV1(value: unknown, source = "behavioral scenario"): EvalBehaviorScenarioV1 {
|
|
104
|
+
const record = asRecord(value, source);
|
|
105
|
+
if (record.schema !== EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1) {
|
|
106
|
+
throw new Error(`${source}.schema: expected ${EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1}`);
|
|
107
|
+
}
|
|
108
|
+
const corpus = asRecord(record.corpus, `${source}.corpus`);
|
|
109
|
+
const execution = asRecord(record.execution, `${source}.execution`);
|
|
110
|
+
const subject = asRecord(execution.subject, `${source}.execution.subject`);
|
|
111
|
+
const mode = execution.mode;
|
|
112
|
+
if (mode !== "machinery-only" && mode !== "model-required") {
|
|
113
|
+
throw new Error(`${source}.execution.mode: expected machinery-only or model-required`);
|
|
114
|
+
}
|
|
115
|
+
const subjectKind = subject.kind;
|
|
116
|
+
if (subjectKind !== "main-agent" && subjectKind !== "worker") {
|
|
117
|
+
throw new Error(`${source}.execution.subject.kind: expected main-agent or worker`);
|
|
118
|
+
}
|
|
119
|
+
const toolTarget = execution.toolTarget;
|
|
120
|
+
if (toolTarget !== "available" && toolTarget !== "none") {
|
|
121
|
+
throw new Error(`${source}.execution.toolTarget: expected available or none`);
|
|
122
|
+
}
|
|
123
|
+
const expectedBehavior = parseRules(record.expectedBehavior, `${source}.expectedBehavior`);
|
|
124
|
+
const forbiddenBehavior = parseRules(record.forbiddenBehavior, `${source}.forbiddenBehavior`);
|
|
125
|
+
const ruleIds = new Set<string>();
|
|
126
|
+
for (const rule of [...expectedBehavior, ...forbiddenBehavior]) {
|
|
127
|
+
if (ruleIds.has(rule.id)) throw new Error(`${source}: duplicate behavioral rule id ${rule.id}`);
|
|
128
|
+
ruleIds.add(rule.id);
|
|
129
|
+
}
|
|
130
|
+
if (ruleIds.size === 0) throw new Error(`${source}: expected at least one behavioral rule`);
|
|
131
|
+
const judge = asRecord(record.judge, `${source}.judge`);
|
|
132
|
+
const maxEvidenceItems = readBoundedInteger(judge.maxEvidenceItems, `${source}.judge.maxEvidenceItems`, 1, MAX_FACTS);
|
|
133
|
+
const maxExplanationChars = readBoundedInteger(
|
|
134
|
+
judge.maxExplanationChars,
|
|
135
|
+
`${source}.judge.maxExplanationChars`,
|
|
136
|
+
1,
|
|
137
|
+
MAX_EXPLANATION_CHARS,
|
|
138
|
+
);
|
|
139
|
+
return {
|
|
140
|
+
schema: EVAL_BEHAVIOR_SCENARIO_SCHEMA_V1,
|
|
141
|
+
corpus: {
|
|
142
|
+
id: readId(corpus.id, `${source}.corpus.id`),
|
|
143
|
+
version: readText(corpus.version, `${source}.corpus.version`, MAX_ID_CHARS),
|
|
144
|
+
},
|
|
145
|
+
execution: {
|
|
146
|
+
mode,
|
|
147
|
+
subject: { kind: subjectKind, role: readId(subject.role, `${source}.execution.subject.role`) },
|
|
148
|
+
toolTarget,
|
|
149
|
+
},
|
|
150
|
+
expectedBehavior,
|
|
151
|
+
forbiddenBehavior,
|
|
152
|
+
judge: { maxEvidenceItems, maxExplanationChars },
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export function canonicalizeEvalBehaviorJudgeInputV1(
|
|
157
|
+
value: unknown,
|
|
158
|
+
scenario: EvalBehaviorScenarioV1,
|
|
159
|
+
source = "behavioral judge input",
|
|
160
|
+
): { input: EvalBehaviorJudgeInputV1; digest: string } {
|
|
161
|
+
const record = asRecord(value, source);
|
|
162
|
+
if (!Array.isArray(record.facts)) throw new Error(`${source}.facts: expected array`);
|
|
163
|
+
if (record.facts.length > scenario.judge.maxEvidenceItems || record.facts.length > MAX_FACTS) {
|
|
164
|
+
throw new Error(`${source}.facts: exceeds bounded evidence limit`);
|
|
165
|
+
}
|
|
166
|
+
const facts = record.facts.map((fact, index) => parseFact(fact, `${source}.facts[${index}]`));
|
|
167
|
+
const factIds = new Set<string>();
|
|
168
|
+
const factKeys = new Set<string>();
|
|
169
|
+
for (const fact of facts) {
|
|
170
|
+
if (factIds.has(fact.id)) throw new Error(`${source}: duplicate fact id ${fact.id}`);
|
|
171
|
+
const key = `${fact.source}\u0000${fact.key}`;
|
|
172
|
+
if (factKeys.has(key)) throw new Error(`${source}: conflicting fact ${fact.source}.${fact.key}`);
|
|
173
|
+
factIds.add(fact.id);
|
|
174
|
+
factKeys.add(key);
|
|
175
|
+
}
|
|
176
|
+
const unavailableSources = parseSources(record.unavailableSources, `${source}.unavailableSources`);
|
|
177
|
+
const infrastructureFailure = record.infrastructureFailure;
|
|
178
|
+
if (typeof infrastructureFailure !== "boolean") {
|
|
179
|
+
throw new Error(`${source}.infrastructureFailure: expected boolean`);
|
|
180
|
+
}
|
|
181
|
+
const input: EvalBehaviorJudgeInputV1 = {
|
|
182
|
+
facts: facts.sort((left, right) => left.source.localeCompare(right.source) || left.key.localeCompare(right.key)),
|
|
183
|
+
unavailableSources: [...new Set(unavailableSources)].sort(),
|
|
184
|
+
infrastructureFailure,
|
|
185
|
+
};
|
|
186
|
+
return { input, digest: sha256(stableJson(input)) };
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export function judgeEvalBehaviorV1(
|
|
190
|
+
scenarioValue: EvalBehaviorScenarioV1,
|
|
191
|
+
verdict: Pick<EvalVerdictEnvelopeV1, "schema" | "scenarioId" | "trialIndex">,
|
|
192
|
+
inputValue: unknown,
|
|
193
|
+
): EvalBehaviorVerdictV1 {
|
|
194
|
+
const scenario = parseEvalBehaviorScenarioV1(scenarioValue);
|
|
195
|
+
const { input, digest } = canonicalizeEvalBehaviorJudgeInputV1(inputValue, scenario);
|
|
196
|
+
const labels = EVAL_BEHAVIOR_CATEGORIES.map((category) =>
|
|
197
|
+
judgeCategory(category, scenario, input, scenario.judge.maxExplanationChars),
|
|
198
|
+
);
|
|
199
|
+
const outcome: EvalBehaviorOutcomeV1 = input.infrastructureFailure
|
|
200
|
+
? "infrastructure_failure"
|
|
201
|
+
: labels.some((label) => label.label === "violated")
|
|
202
|
+
? "behavioral_failure"
|
|
203
|
+
: labels.some((label) => label.label === "unknown")
|
|
204
|
+
? "unknown"
|
|
205
|
+
: labels.every((label) => label.label === "unmeasured")
|
|
206
|
+
? "unmeasured"
|
|
207
|
+
: "pass";
|
|
208
|
+
return parseEvalBehaviorVerdictV1({
|
|
209
|
+
schema: EVAL_BEHAVIOR_SCHEMA_V1,
|
|
210
|
+
verdictRef: {
|
|
211
|
+
schema: verdict.schema,
|
|
212
|
+
scenarioId: verdict.scenarioId,
|
|
213
|
+
trialIndex: verdict.trialIndex,
|
|
214
|
+
},
|
|
215
|
+
corpus: scenario.corpus,
|
|
216
|
+
judgeInputDigest: digest,
|
|
217
|
+
outcome,
|
|
218
|
+
labels,
|
|
219
|
+
});
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
export function parseEvalBehaviorVerdictV1(value: unknown, source = "behavioral verdict"): EvalBehaviorVerdictV1 {
|
|
223
|
+
const record = asRecord(value, source);
|
|
224
|
+
if (record.schema !== EVAL_BEHAVIOR_SCHEMA_V1)
|
|
225
|
+
throw new Error(`${source}.schema: expected ${EVAL_BEHAVIOR_SCHEMA_V1}`);
|
|
226
|
+
const verdictRef = asRecord(record.verdictRef, `${source}.verdictRef`);
|
|
227
|
+
if (verdictRef.schema !== EVAL_VERDICT_SCHEMA_V1) {
|
|
228
|
+
throw new Error(`${source}.verdictRef.schema: expected ${EVAL_VERDICT_SCHEMA_V1}`);
|
|
229
|
+
}
|
|
230
|
+
const corpus = asRecord(record.corpus, `${source}.corpus`);
|
|
231
|
+
const outcome = readOutcome(record.outcome, `${source}.outcome`);
|
|
232
|
+
if (!Array.isArray(record.labels)) throw new Error(`${source}.labels: expected array`);
|
|
233
|
+
const labels = record.labels.map((label, index) => parseLabel(label, `${source}.labels[${index}]`));
|
|
234
|
+
const categories = labels.map((label) => label.category);
|
|
235
|
+
if (
|
|
236
|
+
labels.length !== EVAL_BEHAVIOR_CATEGORIES.length ||
|
|
237
|
+
new Set(categories).size !== EVAL_BEHAVIOR_CATEGORIES.length
|
|
238
|
+
) {
|
|
239
|
+
throw new Error(`${source}.labels: expected every behavioral category exactly once`);
|
|
240
|
+
}
|
|
241
|
+
for (const category of EVAL_BEHAVIOR_CATEGORIES) {
|
|
242
|
+
if (!categories.includes(category)) throw new Error(`${source}.labels: missing category ${category}`);
|
|
243
|
+
}
|
|
244
|
+
const derived = deriveOutcome(labels, outcome === "infrastructure_failure");
|
|
245
|
+
if (outcome !== derived) throw new Error(`${source}.outcome: ${outcome} conflicts with labels (expected ${derived})`);
|
|
246
|
+
return {
|
|
247
|
+
schema: EVAL_BEHAVIOR_SCHEMA_V1,
|
|
248
|
+
verdictRef: {
|
|
249
|
+
schema: EVAL_VERDICT_SCHEMA_V1,
|
|
250
|
+
scenarioId: readId(verdictRef.scenarioId, `${source}.verdictRef.scenarioId`),
|
|
251
|
+
trialIndex: readBoundedInteger(verdictRef.trialIndex, `${source}.verdictRef.trialIndex`, 0, Number.MAX_SAFE_INTEGER),
|
|
252
|
+
},
|
|
253
|
+
corpus: {
|
|
254
|
+
id: readId(corpus.id, `${source}.corpus.id`),
|
|
255
|
+
version: readText(corpus.version, `${source}.corpus.version`, MAX_ID_CHARS),
|
|
256
|
+
},
|
|
257
|
+
judgeInputDigest: readDigest(record.judgeInputDigest, `${source}.judgeInputDigest`),
|
|
258
|
+
outcome,
|
|
259
|
+
labels,
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
export function assertEvalBehaviorReferencesVerdictV1(
|
|
264
|
+
behavior: EvalBehaviorVerdictV1,
|
|
265
|
+
verdict: EvalVerdictEnvelopeV1,
|
|
266
|
+
source = "behavioral verdict",
|
|
267
|
+
): void {
|
|
268
|
+
if (behavior.verdictRef.scenarioId !== verdict.scenarioId || behavior.verdictRef.trialIndex !== verdict.trialIndex) {
|
|
269
|
+
throw new Error(`${source}.verdictRef: conflicts with result verdict identity`);
|
|
270
|
+
}
|
|
271
|
+
if (verdict.machinery === "infrastructure_failure" && behavior.outcome !== "infrastructure_failure") {
|
|
272
|
+
throw new Error(`${source}.outcome: machinery failure must remain infrastructure_failure`);
|
|
273
|
+
}
|
|
274
|
+
if (behavior.outcome === "pass" && verdict.outcome !== "pass") {
|
|
275
|
+
throw new Error(`${source}.outcome: behavioral pass cannot override a failed or unmeasured result`);
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
function judgeCategory(
|
|
280
|
+
category: EvalBehaviorCategoryV1,
|
|
281
|
+
scenario: EvalBehaviorScenarioV1,
|
|
282
|
+
input: EvalBehaviorJudgeInputV1,
|
|
283
|
+
maxExplanationChars: number,
|
|
284
|
+
): EvalBehaviorLabelResultV1 {
|
|
285
|
+
const expected = scenario.expectedBehavior.filter((rule) => rule.category === category);
|
|
286
|
+
const forbidden = scenario.forbiddenBehavior.filter((rule) => rule.category === category);
|
|
287
|
+
const rules = [...expected, ...forbidden];
|
|
288
|
+
if (input.infrastructureFailure)
|
|
289
|
+
return labelResult(category, "unknown", rules, [], "infrastructure failure", maxExplanationChars);
|
|
290
|
+
if (rules.length === 0) return labelResult(category, "unmeasured", [], [], null, maxExplanationChars);
|
|
291
|
+
const unavailable = rules.some((rule) => input.unavailableSources.includes(rule.fact.source));
|
|
292
|
+
if (unavailable)
|
|
293
|
+
return labelResult(category, "unmeasured", rules, [], "required evidence source unavailable", maxExplanationChars);
|
|
294
|
+
const evaluations = rules.map((rule) => {
|
|
295
|
+
const fact = input.facts.find(
|
|
296
|
+
(candidate) => candidate.source === rule.fact.source && candidate.key === rule.fact.key,
|
|
297
|
+
);
|
|
298
|
+
if (fact === undefined) return { rule, fact: null, holds: null };
|
|
299
|
+
const matches = compare(fact.value, rule.fact.op, rule.fact.value);
|
|
300
|
+
return { rule, fact, holds: expected.includes(rule) ? matches : !matches };
|
|
301
|
+
});
|
|
302
|
+
const evidence = evaluations
|
|
303
|
+
.flatMap(({ fact }) => (fact === null ? [] : [toEvidence(fact)]))
|
|
304
|
+
.slice(0, MAX_EVIDENCE_PER_LABEL);
|
|
305
|
+
if (evaluations.some(({ holds }) => holds === false)) {
|
|
306
|
+
return labelResult(category, "violated", rules, evidence, "one or more declared rules failed", maxExplanationChars);
|
|
307
|
+
}
|
|
308
|
+
if (evaluations.some(({ holds }) => holds === null)) {
|
|
309
|
+
return labelResult(category, "unknown", rules, evidence, "required observable fact missing", maxExplanationChars);
|
|
310
|
+
}
|
|
311
|
+
return labelResult(category, "satisfied", rules, evidence, null, maxExplanationChars);
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
function labelResult(
|
|
315
|
+
category: EvalBehaviorCategoryV1,
|
|
316
|
+
label: EvalBehaviorLabelV1,
|
|
317
|
+
rules: ReadonlyArray<EvalBehaviorRuleV1>,
|
|
318
|
+
evidence: EvalBehaviorEvidenceV1[],
|
|
319
|
+
explanation: string | null,
|
|
320
|
+
maxExplanationChars: number,
|
|
321
|
+
): EvalBehaviorLabelResultV1 {
|
|
322
|
+
return {
|
|
323
|
+
category,
|
|
324
|
+
label,
|
|
325
|
+
ruleIds: rules.map((rule) => rule.id).sort(),
|
|
326
|
+
evidence,
|
|
327
|
+
explanation: explanation === null ? null : explanation.slice(0, maxExplanationChars),
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
function deriveOutcome(
|
|
332
|
+
labels: ReadonlyArray<EvalBehaviorLabelResultV1>,
|
|
333
|
+
infrastructure: boolean,
|
|
334
|
+
): EvalBehaviorOutcomeV1 {
|
|
335
|
+
if (infrastructure) return "infrastructure_failure";
|
|
336
|
+
if (labels.some((label) => label.label === "violated")) return "behavioral_failure";
|
|
337
|
+
if (labels.some((label) => label.label === "unknown")) return "unknown";
|
|
338
|
+
if (labels.every((label) => label.label === "unmeasured")) return "unmeasured";
|
|
339
|
+
return "pass";
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
function parseRules(value: unknown, source: string): EvalBehaviorRuleV1[] {
|
|
343
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
344
|
+
if (value.length > MAX_RULES) throw new Error(`${source}: exceeds ${MAX_RULES} rules`);
|
|
345
|
+
return value.map((entry, index) => {
|
|
346
|
+
const record = asRecord(entry, `${source}[${index}]`);
|
|
347
|
+
const fact = asRecord(record.fact, `${source}[${index}].fact`);
|
|
348
|
+
const rationale =
|
|
349
|
+
record.rationale === undefined
|
|
350
|
+
? undefined
|
|
351
|
+
: readText(record.rationale, `${source}[${index}].rationale`, MAX_TEXT_CHARS);
|
|
352
|
+
return {
|
|
353
|
+
id: readId(record.id, `${source}[${index}].id`),
|
|
354
|
+
category: readCategory(record.category, `${source}[${index}].category`),
|
|
355
|
+
fact: {
|
|
356
|
+
source: readSource(fact.source, `${source}[${index}].fact.source`),
|
|
357
|
+
key: readText(fact.key, `${source}[${index}].fact.key`, MAX_TEXT_CHARS),
|
|
358
|
+
op: readOp(fact.op, `${source}[${index}].fact.op`),
|
|
359
|
+
value: readScalar(fact.value, `${source}[${index}].fact.value`),
|
|
360
|
+
},
|
|
361
|
+
...(rationale === undefined ? {} : { rationale }),
|
|
362
|
+
};
|
|
363
|
+
});
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
function parseFact(value: unknown, source: string): EvalBehaviorJudgeFactV1 {
|
|
367
|
+
const record = asRecord(value, source);
|
|
368
|
+
const evidence = asRecord(record.evidence, `${source}.evidence`);
|
|
369
|
+
return {
|
|
370
|
+
id: readId(record.id, `${source}.id`),
|
|
371
|
+
source: readSource(record.source, `${source}.source`),
|
|
372
|
+
key: readText(record.key, `${source}.key`, MAX_TEXT_CHARS),
|
|
373
|
+
value: readScalar(record.value, `${source}.value`),
|
|
374
|
+
evidence: {
|
|
375
|
+
locator: readText(evidence.locator, `${source}.evidence.locator`, MAX_TEXT_CHARS),
|
|
376
|
+
digest: readDigest(evidence.digest, `${source}.evidence.digest`),
|
|
377
|
+
excerpt: evidence.excerpt === null ? null : readText(evidence.excerpt, `${source}.evidence.excerpt`, MAX_TEXT_CHARS),
|
|
378
|
+
},
|
|
379
|
+
};
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
function parseLabel(value: unknown, source: string): EvalBehaviorLabelResultV1 {
|
|
383
|
+
const record = asRecord(value, source);
|
|
384
|
+
const label = record.label;
|
|
385
|
+
if (label !== "satisfied" && label !== "violated" && label !== "unknown" && label !== "unmeasured") {
|
|
386
|
+
throw new Error(`${source}.label: expected satisfied, violated, unknown, or unmeasured`);
|
|
387
|
+
}
|
|
388
|
+
if (!Array.isArray(record.ruleIds) || !Array.isArray(record.evidence)) {
|
|
389
|
+
throw new Error(`${source}: expected ruleIds and evidence arrays`);
|
|
390
|
+
}
|
|
391
|
+
if (record.evidence.length > MAX_EVIDENCE_PER_LABEL) throw new Error(`${source}.evidence: exceeds bounded limit`);
|
|
392
|
+
if ((label === "satisfied" || label === "violated") && (record.ruleIds.length === 0 || record.evidence.length === 0)) {
|
|
393
|
+
throw new Error(`${source}: ${label} label requires a rule and observable evidence`);
|
|
394
|
+
}
|
|
395
|
+
const explanation = record.explanation;
|
|
396
|
+
if (explanation !== null && (typeof explanation !== "string" || explanation.length > MAX_EXPLANATION_CHARS)) {
|
|
397
|
+
throw new Error(`${source}.explanation: expected bounded string or null`);
|
|
398
|
+
}
|
|
399
|
+
return {
|
|
400
|
+
category: readCategory(record.category, `${source}.category`),
|
|
401
|
+
label,
|
|
402
|
+
ruleIds: record.ruleIds.map((id, index) => readId(id, `${source}.ruleIds[${index}]`)),
|
|
403
|
+
evidence: record.evidence.map((entry, index) => {
|
|
404
|
+
const evidence = asRecord(entry, `${source}.evidence[${index}]`);
|
|
405
|
+
return {
|
|
406
|
+
factId: readId(evidence.factId, `${source}.evidence[${index}].factId`),
|
|
407
|
+
source: readSource(evidence.source, `${source}.evidence[${index}].source`),
|
|
408
|
+
locator: readText(evidence.locator, `${source}.evidence[${index}].locator`, MAX_TEXT_CHARS),
|
|
409
|
+
digest: readDigest(evidence.digest, `${source}.evidence[${index}].digest`),
|
|
410
|
+
excerpt:
|
|
411
|
+
evidence.excerpt === null
|
|
412
|
+
? null
|
|
413
|
+
: readText(evidence.excerpt, `${source}.evidence[${index}].excerpt`, MAX_TEXT_CHARS),
|
|
414
|
+
};
|
|
415
|
+
}),
|
|
416
|
+
explanation: explanation as string | null,
|
|
417
|
+
};
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
function toEvidence(fact: EvalBehaviorJudgeFactV1): EvalBehaviorEvidenceV1 {
|
|
421
|
+
return { factId: fact.id, source: fact.source, ...fact.evidence };
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
function compare(actual: EvalBehaviorScalarV1, op: EvalBehaviorRuleOpV1, expected: EvalBehaviorScalarV1): boolean {
|
|
425
|
+
if (op === "eq") return actual === expected;
|
|
426
|
+
if (op === "neq") return actual !== expected;
|
|
427
|
+
if (typeof actual !== "number" || typeof expected !== "number") return false;
|
|
428
|
+
if (op === "lt") return actual < expected;
|
|
429
|
+
if (op === "lte") return actual <= expected;
|
|
430
|
+
if (op === "gt") return actual > expected;
|
|
431
|
+
return actual >= expected;
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
function parseSources(value: unknown, source: string): EvalBehaviorFactSourceV1[] {
|
|
435
|
+
if (!Array.isArray(value)) throw new Error(`${source}: expected array`);
|
|
436
|
+
return value.map((entry, index) => readSource(entry, `${source}[${index}]`));
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
function readOutcome(value: unknown, source: string): EvalBehaviorOutcomeV1 {
|
|
440
|
+
if (
|
|
441
|
+
value === "pass" ||
|
|
442
|
+
value === "behavioral_failure" ||
|
|
443
|
+
value === "unknown" ||
|
|
444
|
+
value === "unmeasured" ||
|
|
445
|
+
value === "infrastructure_failure"
|
|
446
|
+
)
|
|
447
|
+
return value;
|
|
448
|
+
throw new Error(`${source}: expected a closed behavioral outcome`);
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function readCategory(value: unknown, source: string): EvalBehaviorCategoryV1 {
|
|
452
|
+
if (typeof value === "string" && (EVAL_BEHAVIOR_CATEGORIES as readonly string[]).includes(value)) {
|
|
453
|
+
return value as EvalBehaviorCategoryV1;
|
|
454
|
+
}
|
|
455
|
+
throw new Error(`${source}: expected a closed behavioral category`);
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
function readSource(value: unknown, source: string): EvalBehaviorFactSourceV1 {
|
|
459
|
+
if (value === "transcript" || value === "tool" || value === "receipt" || value === "grader") return value;
|
|
460
|
+
throw new Error(`${source}: expected transcript, tool, receipt, or grader`);
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
function readOp(value: unknown, source: string): EvalBehaviorRuleOpV1 {
|
|
464
|
+
if (value === "eq" || value === "neq" || value === "lt" || value === "lte" || value === "gt" || value === "gte") {
|
|
465
|
+
return value;
|
|
466
|
+
}
|
|
467
|
+
throw new Error(`${source}: expected eq, neq, lt, lte, gt, or gte`);
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
function readScalar(value: unknown, source: string): EvalBehaviorScalarV1 {
|
|
471
|
+
if (typeof value === "boolean") return value;
|
|
472
|
+
if (typeof value === "number" && Number.isFinite(value)) return value;
|
|
473
|
+
if (typeof value === "string" && value.length <= MAX_TEXT_CHARS) return value;
|
|
474
|
+
throw new Error(`${source}: expected bounded string, finite number, or boolean`);
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
function readId(value: unknown, source: string): string {
|
|
478
|
+
const id = readText(value, source, MAX_ID_CHARS);
|
|
479
|
+
if (!/^[A-Za-z0-9._-]+$/u.test(id)) throw new Error(`${source}: expected stable id`);
|
|
480
|
+
return id;
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
function readText(value: unknown, source: string, maxChars: number): string {
|
|
484
|
+
if (typeof value !== "string" || value.trim().length === 0 || value.length > maxChars) {
|
|
485
|
+
throw new Error(`${source}: expected non-empty string no longer than ${maxChars} characters`);
|
|
486
|
+
}
|
|
487
|
+
return value;
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
function readDigest(value: unknown, source: string): string {
|
|
491
|
+
if (typeof value !== "string" || !/^[a-f0-9]{64}$/u.test(value)) throw new Error(`${source}: expected sha256 digest`);
|
|
492
|
+
return value;
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
function readBoundedInteger(value: unknown, source: string, min: number, max: number): number {
|
|
496
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < min || value > max) {
|
|
497
|
+
throw new Error(`${source}: expected integer from ${min} through ${max}`);
|
|
498
|
+
}
|
|
499
|
+
return value;
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
function asRecord(value: unknown, source: string): Record<string, unknown> {
|
|
503
|
+
if (typeof value === "object" && value !== null && !Array.isArray(value)) return value as Record<string, unknown>;
|
|
504
|
+
throw new Error(`${source}: expected object`);
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
function stableJson(value: unknown): string {
|
|
508
|
+
if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`;
|
|
509
|
+
if (typeof value === "object" && value !== null) {
|
|
510
|
+
return `{${Object.entries(value as Record<string, unknown>)
|
|
511
|
+
.sort(([left], [right]) => left.localeCompare(right))
|
|
512
|
+
.map(([key, entry]) => `${JSON.stringify(key)}:${stableJson(entry)}`)
|
|
513
|
+
.join(",")}}`;
|
|
514
|
+
}
|
|
515
|
+
return JSON.stringify(value);
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
function sha256(value: string): string {
|
|
519
|
+
return createHash("sha256").update(value, "utf8").digest("hex");
|
|
520
|
+
}
|