@iowarp/clio-coder 0.3.7 → 0.3.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/README.md +13 -4
- package/dist/{acp-SK4MD6MM.js → acp-7LOELQFP.js} +13 -13
- package/dist/{agents-2FN2K6ME.js → agents-FIBG2SHA.js} +41 -37
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-QIYZWM5I.js → auth-OI4LIH2I.js} +31 -24
- package/dist/builtins-AD25UL3C.js +17 -0
- package/dist/{chunk-5FR74PWO.js → chunk-2JDWVJND.js} +2 -2
- package/dist/chunk-3DPEIQKN.js +113 -0
- package/dist/{chunk-EOOQZZDE.js → chunk-3DUR4WUA.js} +19 -19
- package/dist/{chunk-WHJYKASB.js → chunk-3MRC2YSQ.js} +2 -2
- package/dist/{chunk-EBEFWSGL.js → chunk-3UUY7R3Z.js} +14 -10
- package/dist/{chunk-LADCF22A.js → chunk-3V5AYSEQ.js} +113 -54
- package/dist/{chunk-BMWK7ZIZ.js → chunk-465CC7FK.js} +16 -13
- package/dist/{chunk-5WIGXA4T.js → chunk-47CMYGET.js} +111 -4
- package/dist/{chunk-CEYBNUGC.js → chunk-4H6ULJ3H.js} +378 -36
- package/dist/{chunk-YTYFXUI3.js → chunk-4LJX2PUC.js} +9 -9
- package/dist/{chunk-DOOEX22V.js → chunk-56KB5IJP.js} +5 -5
- package/dist/{chunk-SPULKLCF.js → chunk-5DQRIYDZ.js} +2 -2
- package/dist/{chunk-TSHXZTOQ.js → chunk-5HFBWUMU.js} +23 -11
- package/dist/{chunk-5UJ6ECTS.js → chunk-5PVQ4SRS.js} +80 -8
- package/dist/{chunk-ZWLZP4ZT.js → chunk-5QKCQQ3E.js} +359 -17
- package/dist/{chunk-6M7VS3J3.js → chunk-5T7RBWN2.js} +111 -5
- package/dist/chunk-774ILSRL.js +172 -0
- package/dist/chunk-7C6RYZGQ.js +391 -0
- package/dist/{chunk-GH5622CP.js → chunk-A2NJGIB3.js} +2 -2
- package/dist/{chunk-C4JBQ5SR.js → chunk-AD7Y7STJ.js} +6 -6
- package/dist/{chunk-GEYXPTRF.js → chunk-AEYBF3TB.js} +33 -12
- package/dist/{chunk-2SFS6XQE.js → chunk-AMKHQW3C.js} +3 -2
- package/dist/{chunk-D4MDIG46.js → chunk-B5CSFE7B.js} +7 -7
- package/dist/{chunk-MXI6J5JF.js → chunk-B5XRQOLB.js} +10 -10
- package/dist/{chunk-X2KV5FXT.js → chunk-BVDVID7E.js} +2 -2
- package/dist/{chunk-JNXPYBB4.js → chunk-CA42X6KT.js} +3 -3
- package/dist/{chunk-VREKEFLL.js → chunk-D73KXYPF.js} +3 -3
- package/dist/{chunk-JTSEDYVQ.js → chunk-DG4M6ZUE.js} +7 -7
- package/dist/{chunk-DQA7QLMD.js → chunk-EBOC7MT3.js} +10 -25
- package/dist/{chunk-KZ2H5X4G.js → chunk-ECUO3KDP.js} +129 -14
- package/dist/{chunk-JRIO5UD2.js → chunk-EQ63NRB7.js} +5 -5
- package/dist/{chunk-YD734TPH.js → chunk-FALJGAWU.js} +2 -2
- package/dist/{chunk-GWS3VEIW.js → chunk-FWDFM5ZU.js} +24 -3
- package/dist/{chunk-XEGB6BCN.js → chunk-GAYUJ7LE.js} +68 -14
- package/dist/{chunk-UND3GU2L.js → chunk-H7IXIC72.js} +2 -2
- package/dist/{chunk-IR4CFBFN.js → chunk-HAY4ZE2P.js} +12 -12
- package/dist/{chunk-UVDSQ6LW.js → chunk-HCBCAYZU.js} +74 -147
- package/dist/{chunk-4DWFMQDR.js → chunk-HJB5IUKP.js} +89 -145
- package/dist/{chunk-M4AKACEO.js → chunk-HKO36JWF.js} +33 -5
- package/dist/{chunk-KCMKRQX4.js → chunk-HPCTNZM2.js} +45 -82
- package/dist/{chunk-465YSENW.js → chunk-IFBNV6H6.js} +3 -3
- package/dist/{chunk-FJ3H4MN5.js → chunk-IHKBWSXF.js} +2 -2
- package/dist/chunk-JEQQR47K.js +3025 -0
- package/dist/{chunk-FO5ZOVUY.js → chunk-KV2AOLDF.js} +27 -7
- package/dist/chunk-LU7P4LHA.js +33 -0
- package/dist/{chunk-6TUKSZVF.js → chunk-LXPJXFM5.js} +11 -11
- package/dist/{chunk-VQNODYQ4.js → chunk-MIX5N5AC.js} +488 -3668
- package/dist/chunk-MLOK6ZOS.js +2888 -0
- package/dist/{chunk-ZZMN5OM4.js → chunk-MV2VUEJC.js} +2 -2
- package/dist/{chunk-MVVUPGPW.js → chunk-MXHC5QYU.js} +6 -6
- package/dist/{chunk-OBMAI2DP.js → chunk-N3PBVRTZ.js} +12 -388
- package/dist/{chunk-WJHBC77E.js → chunk-N5XKWMDW.js} +17 -7
- package/dist/{chunk-5C3AQNDW.js → chunk-NNNWO6F2.js} +124 -36
- package/dist/{chunk-UFQ3F4FW.js → chunk-NQ6UCCOD.js} +4 -4
- package/dist/chunk-NUGM5KR6.js +165 -0
- package/dist/{chunk-DMD2AGVS.js → chunk-NZU6YDNV.js} +20 -18
- package/dist/{chunk-WHGPSPT5.js → chunk-O6I4CIEU.js} +151 -13
- package/dist/{chunk-XN3L4EYL.js → chunk-OEDBCISO.js} +2 -2
- package/dist/{chunk-PD3MESLB.js → chunk-P3JGPQFL.js} +4 -4
- package/dist/{chunk-UHXRNZ2J.js → chunk-PNY46YEY.js} +23 -6
- package/dist/{chunk-THKY7CD7.js → chunk-PZ4I4JE2.js} +134 -29
- package/dist/{chunk-SROCI7ZU.js → chunk-QQ7EKM72.js} +5 -5
- package/dist/{chunk-QCTRSGHQ.js → chunk-R7LNVMCS.js} +91 -53
- package/dist/{chunk-GOXNB3AO.js → chunk-RAPCMZL4.js} +75 -4
- package/dist/chunk-RKKLTLYB.js +45 -0
- package/dist/{chunk-OB5HIGJY.js → chunk-RKRLDWD3.js} +4 -1
- package/dist/{chunk-DJNLUABN.js → chunk-S4COXYBG.js} +588 -32
- package/dist/{chunk-3HAPLH5M.js → chunk-T3Z6VAAF.js} +172 -11
- package/dist/{chunk-FOT2FX5J.js → chunk-TD7UE2L5.js} +12 -10
- package/dist/{chunk-UUANF5CR.js → chunk-TEO2TLVN.js} +856 -967
- package/dist/{chunk-FCSXB6T2.js → chunk-UOSL25KY.js} +14 -2
- package/dist/{chunk-EELBMBT6.js → chunk-VKBMFOYV.js} +74 -15
- package/dist/chunk-VO2LKSTM.js +165 -0
- package/dist/{chunk-5C77SEEY.js → chunk-VPTUJU4P.js} +3 -3
- package/dist/{chunk-WEH5XRJQ.js → chunk-WIE7ZOSW.js} +2 -2
- package/dist/{chunk-J7PIKKWC.js → chunk-WXCJ7VME.js} +8 -8
- package/dist/{chunk-4DGYLA73.js → chunk-XDOQXGFO.js} +22 -7
- package/dist/{chunk-PPAMZ32Z.js → chunk-XK56QHLX.js} +6 -1
- package/dist/{chunk-AB4XIIVB.js → chunk-YKOFT37S.js} +6 -6
- package/dist/chunk-YSEHGPCT.js +127 -0
- package/dist/{chunk-HFSBBKSQ.js → chunk-YW7UVM5V.js} +138 -3
- package/dist/cli/index.js +32 -32
- package/dist/{clio-WBVQEBKO.js → clio-LT5V7SSZ.js} +9 -9
- package/dist/{code-nav-FGGFIE7L.js → code-nav-LMW275PA.js} +5 -5
- package/dist/codewiki/build-worker.js +4 -4
- package/dist/{components-F7OEATSO.js → components-ZFA3SAER.js} +8 -8
- package/dist/{config-TRBL3RCF.js → config-RXS5T3JT.js} +98 -65
- package/dist/{configure-OLCVPHNM.js → configure-2WYWSCSD.js} +26 -22
- package/dist/{context-MJIJ6GOX.js → context-I3BTOTCS.js} +12 -12
- package/dist/{context-XEWE3MOJ.js → context-MVOORGMF.js} +54 -47
- package/dist/{context-WFPKQSM6.js → context-PALKKQYL.js} +28 -28
- package/dist/{context-clear-KNOS2JPB.js → context-clear-N2WOYZ2K.js} +53 -46
- package/dist/{context-index-SSR5ECNE.js → context-index-HNG3MOME.js} +6 -6
- package/dist/{context-working-set-EUXAZI6N.js → context-working-set-MIEVECVZ.js} +17 -18
- package/dist/{dispatch-runner-B7MTOVKL.js → dispatch-runner-VVA4SRRH.js} +90 -61
- package/dist/{docs-FLJTIDSE.js → docs-7LQ23DLM.js} +8 -8
- package/dist/doctor-TWBWFK5V.js +165 -0
- package/dist/eval-IJ5VEZDJ.js +4483 -0
- package/dist/{evidence-JZNBUOQZ.js → evidence-L5APPXNV.js} +68 -61
- package/dist/{evolve-FJVC4KKI.js → evolve-RGNKFJ52.js} +47 -40
- package/dist/{extensions-IQL36S7K.js → extensions-7WYWUX5A.js} +13 -7
- package/dist/{fleet-BDKYJFCP.js → fleet-6CNVBZZP.js} +113 -76
- package/dist/{fleet-commands-ZFIWZSB3.js → fleet-commands-L2SXSYEI.js} +10 -10
- package/dist/{fleet-graph-Y6HPXIVF.js → fleet-graph-2J3OOIPO.js} +17 -15
- package/dist/{fleet-preflight-BHSNPBMH.js → fleet-preflight-CZRJ4JP5.js} +5 -6
- package/dist/{fleet-validate-BIYREGIK.js → fleet-validate-C5RI6DP7.js} +20 -19
- package/dist/{init-LQUB5COQ.js → init-VBN2ACVA.js} +70 -63
- package/dist/{library-NJAHIGG4.js → library-JHGUMLY2.js} +22 -20
- package/dist/{memory-OG6HOYKM.js → memory-K4OQIYWG.js} +49 -42
- package/dist/{models-5ZG5XY7J.js → models-2NCZUWDD.js} +35 -29
- package/dist/{monitor-TJ7AMTGB.js → monitor-MMVTJABD.js} +64 -45
- package/dist/{orchestrator-WZYB54DM.js → orchestrator-ZKBPCHW6.js} +1971 -520
- package/dist/{paths-XUC7GS6E.js → paths-DBXMZMDU.js} +5 -5
- package/dist/registry-LG64LTF4.js +11 -0
- package/dist/{reset-PXQT45IY.js → reset-DD5JGOY3.js} +11 -11
- package/dist/{run-FQ74YF62.js → run-QEGNX7FL.js} +89 -83
- package/dist/{share-FW7SVCL3.js → share-JKD3BQMW.js} +20 -18
- package/dist/{skills-7E7IRB3R.js → skills-LMQIKDOZ.js} +23 -21
- package/dist/{skills-eval-LI75W6OK.js → skills-eval-I7X2774U.js} +59 -52
- package/dist/{steer-GGWFUJUD.js → steer-CF5TDANS.js} +3 -3
- package/dist/support-I7LOJLIF.js +38 -0
- package/dist/{targets-4CIFKCTW.js → targets-RUSR6B5Z.js} +77 -42
- package/dist/{terminal-lease-WUZY7ZV5.js → terminal-lease-QYVORFR4.js} +6 -4
- package/dist/{trace-PNCASAXC.js → trace-ODOQIVIW.js} +61 -6
- package/dist/{uninstall-7FV7IP4E.js → uninstall-ZJF5H5ZN.js} +8 -8
- package/dist/{upgrade-K2HVIVMQ.js → upgrade-XANW3FXB.js} +29 -26
- package/dist/{usage-GTZELZQX.js → usage-4H7ZRXQT.js} +110 -61
- package/dist/{verifiers-RLAHT27O.js → verifiers-UZXNBZEB.js} +13 -13
- package/dist/{verify-BX3BRKH5.js → verify-BVKWTNDL.js} +9 -9
- package/dist/{wiki-generate-ASIFASCN.js → wiki-generate-MY7WV2QI.js} +76 -69
- package/dist/worker/entry.js +69 -66
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +1 -1
- package/docs/artifact-versions.md +11 -5
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +23 -2
- package/docs/commands-and-modes.md +2 -2
- package/docs/configuration-and-targets.md +37 -5
- package/docs/context-engine.md +63 -4
- package/docs/documentation-coverage.md +3 -3
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +2 -0
- package/docs/eval-runner.md +262 -11
- package/docs/evals-internal.md +72 -2
- package/docs/evidence-and-memory.md +77 -12
- package/docs/evolution.md +1 -1
- package/docs/extensions-and-sharing.md +3 -1
- package/docs/fleet-dispatch.md +34 -9
- package/docs/glossary.md +21 -1
- package/docs/installation-and-lifecycle.md +1 -1
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +1 -1
- package/docs/observability.md +54 -3
- package/docs/proactive-memory.md +127 -14
- package/docs/prompt-envelope-and-tools.md +22 -2
- package/docs/provider-adapter-cookbook.md +1 -1
- package/docs/release-cut-checklist.md +60 -41
- package/docs/safety-model.md +1 -1
- package/docs/scientific-validation.md +1 -1
- package/docs/skills-marketplace.md +1 -1
- package/docs/tool-usage.md +1 -1
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +87 -0
- package/docs/tui-design.md +1 -1
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +2 -2
- package/src/cli/agents.ts +1 -1
- package/src/cli/argv.ts +5 -0
- package/src/cli/config-inspect.ts +33 -6
- package/src/cli/config.ts +1 -1
- package/src/cli/configure.ts +107 -23
- package/src/cli/doctor-state-size.ts +82 -0
- package/src/cli/doctor.ts +7 -1
- package/src/cli/eval.ts +80 -16
- package/src/cli/evidence.ts +30 -25
- package/src/cli/extensions.ts +5 -1
- package/src/cli/fleet-preflight.ts +2 -12
- package/src/cli/fleet.ts +32 -3
- package/src/cli/shared.ts +1 -0
- package/src/cli/targets.ts +45 -11
- package/src/cli/trace.ts +63 -4
- package/src/cli/usage.ts +63 -14
- package/src/cli/validate-model.ts +60 -5
- package/src/core/bus-events.ts +54 -1
- package/src/core/cache-telemetry.ts +42 -0
- package/src/core/commit-attribution.ts +4 -4
- package/src/core/config.ts +18 -0
- package/src/core/defaults.ts +36 -6
- package/src/core/endpoint-key.ts +27 -0
- package/src/core/path-boundary.ts +100 -0
- package/src/core/residency-target-key.ts +25 -0
- package/src/core/response-schema.ts +36 -2
- package/src/domains/agents/extension.ts +2 -11
- package/src/domains/agents/fleet-contract.ts +30 -12
- package/src/domains/agents/recipe.ts +7 -1
- package/src/domains/agents/registry.ts +73 -5
- package/src/domains/agents/result-contract.ts +128 -17
- package/src/domains/agents/write-boundary.ts +15 -50
- package/src/domains/config/classify.ts +3 -0
- package/src/domains/context/codewiki/coordinator.ts +12 -4
- package/src/domains/context/project-rules.ts +51 -1
- package/src/domains/dispatch/admission.ts +40 -3
- package/src/domains/dispatch/assignment-reconcile.ts +22 -5
- package/src/domains/dispatch/assignment-store.ts +151 -14
- package/src/domains/dispatch/capacity-lease.ts +98 -9
- package/src/domains/dispatch/contract.ts +26 -1
- package/src/domains/dispatch/delegation-plan.ts +2 -5
- package/src/domains/dispatch/execution-plan.ts +44 -4
- package/src/domains/dispatch/execution-role.ts +9 -1
- package/src/domains/dispatch/extension.ts +309 -96
- package/src/domains/dispatch/fleet-run.ts +78 -4
- package/src/domains/dispatch/gate-role-prompts.ts +38 -0
- package/src/domains/dispatch/heartbeat.ts +32 -8
- package/src/domains/dispatch/index.ts +6 -1
- package/src/domains/dispatch/intent-requirements.ts +40 -0
- package/src/domains/dispatch/intent.ts +84 -8
- package/src/domains/dispatch/orphan-recovery.ts +5 -0
- package/src/domains/dispatch/path-scope.ts +370 -0
- package/src/domains/dispatch/receipt-integrity.ts +2 -1
- package/src/domains/dispatch/reservation-store.ts +116 -8
- package/src/domains/dispatch/state.ts +4 -0
- package/src/domains/dispatch/types.ts +14 -7
- package/src/domains/dispatch/validation.ts +6 -3
- package/src/domains/dispatch/worker-spawn.ts +25 -11
- package/src/domains/dispatch/write-boundary-enforcer.ts +62 -0
- package/src/domains/dispatch/write-boundary.ts +262 -22
- package/src/domains/eval/artifacts/store.ts +62 -0
- package/src/domains/eval/compare/behavioral.ts +224 -0
- package/src/domains/eval/compare/compare.ts +355 -2
- package/src/domains/eval/compare/envelope.ts +128 -0
- package/src/domains/eval/compare/gates.ts +24 -6
- package/src/domains/eval/compare/thresholds.ts +30 -3
- package/src/domains/eval/execution-provenance.ts +240 -0
- package/src/domains/eval/metrics/aggregate.ts +136 -0
- package/src/domains/eval/metrics/call-ledger-stream.ts +112 -0
- package/src/domains/eval/metrics/evidence.ts +79 -2
- package/src/domains/eval/metrics/tracked.ts +413 -0
- package/src/domains/eval/provenance.ts +117 -0
- package/src/domains/eval/reports/comparison.ts +128 -0
- package/src/domains/eval/reports/junit.ts +17 -3
- package/src/domains/eval/reports/markdown.ts +3 -3
- package/src/domains/eval/reports/text.ts +14 -0
- package/src/domains/eval/run-compare.ts +20 -0
- package/src/domains/eval/runners/clio-run.ts +139 -2
- package/src/domains/eval/runners/external-command.ts +28 -3
- package/src/domains/eval/schema/adapter.ts +111 -0
- package/src/domains/eval/schema/artifact.ts +20 -0
- package/src/domains/eval/schema/behavioral-metrics.ts +204 -0
- package/src/domains/eval/schema/behavioral.ts +520 -0
- package/src/domains/eval/schema/execution-envelope.ts +194 -0
- package/src/domains/eval/schema/serving.ts +74 -0
- package/src/domains/eval/schema/suite.ts +38 -8
- package/src/domains/eval/schema/validate.ts +58 -3
- package/src/domains/eval/schema/verdict.ts +237 -0
- package/src/domains/eval/suites/resolve.ts +2 -0
- package/src/domains/eval/suites/run.ts +264 -33
- package/src/domains/eval/verifiers/command.ts +2 -1
- package/src/domains/eval/workspaces/temp-copy.ts +145 -13
- package/src/domains/evidence/build.ts +68 -21
- package/src/domains/evidence/eval.ts +2 -12
- package/src/domains/evidence/findings-markdown.ts +33 -0
- package/src/domains/evidence/index.ts +21 -0
- package/src/domains/evidence/provenance.ts +46 -11
- package/src/domains/evidence/run-trust.ts +7 -113
- package/src/domains/evidence/trust-projection.ts +274 -0
- package/src/domains/evidence/trust-status.ts +145 -17
- package/src/domains/evidence/types.ts +4 -0
- package/src/domains/extensions/compatibility.ts +285 -0
- package/src/domains/extensions/discovery.ts +126 -4
- package/src/domains/extensions/resources.ts +21 -9
- package/src/domains/extensions/state.ts +18 -5
- package/src/domains/extensions/types.ts +6 -1
- package/src/domains/lifecycle/doctor.ts +209 -2
- package/src/domains/memory/index.ts +14 -0
- package/src/domains/memory/task-bank-promotion.ts +64 -0
- package/src/domains/memory/task-memory-policy.ts +77 -8
- package/src/domains/memory/task-memory-spend.ts +131 -0
- package/src/domains/memory/task-memory-status.ts +7 -0
- package/src/domains/memory/task-memory-telemetry.ts +2 -0
- package/src/domains/middleware/index.ts +1 -0
- package/src/domains/middleware/memory-intervention.ts +69 -5
- package/src/domains/middleware/memory-step-endpoint.ts +71 -0
- package/src/domains/observability/background-memory-usage.ts +140 -0
- package/src/domains/observability/cost.ts +1 -1
- package/src/domains/observability/index.ts +7 -0
- package/src/domains/observability/out-of-turn-usage.ts +51 -2
- package/src/domains/observability/trace-store.ts +192 -2
- package/src/domains/prompts/compiler.ts +100 -13
- package/src/domains/prompts/contract.ts +3 -5
- package/src/domains/providers/endpoint-capacity.ts +96 -0
- package/src/domains/providers/extension.ts +30 -2
- package/src/domains/providers/index.ts +10 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +243 -1
- package/src/domains/providers/runtime-resolution.ts +8 -1
- package/src/domains/providers/runtimes/common/probe-helpers.ts +31 -9
- package/src/domains/providers/runtimes/local-native/llamacpp-anthropic.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-completion.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-embed.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp-rerank.ts +1 -1
- package/src/domains/providers/runtimes/local-native/llamacpp.ts +4 -1
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +4 -1
- package/src/domains/providers/runtimes/local-native/ollama-native.ts +6 -1
- package/src/domains/providers/types/capability-flags.ts +2 -0
- package/src/domains/providers/types/target-descriptor.ts +2 -0
- package/src/domains/resources/common-loader.ts +3 -0
- package/src/domains/resources/prompts/loader.ts +184 -17
- package/src/domains/safety/call-target.ts +52 -0
- package/src/domains/safety/policy-engine.ts +5 -5
- package/src/domains/safety/run-effects.ts +96 -2
- package/src/domains/safety/scope.ts +7 -12
- package/src/domains/session/context-accounting.ts +52 -1
- package/src/domains/session/context-ledger.ts +37 -13
- package/src/domains/session/index.ts +6 -0
- package/src/domains/session/prompt-cache.ts +140 -0
- package/src/domains/session/prompt-manifest.ts +42 -0
- package/src/engine/acp/adapter.ts +18 -3
- package/src/engine/acp/server.ts +4 -1
- package/src/engine/ai.ts +35 -0
- package/src/engine/apis/llamacpp-residency.ts +55 -3
- package/src/engine/apis/lmstudio.ts +25 -5
- package/src/engine/apis/ollama-native.ts +2 -1
- package/src/engine/apis/openai-completions.ts +80 -17
- package/src/engine/apis/residency-lock.ts +3 -1
- package/src/engine/apis/residency.ts +34 -1
- package/src/engine/prompt-templates.ts +18 -1
- package/src/engine/provider-payload.ts +29 -1
- package/src/engine/worker-runtime.ts +6 -3
- package/src/entry/orchestrator.ts +176 -30
- package/src/interactive/chat-loop-messages.ts +26 -7
- package/src/interactive/chat-loop.ts +318 -41
- package/src/interactive/chat-panel.ts +62 -8
- package/src/interactive/clio-editor.ts +45 -8
- package/src/interactive/context-activity.ts +5 -1
- package/src/interactive/context-meter.ts +1 -1
- package/src/interactive/context-overlay.ts +40 -10
- package/src/interactive/cost-overlay.ts +64 -6
- package/src/interactive/dispatch-board.ts +127 -5
- package/src/interactive/fleet-run-preview.ts +41 -15
- package/src/interactive/handoff-round.ts +41 -2
- package/src/interactive/interactive-application.ts +24 -1
- package/src/interactive/interactive-event-projection.ts +14 -0
- package/src/interactive/interactive-input-runtime.ts +8 -0
- package/src/interactive/interactive-presentation.ts +4 -0
- package/src/interactive/interactive-shell.ts +20 -17
- package/src/interactive/interactive-slash-runtime.ts +27 -4
- package/src/interactive/memory-overlay.ts +8 -0
- package/src/interactive/mutation-preview.ts +295 -0
- package/src/interactive/overlay-general-openers.ts +16 -0
- package/src/interactive/overlay-key-routing.ts +38 -0
- package/src/interactive/overlay-lifecycle.ts +38 -5
- package/src/interactive/overlay-permission-lifecycle.ts +22 -2
- package/src/interactive/overlay-session-lifecycle.ts +73 -9
- package/src/interactive/overlays/ask-user.ts +91 -19
- package/src/interactive/overlays/help-reference.ts +4 -0
- package/src/interactive/overlays/prompts.ts +11 -1
- package/src/interactive/overlays/settings.ts +176 -48
- package/src/interactive/permission-hint.ts +34 -2
- package/src/interactive/permission-overlay.ts +159 -9
- package/src/interactive/prewarm.ts +197 -0
- package/src/interactive/render-trace.ts +162 -15
- package/src/interactive/renderers/tool-execution.ts +4 -0
- package/src/interactive/side-question.ts +58 -1
- package/src/interactive/slash-commands.ts +7 -2
- package/src/interactive/status/controller.ts +11 -0
- package/src/interactive/status/state-machine.ts +54 -2
- package/src/interactive/status/types.ts +7 -0
- package/src/interactive/terminal-lease.ts +2 -0
- package/src/interactive/turn-context.ts +299 -31
- package/src/interactive/turn-persistence.ts +14 -4
- package/src/interactive/turn-prewarm.ts +364 -0
- package/src/interactive/turn-queues.ts +7 -4
- package/src/interactive/turn-runtime.ts +8 -1
- package/src/interactive/turn-state.ts +23 -0
- package/src/interactive/view/artifacts.ts +42 -9
- package/src/interactive/view/view-overlay.ts +43 -6
- package/src/interactive/worker-receipts.ts +14 -2
- package/src/interactive/worker-stream.ts +8 -0
- package/src/tools/ask-user.ts +43 -2
- package/src/tools/dispatch-admission.ts +12 -13
- package/src/tools/dispatch-arguments.ts +27 -0
- package/src/tools/dispatch-plan.ts +46 -9
- package/src/tools/dispatch-runner.ts +48 -13
- package/src/tools/dispatch-scout.ts +1 -1
- package/src/tools/monitor.ts +13 -0
- package/src/tools/registry.ts +16 -0
- package/src/tools/worker-evidence.ts +19 -13
- package/src/worker/spec-contract.ts +2 -1
- package/dist/chunk-AOCYTWAV.js +0 -449
- package/dist/chunk-HWUFFB6L.js +0 -83
- package/dist/chunk-R346GLFC.js +0 -31
- package/dist/chunk-ZGH7FGS5.js +0 -1079
- package/dist/doctor-RN4YKO2X.js +0 -87
- package/dist/eval-RUBJVSNQ.js +0 -2557
|
@@ -11,7 +11,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
|
|
|
11
11
|
| `src/engine/` (Core) | Engine turn loop, prompt priming, streaming message adapters, turn execution | [architecture.md](architecture.md), [context-engine.md](context-engine.md) | `documented` | Documented across architecture and context engine guides. |
|
|
12
12
|
| `src/engine/acp/` | ACP protocol server, transport adapters, tool mediators, permission forwarding, error taxonomy | [acp.md](acp.md) | `documented` | Dedicated ACP specification covering server wiring, permission mediation, timeouts, error taxonomy, and security boundaries. |
|
|
13
13
|
| `src/entry/` | Application bootstrapping, CLI router, interactive loop entry point | [architecture.md](architecture.md), [installation-and-lifecycle.md](installation-and-lifecycle.md) | `documented` | Documented in architecture compilation boundaries and lifecycle guides. |
|
|
14
|
-
| `src/interactive/` | TUI architecture, screens, overlays, keybindings, panels, theme tokens, width matrices | [tui-design.md](tui-design.md), [commands-and-modes.md](commands-and-modes.md) | `documented` | Fully documented in TUI design specification and commands reference. |
|
|
14
|
+
| `src/interactive/` | TUI architecture, screens, overlays, keybindings, panels, theme tokens, width matrices, prompt pre-warm rounds and their gating, expected-cold reason stamping | [tui-design.md](tui-design.md), [commands-and-modes.md](commands-and-modes.md), [context-engine.md](context-engine.md) | `documented` | Fully documented in TUI design specification and commands reference; the pre-warm and the cache-honesty surfaces `/context` renders are in the context engine reference. |
|
|
15
15
|
| `src/tools/` | 20 built-in tools across 7 planes, registry, policy engine bindings, observation envelope bounds | [tool-usage.md](tool-usage.md), [prompt-envelope-and-tools.md](prompt-envelope-and-tools.md) | `documented` | Comprehensive 20-tool reference with schemas, examples, and envelope size constraints. |
|
|
16
16
|
| `src/utils/` | Image manipulation, photon operations, git execution utilities | [architecture.md](architecture.md), [tool-usage.md](tool-usage.md) | `documented` | Utility helpers documented within tool usage and architectural boundaries. |
|
|
17
17
|
| `src/worker/` | Worker subprocess lifecycle, NDJSON transport, heartbeat timers, control lane demuxing, spec contracts | [worker-dispatch-mechanics.md](worker-dispatch-mechanics.md) | `documented` | Complete reference for NDJSON socket protocols, watchdog timers, and exit status mapping. |
|
|
@@ -20,7 +20,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
|
|
|
20
20
|
| `src/domains/config/` | Configuration contracts, file watcher, keybinding definitions, setting classifiers | [configuration-and-targets.md](configuration-and-targets.md), [commands-and-modes.md](commands-and-modes.md) | `documented` | Documented in configuration targets and command/keybinding reference. |
|
|
21
21
|
| `src/domains/context/` | `CLIO-CODER.md` bootstrap, codewiki generation, prompt context assembly, project rules, non-destructive working-set eviction (`age-horizon` and `structural-v1` policies, protection predicates, path index, byte-stable markers, recall by ref) | [context-engine.md](context-engine.md), [context-working-set.md](context-working-set.md) | `documented` | Context window, token accounting, and the three compaction mechanisms in the engine reference; the working-set layer has its own guide covering the vocabulary, both ledger record kinds and format v4, the marker contract, both policies with their rule order, recall semantics, and the operator surfaces. |
|
|
22
22
|
| `src/domains/dispatch/` | Fleet orchestration, assignment store, batch tracker, admission, route planner, receipt integrity v16 | [fleet-dispatch.md](fleet-dispatch.md), [dispatch-architecture-rationale.md](dispatch-architecture-rationale.md), [worker-dispatch-mechanics.md](worker-dispatch-mechanics.md) | `documented` | Multi-node fleet dispatch, admission invariants, and receipt verification fully documented. |
|
|
23
|
-
| `src/domains/eval/` | Suite v2 YAML schema, eval runner, metrics, reporters, workspace sandboxing | [eval-runner.md](eval-runner.md), [evals-internal.md](evals-internal.md) | `documented` | Product evals are documented independently from external benchmarks. |
|
|
23
|
+
| `src/domains/eval/` | Suite v2 YAML schema, eval runner, `clio.eval.verdict.v1` envelope and its Suite v2 adapter, tracked metrics and scenario aggregates, serving-configuration provenance, reporters, workspace sandboxing | [eval-runner.md](eval-runner.md), [evals-internal.md](evals-internal.md) | `documented` | Product evals are documented independently from external benchmarks. The verdict envelope, `trackedMetrics` and their sources, `--trials`, and the config-drift and estimated-versus-measured refusals are in the runner reference. |
|
|
24
24
|
| `src/domains/evidence/` | Evidence bundles, findings taxonomy, provenance store, failure attribution | [evidence-and-memory.md](evidence-and-memory.md) | `documented` | Documented in evidence directory structures and memory retrieval guide. |
|
|
25
25
|
| `src/domains/evolution/` | Falsifiable Change Manifest JSON templates and `clio-coder evolve` self-edit gates | [evolution.md](evolution.md) | `documented` | Documented in evolution manifest reference and mutation validation rules. |
|
|
26
26
|
| `src/domains/extensions/` | Extension manifest schemas, resource roots, portable share archives | [extensions-and-sharing.md](extensions-and-sharing.md) | `documented` | Documented in extensions and sharing guide. |
|
|
@@ -29,7 +29,7 @@ This matrix maps every top-level directory in `src/` and every domain directory
|
|
|
29
29
|
| `src/domains/middleware/` | Middleware hooks (`turn_start`, `tool_call`, `tool_result`, `turn_end`), reminders, budgets | [middleware-and-components.md](middleware-and-components.md) | `documented` | Documented in middleware hooks and active component snapshot guide. |
|
|
30
30
|
| `src/domains/observability/` | Trace store (`node:sqlite` WAL mirror), metrics, cost accounting, evidence index | [trace-store.md](trace-store.md), [observability.md](observability.md) | `documented` | Database schema, rowid cursor queries, and receipt provenance documented. |
|
|
31
31
|
| `src/domains/prompts/` | Prompt compiler, fragment loaders, static cache stability, memory intervention injection | [prompt-envelope-and-tools.md](prompt-envelope-and-tools.md) | `documented` | Documented in prompt envelope and tool delivery guide. |
|
|
32
|
-
| `src/domains/providers/` | Runtime adapters, capability probes, model catalog, thinking control, ALCF OAuth | [configuration-and-targets.md](configuration-and-targets.md), [model-catalog.md](model-catalog.md), [provider-adapter-cookbook.md](provider-adapter-cookbook.md), [alcf-provider.md](alcf-provider.md) | `documented` | Complete provider adapter contracts, model catalogs, and ALCF Globus targets documented. |
|
|
32
|
+
| `src/domains/providers/` | Runtime adapters, capability probes, model catalog and its per-family `measuredUnder` provenance, canonical endpoint keys and per-endpoint request-slot capacity, thinking control, ALCF OAuth | [configuration-and-targets.md](configuration-and-targets.md), [model-catalog.md](model-catalog.md), [provider-adapter-cookbook.md](provider-adapter-cookbook.md), [alcf-provider.md](alcf-provider.md) | `documented` | Complete provider adapter contracts, model catalogs, and ALCF Globus targets documented. |
|
|
33
33
|
| `src/domains/resources/` | Skill package discovery, marketplace index resolution, prompt resources | [skills-marketplace.md](skills-marketplace.md), [extensions-and-sharing.md](extensions-and-sharing.md) | `documented` | Skills marketplace, publishing flows, and resource managers documented. |
|
|
34
34
|
| `src/domains/safety/` | Policy engine, action classifiers, damage-control rules, path policies, finish contract, audit log | [safety-model.md](safety-model.md), [scientific-validation.md](scientific-validation.md) | `documented` | Policy evaluation order, 10-step sequence, write containment, and finish contract documented. |
|
|
35
35
|
| `src/domains/scheduling/` | Capacity lease acquisition, heartbeats, expiry, cross-process locks, cluster scheduling | [capacity-and-scheduling.md](capacity-and-scheduling.md), [fleet-dispatch.md](fleet-dispatch.md) | `documented` | Dedicated capacity leasing, heartbeat TTL, and cross-process lock reference. |
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Documentation Standards and Codebase Alignment
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive documentation link linter, phrasing/claim evaluator, and alignment portal is located at [docs/html/documentation_blueprint.html](html/documentation_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive documentation link linter, phrasing/claim evaluator, and alignment portal is located at [docs/html/documentation_blueprint.html](html/documentation_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Clio Coder is an experimental community alpha. Documentation should help contributors and early users work from the source of truth without overstating maturity. When docs drift, prefer the current source and tests over older prose or aspirational roadmap notes.
|
|
7
7
|
|
|
@@ -47,6 +47,8 @@ Durable values live in the `guardrails:` section of settings.yaml (see [configur
|
|
|
47
47
|
| `CLIO_CODER_REDUCE_MOTION` | off | `1` makes smooth-streaming `auto` use the immediate coalescer. Explicit `on` remains an operator request, while stdout backpressure still pauses frame production. |
|
|
48
48
|
| `CLIO_CODER_SCREEN_READER` | off | `1` makes smooth-streaming `auto` use the immediate coalescer so a screen reader receives the existing low-motion update behavior. |
|
|
49
49
|
| `CLIO_CODER_INSTANT_SHELL` | on | `0` disables the single-owner Stage 0 interactive shell for immediate rollback. Unset or `1` mounts one terminal/editor owner before service hydration; ACP, headless, ordinary non-TTY, and subcommand paths never mount it. An explicit `CLIO_CODER_INTERACTIVE=1` keeps its force-interactive non-TTY behavior. |
|
|
50
|
+
| `CLIO_CODER_TRACE_RETENTION_DAYS` | 30 | Maximum age in days for terminal rows in the rebuildable SQLite trace mirror. The value is an integer of at least 1 (`src/domains/observability/trace-store.ts`). |
|
|
51
|
+
| `CLIO_CODER_TRACE_MAX_BYTES` | 134217728 | Maximum allocated size for the SQLite trace mirror before the oldest terminal runs are pruned. The value is an integer of at least 1,048,576 (`src/domains/observability/trace-store.ts`). |
|
|
50
52
|
|
|
51
53
|
## Directory and install layout
|
|
52
54
|
|
package/docs/eval-runner.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Clio Coder Local Evaluation Runner
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive task suite validator, subprocess execution simulator, and compare calculator is located at [docs/html/eval_blueprint.html](html/eval_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive task suite validator, subprocess execution simulator, and compare calculator is located at [docs/html/eval_blueprint.html](html/eval_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
The local evaluation runner executes repository-local YAML task suites as deterministic subprocess checks. It is useful for comparing harness changes, prompts, tools, or local workflows.
|
|
7
7
|
|
|
@@ -15,10 +15,10 @@ The CLI commands under `clio-coder eval` support running, validating, reporting,
|
|
|
15
15
|
|
|
16
16
|
```bash
|
|
17
17
|
clio-coder eval validate --suite <suite.yaml>
|
|
18
|
-
clio-coder eval run --suite <suite.yaml> [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
|
|
18
|
+
clio-coder eval run --suite <suite.yaml> [--trials <n>] [--target <id>] [--model <id>] [--out <path>] [--clio-coder-entry <path>]
|
|
19
19
|
clio-coder eval run --task-file <tasks.yaml> [--repeat <n>] [--out <path>] [--clio-coder-entry <path>]
|
|
20
20
|
clio-coder eval report <evalId> --format text|json|md|swe-jsonl|junit
|
|
21
|
-
clio-coder eval compare <baselineEvalId> <candidateEvalId>
|
|
21
|
+
clio-coder eval compare <baselineEvalId> <candidateEvalId> [--metric <name>] [--format text|json|md|junit] [--allow-config-drift]
|
|
22
22
|
clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds <file>]
|
|
23
23
|
```
|
|
24
24
|
|
|
@@ -32,7 +32,7 @@ clio-coder eval gate <candidateEvalId> --baseline <baselineEvalId> [--thresholds
|
|
|
32
32
|
* `swe-jsonl`: Standardized JSONL format representing task runs (e.g. for SWE-bench comparisons).
|
|
33
33
|
* `junit`: XML report for CI/CD integration.
|
|
34
34
|
* **`compare`**: Compares two evaluation artifacts (baseline and candidate) by matching tasks.
|
|
35
|
-
* **`gate`**: Compares candidate metrics against baseline
|
|
35
|
+
* **`gate`**: Compares candidate metrics against baseline and absolute thresholds. Correctness and safety regressions fail independently of informational budgets.
|
|
36
36
|
|
|
37
37
|
Exit codes:
|
|
38
38
|
|
|
@@ -41,8 +41,8 @@ Exit codes:
|
|
|
41
41
|
| `eval validate` | `0` when validation passes | `2` for validation issues |
|
|
42
42
|
| `eval run` | `0` when all task repetitions pass | `1` when any task fails, `2` for invalid configs |
|
|
43
43
|
| `eval report` | `0` when artifact loads | `1` if artifact cannot be read, `2` for invalid ID |
|
|
44
|
-
| `eval compare` | `0` when both artifacts
|
|
45
|
-
| `eval gate` | `0` when
|
|
44
|
+
| `eval compare` | `0` when both artifacts compare and the behavioral hard gate passes | `1` for a hard regression or unreadable artifact, `2` for invalid ID |
|
|
45
|
+
| `eval gate` | `0` when correctness, safety, and hard threshold assertions pass | `1` for any hard failure, `2` for config/invalid ID errors |
|
|
46
46
|
|
|
47
47
|
---
|
|
48
48
|
|
|
@@ -103,10 +103,11 @@ tasks:
|
|
|
103
103
|
| --- | --- | --- |
|
|
104
104
|
| `version` | - | Must equal `2`. |
|
|
105
105
|
| `suite` | `id`, `title`, `visibility`, `description` | Metadata identifying the evaluation suite. |
|
|
106
|
-
| `matrix` | `targets[]`, `repeats` | Matrix of execution targets
|
|
106
|
+
| `matrix` | `targets[]`, `repeats`, `dimensions[]` | Matrix of execution targets, repetition count, and the execution-envelope fields intentionally varied by the suite. |
|
|
107
107
|
| `workspace` | `kind`, `path`, `url`, `commit`, `checkout`, `excludes` | Workspace strategy: `local` (run in-place), `git` (clone from URL), or `temp-copy` (isolated copy of a directory). |
|
|
108
108
|
| `runner` | `kind`, `prompt`, `command`, `commands`, `args`, `timeoutMs` | Runner type: `clio-run` (starts Clio agent loop), `context-index` (runs indexer), `context-init` (initializes context), `external-command` (spawns subprocess). |
|
|
109
|
-
| `
|
|
109
|
+
| `behavioral` | `schema`, `corpus`, `execution`, `expectedBehavior`, `forbiddenBehavior`, `judge` | Optional `clio.eval.scenario.v1` behavioral contract. Rules name a closed category and a typed predicate over transcript, tool, receipt, or grader facts. |
|
|
110
|
+
| `verify` | `commands`, `measure`, `assertions`, `forbidPaths` | Validation steps: shell commands, a task-outcome grader, metric assertions (e.g. `op: lt` for max token counts), and files/directories that must not be created or modified (`forbidPaths`). |
|
|
110
111
|
| `metrics` | `collect` | List of metric names to compile for the evaluation runs. |
|
|
111
112
|
|
|
112
113
|
---
|
|
@@ -114,7 +115,7 @@ tasks:
|
|
|
114
115
|
## Workspace Kinds
|
|
115
116
|
* **`local`**: Executes the task directly in the specified local path.
|
|
116
117
|
* **`git`**: Clones the repository from `url`, checks out the specified `commit` or `checkout` ref, and runs there.
|
|
117
|
-
* **`temp-copy`**: Copies the directory at `path` to a temporary workspace location before
|
|
118
|
+
* **`temp-copy`**: Copies the directory at `path` to a temporary workspace location immediately before the matrix item runs and removes it afterward. In a Git checkout, the copy contains exactly tracked files plus untracked files not excluded by Git ignore rules (`git ls-files --cached --others --exclude-standard`), with `excludes` applied afterward. Outside Git it retains the recursive directory copy. This prevents side-effects from polluting other task runs without copying ignored datasets or build trees.
|
|
118
119
|
|
|
119
120
|
---
|
|
120
121
|
|
|
@@ -194,12 +195,262 @@ export interface EvalArtifactV4 {
|
|
|
194
195
|
matrix: { target: string; model: string | null; thinking: string | null };
|
|
195
196
|
summary: EvalArtifactSummaryV4;
|
|
196
197
|
results: EvalArtifactResultV4[];
|
|
198
|
+
servingConfiguration?: EvalServingConfigurationV1;
|
|
199
|
+
aggregates?: EvalScenarioAggregateV1[];
|
|
197
200
|
}
|
|
198
201
|
```
|
|
199
202
|
|
|
203
|
+
`servingConfiguration` and `aggregates` are additive. The v4 reader still accepts an artifact that omits them, and each result's `verdict` is optional for the same reason, so an artifact written before this release loads unchanged.
|
|
204
|
+
|
|
200
205
|
---
|
|
201
206
|
|
|
202
|
-
##
|
|
207
|
+
## The verdict envelope
|
|
208
|
+
|
|
209
|
+
Every result carries a strictly parsed `clio.eval.verdict.v1` envelope (`src/domains/eval/schema/verdict.ts`). Suite v2 results are adapted into it at one explicit boundary (`src/domains/eval/schema/adapter.ts`) rather than by widening the artifact version, because the envelope carries no information a v4 artifact cannot hold.
|
|
210
|
+
|
|
211
|
+
```json
|
|
212
|
+
{
|
|
213
|
+
"schema": "clio.eval.verdict.v1",
|
|
214
|
+
"scenarioId": "latency-nonnegative",
|
|
215
|
+
"trialIndex": 0,
|
|
216
|
+
"outcome": "pass",
|
|
217
|
+
"machinery": "ok",
|
|
218
|
+
"reason": null,
|
|
219
|
+
"trackedMetrics": { "...": "see below" },
|
|
220
|
+
"behavioral": null,
|
|
221
|
+
"evidence": {
|
|
222
|
+
"assignmentId": "qcy5rfopdrfw",
|
|
223
|
+
"terminalReceiptDigest": "d85a3ad4f8ae...",
|
|
224
|
+
"graderExitCode": 0
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
The envelope is fail-closed by construction. `outcome` is one of `pass`, `fail`, or `unmeasured`; `machinery` is `ok` or `infrastructure_failure`; `reason` is null for a pass or unmeasured outcome and names the rule or failure class for every failure; the original `behavioral` reservation remains exactly `null`; and an envelope claiming both `infrastructure_failure` and `pass` is rejected at parse rather than recorded. A run whose harness broke therefore cannot be read as a model that succeeded. Behavioral results use the separately versioned sibling document below rather than changing this persisted schema.
|
|
230
|
+
|
|
231
|
+
### Behavioral scenario and verdict documents
|
|
232
|
+
|
|
233
|
+
Behavioral evaluation is additive and does not change the persisted `clio.eval.verdict.v1` reader. A Suite v2 task may declare a `clio.eval.scenario.v1` block, and its Artifact v4 result then carries a sibling `clio.eval.behavior.v1` document whose `verdictRef` names the verdict schema, scenario id, and trial index. This preserves existing verdicts and the tracked-metrics baseline while making a cross-linked behavioral document independently parseable.
|
|
234
|
+
|
|
235
|
+
The closed categories are `tool_choice`, `exploration`, `delegation`, `safety_comprehension`, `claim_grounding`, `denied_tool_recovery`, `completion_behavior`, and `task_correctness`. Each category result is exactly one of `satisfied`, `violated`, `unknown`, or `unmeasured`. The document outcome is `pass`, `behavioral_failure`, `unknown`, `unmeasured`, or `infrastructure_failure`; missing facts are never invented as successes, and an infrastructure failure cannot become a behavioral pass.
|
|
236
|
+
|
|
237
|
+
Expected and forbidden rules contain typed predicates over facts sourced from `transcript`, `tool`, `receipt`, or `grader`. Facts cite a locator, SHA-256 digest, and optional bounded excerpt. The parser caps rules, facts, evidence per category, ids, and explanations. Before judging, facts and unavailable sources are sorted into a canonical representation and hashed as `judgeInputDigest`, so input order cannot change the judge result. Duplicate or conflicting facts, missing categories, malformed evidence, contradictory outcomes, and a behavioral document that references a different result are refused.
|
|
238
|
+
|
|
239
|
+
Suite execution adapts scalar run metrics into these observable facts at the Suite v2 to Artifact v4 boundary. A declared no-tool target leaves tool-dependent rules `unmeasured`, while an available evidence source that omits a required fact produces `unknown`. Categories a role-specific scenario does not claim to measure remain `unmeasured`; they are not numeric zero and do not silently satisfy a rule.
|
|
240
|
+
|
|
241
|
+
### Public built-in behavioral corpus
|
|
242
|
+
|
|
243
|
+
The repository ships corpus `public-built-in-behavior` version `1.0.0` under
|
|
244
|
+
`benchmarks/eval/`. It contains no private prompts, endpoints, credentials, or
|
|
245
|
+
mutable external dataset:
|
|
246
|
+
|
|
247
|
+
- `behavioral-machinery.yaml` provides one positive and one adversarial
|
|
248
|
+
machinery-only check for each of the 13 shipped built-in worker recipes. Its
|
|
249
|
+
deterministic driver loads the production recipe catalog, admits a real
|
|
250
|
+
dispatch through the production gate, runs a scripted worker, and verifies
|
|
251
|
+
the sealed receipt and result-contract outcome. The 26 scenarios require no
|
|
252
|
+
model; they do not infer behavior by grepping recipe frontmatter.
|
|
253
|
+
- `behavioral-model.yaml` provides four isolated main-agent scenarios on the
|
|
254
|
+
`mini` target: a focused edit, adversarial scope control, required
|
|
255
|
+
delegation, and recovery after Bash is denied. Together they cover all eight
|
|
256
|
+
behavioral categories with per-tool call and blocked-call counts, distinct
|
|
257
|
+
and allowlisted read-path counts, declared decoy hits, and grader-emitted
|
|
258
|
+
claim-support and completion facts.
|
|
259
|
+
- `behavioral-model-negative-control.yaml` intentionally reads a declared
|
|
260
|
+
decoy. A healthy run solves its literal task while recording
|
|
261
|
+
`behavioral_failure` with violated exploration and safety labels, proving
|
|
262
|
+
that the rules can reject observed model behavior rather than merely restate
|
|
263
|
+
aggregate success counters.
|
|
264
|
+
|
|
265
|
+
Build once, then run either focused suite from the repository root:
|
|
266
|
+
|
|
267
|
+
```sh
|
|
268
|
+
node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-machinery.yaml --clio-coder-entry dist/cli/index.js
|
|
269
|
+
node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model.yaml --target mini --clio-coder-entry dist/cli/index.js
|
|
270
|
+
node dist/cli/index.js eval run --suite benchmarks/eval/behavioral-model-negative-control.yaml --target mini --clio-coder-entry dist/cli/index.js
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
The machinery tasks use the repository read-only and create only private
|
|
274
|
+
scratch state under `TMPDIR`; model tasks use a fresh `temp-copy` workspace and
|
|
275
|
+
remove it after the matrix item settles. The machinery suite is the fast
|
|
276
|
+
admission, worker, and receipt contract. The model suite is the live behavioral
|
|
277
|
+
measurement: keep its Artifact v4 output as evidence for the exact target and
|
|
278
|
+
serving configuration that ran, rather than treating one observed model result
|
|
279
|
+
as a universal guarantee. Behavioral facts and their evidence store only
|
|
280
|
+
bounded read counters, not path strings. As with other eval runs, the artifact's
|
|
281
|
+
bounded diagnostic stdout may retain the underlying tool event stream.
|
|
282
|
+
|
|
283
|
+
### `trackedMetrics`
|
|
284
|
+
|
|
285
|
+
Eleven numbers plus a reason histogram, each carrying the source it came from. `source` is `ledger` (the per-call ledger folded from the worker's own JSON stream), `receipt` (the sealed run receipt), or `estimated`, and `estimated` is what a missing observation is marked as rather than being silently counted as measured.
|
|
286
|
+
|
|
287
|
+
| Metric | Usual source |
|
|
288
|
+
| --- | --- |
|
|
289
|
+
| `modelCalls` | ledger |
|
|
290
|
+
| `uncachedPrefillTokens` | ledger, from `promptCache.backend` |
|
|
291
|
+
| `cacheReadTokens` | ledger, from `promptCache.backend`, falling back to pi-ai cache reads |
|
|
292
|
+
| `generatedTokens` | ledger |
|
|
293
|
+
| `reasoningTokens` | receipt; nullable, because absent and zero are different claims |
|
|
294
|
+
| `toolCalls`, `toolErrors` | ledger when present, otherwise receipt |
|
|
295
|
+
| `ttftMsFirstCall` | ledger |
|
|
296
|
+
| `wallClockMs` | receipt |
|
|
297
|
+
| `contextTokensAtEnd` | ledger |
|
|
298
|
+
| `compactions` | ledger |
|
|
299
|
+
| `expectedColdReasons` | ledger, one sourced count per reason |
|
|
300
|
+
|
|
301
|
+
A dispatched worker's receipt reports `sessionId: null` and writes no session archive, which is why the ledger source exists at all: the runner folds structured usage, backend timing, cache, and monotonic TTFT facts out of the worker's `message_end` events. It keeps no prompt text, no model prose, and no tool-result content in that fold.
|
|
302
|
+
|
|
303
|
+
### Scenario aggregates
|
|
304
|
+
|
|
305
|
+
`aggregates` groups verdicts by `scenarioId`, sets `k` to the trial count, and records `passAtK` (any trial passed) and `passPowK` (every trial passed). Each tracked numeric metric reports observation, measured, and unmeasured counts, mean, min, max, nearest-rank p90, population variance, standard deviation, and the set of sources observed. A metric with no observation keeps every numeric statistic `null`; it never becomes zero. At `k: 1`, variance and standard deviation are zero only when the value was actually measured.
|
|
306
|
+
|
|
307
|
+
### Behavioral multi-metric results
|
|
308
|
+
|
|
309
|
+
A result with a `clio.eval.behavior.v1` verdict also carries the additive
|
|
310
|
+
`clio.eval.behavior.metrics.v1` projection. The projection binds the scenario
|
|
311
|
+
to its role and target/model envelope and records one `number | null`
|
|
312
|
+
observation for each closed metric. The source travels beside every value:
|
|
313
|
+
|
|
314
|
+
| Family | Metric | Direction | Gate | Source |
|
|
315
|
+
|---|---|---|---|---|
|
|
316
|
+
| correctness | `correctness.taskSolved` | higher | hard | grader |
|
|
317
|
+
| safety | `safety.violations` | lower | hard | behavioral label |
|
|
318
|
+
| behavior | `behavior.labelViolations` | lower | informational | behavioral labels |
|
|
319
|
+
| efficiency | `efficiency.toolCalls` | lower | informational | terminal tool events |
|
|
320
|
+
| exploration | `exploration.unnecessaryReads` | lower | informational | read observation counters |
|
|
321
|
+
| delegation | `delegation.quality` | higher | informational | behavioral label |
|
|
322
|
+
| claims | `claims.unsupported` | lower | informational | grader |
|
|
323
|
+
| tokens | `tokens.total` | lower | informational | runner usage stream |
|
|
324
|
+
| latency | `latency.wallMs` | lower | informational | monotonic runner clock |
|
|
325
|
+
| cost | `cost.usd` | lower | informational | sealed receipt |
|
|
326
|
+
|
|
327
|
+
Label metrics are numeric projections only when the category is `satisfied` or
|
|
328
|
+
`violated`; `unknown` and `unmeasured` remain null. A missing grader, token
|
|
329
|
+
stream, receipt, read observation, or category label likewise remains null.
|
|
330
|
+
The projection therefore records observation coverage without claiming that
|
|
331
|
+
silence was success, safety, or zero cost.
|
|
332
|
+
|
|
333
|
+
### Behavioral comparisons and variance
|
|
334
|
+
|
|
335
|
+
`eval compare` reduces the behavioral projection independently for every
|
|
336
|
+
scenario, role, target id, and model id. Each row contains the baseline and
|
|
337
|
+
candidate distributions, coverage, mean delta, variance delta, and two closed
|
|
338
|
+
classifications: `improved`, `regressed`, `unchanged`, or `incomparable` for the
|
|
339
|
+
mean and for variability. Lower variance is the improvement direction for the
|
|
340
|
+
variability classification.
|
|
341
|
+
|
|
342
|
+
Correctness and safety rows are hard. A measured regression fails the hard
|
|
343
|
+
gate even when pass rate, tokens, latency, or cost improved. Losing a
|
|
344
|
+
correctness or safety measurement that existed in the baseline is also a hard
|
|
345
|
+
failure; a category explicitly unmeasured on both sides stays incomparable but
|
|
346
|
+
does not invent a regression. Other families remain visible informational
|
|
347
|
+
tradeoffs. `--metric` accepts either a behavioral metric or family as well as a
|
|
348
|
+
tracked metric, but filtering displayed rows never filters the hard-gate
|
|
349
|
+
decision.
|
|
350
|
+
|
|
351
|
+
Comparison output supports `text`, `json`, `md`, and `junit`. All four carry
|
|
352
|
+
the same hard-gate result and closed classifications. JUnit failures represent
|
|
353
|
+
only hard behavioral failures; an informational efficiency or cost regression
|
|
354
|
+
is emitted as testcase output rather than a failed testcase.
|
|
355
|
+
|
|
356
|
+
### Execution-envelope provenance and comparability
|
|
357
|
+
|
|
358
|
+
Every newly written behavioral result carries an additive
|
|
359
|
+
`clio.eval.execution-envelope.v1` sibling. Artifact v4,
|
|
360
|
+
`clio.eval.verdict.v1`, and `clio.eval.behavior.metrics.v1` retain their
|
|
361
|
+
existing identities. The envelope records the selected prompt fragment ids,
|
|
362
|
+
authored versions or `unversioned` marker, fragment content hashes, prompt
|
|
363
|
+
composition hash, recipe id/version/fingerprint when a worker recipe applies,
|
|
364
|
+
target, wire model, runtime, thinking level, tool signature, effective
|
|
365
|
+
autonomy, rule-pack and project-policy hashes, bounded project-context
|
|
366
|
+
provenance, and corpus id/version. A machinery-only scenario uses explicit
|
|
367
|
+
nulls for model concepts that did not apply; null is not substituted for a
|
|
368
|
+
fact that was observed.
|
|
369
|
+
|
|
370
|
+
Suite v2 may declare `matrix.dimensions` from `prompt`, `recipe`, `target`,
|
|
371
|
+
`wireModel`, `runtime`, `thinkingLevel`, `toolSignature`, `autonomy`, `policy`,
|
|
372
|
+
`projectContext`, and `corpus`. Comparison ignores only dimensions declared by
|
|
373
|
+
both artifacts. Any other envelope difference marks every metric row for that
|
|
374
|
+
scenario/role/target incomparable and fails the behavioral gate. A missing
|
|
375
|
+
envelope on only one side is also incomparable. Two older artifacts that both
|
|
376
|
+
predate the sibling remain readable and compare under their existing data.
|
|
377
|
+
|
|
378
|
+
Text, JSON, Markdown, and JUnit comparison reports carry the same envelope
|
|
379
|
+
mismatch. Text and Markdown also include independent per-scenario and per-role
|
|
380
|
+
baseline/candidate counts for improved, regressed, unchanged, and incomparable
|
|
381
|
+
metric means and variances. When the prompt or recipe identity changes, the
|
|
382
|
+
generated evidence names each affected corpus scenario and role instead of
|
|
383
|
+
hiding it behind an aggregate score.
|
|
384
|
+
|
|
385
|
+
### Checked behavioral release baseline
|
|
386
|
+
|
|
387
|
+
The checked deterministic baseline is
|
|
388
|
+
`benchmarks/eval/behavioral-machinery-baseline.json`. The release gate runs all
|
|
389
|
+
26 machinery-only scenarios through the built CLI and compares a stable
|
|
390
|
+
projection of their labels, metrics, and execution envelopes with that file.
|
|
391
|
+
It requires no model, private endpoint, credential, or mutable dataset.
|
|
392
|
+
|
|
393
|
+
When an intentional prompt, recipe, policy, or expected-behavior change moves
|
|
394
|
+
the evidence, run the same machinery suite first, inspect the failing diff and
|
|
395
|
+
the named affected corpus results, then update explicitly:
|
|
396
|
+
|
|
397
|
+
```sh
|
|
398
|
+
npm run build
|
|
399
|
+
node benchmarks/eval/check-behavioral-release.mjs --update
|
|
400
|
+
git diff -- benchmarks/eval/behavioral-machinery-baseline.json
|
|
401
|
+
```
|
|
402
|
+
|
|
403
|
+
The baseline update belongs in the reviewed change that caused it. Do not use
|
|
404
|
+
the update command merely to make a red gate green. The model-required and
|
|
405
|
+
negative-control suites remain manual release evidence because their outputs
|
|
406
|
+
depend on a live target; they are never folded into the deterministic baseline.
|
|
407
|
+
The projection excludes `latency.wallMs` because scheduler timing is not stable
|
|
408
|
+
evidence. Behavioral labels, deterministic metrics, and the execution envelope
|
|
409
|
+
remain checked byte for byte.
|
|
410
|
+
|
|
411
|
+
### Hard thresholds and informational budgets
|
|
412
|
+
|
|
413
|
+
Suite and external threshold files keep two separate assertion lists:
|
|
414
|
+
|
|
415
|
+
```yaml
|
|
416
|
+
thresholds:
|
|
417
|
+
fail:
|
|
418
|
+
- metric: task.solved
|
|
419
|
+
op: eq
|
|
420
|
+
value: false
|
|
421
|
+
informational:
|
|
422
|
+
- metric: cost.usd
|
|
423
|
+
op: gt
|
|
424
|
+
value: 0.25
|
|
425
|
+
```
|
|
426
|
+
|
|
427
|
+
`fail` is the backwards-compatible hard list. A firing or unresolved hard
|
|
428
|
+
assertion makes `eval run` or `eval gate` exit nonzero. `informational` uses the
|
|
429
|
+
same typed predicates and reports every firing budget or missing measurement,
|
|
430
|
+
but never changes the exit status. `eval gate` additionally evaluates the
|
|
431
|
+
baseline-to-candidate correctness and safety hard gate, so a cheaper candidate
|
|
432
|
+
cannot offset a task or safety regression.
|
|
433
|
+
|
|
434
|
+
### `--trials N`
|
|
203
435
|
|
|
204
|
-
|
|
436
|
+
`--trials N` overrides the suite's `matrix.repeats` and asks for an isolated workspace per matrix item. A `local` workspace is converted to a temporary copy immediately before that item runs, so an explicit trial run never mutates the directory it was pointed at; `git` and `temp-copy` workspaces already produce a distinct preparation directory per item. Workspace and state directories are removed on the item's `finally` path, including runner, setup, and copy failures. The trial index rides through to each verdict's `trialIndex`.
|
|
437
|
+
|
|
438
|
+
### Serving-configuration provenance and drift refusal
|
|
439
|
+
|
|
440
|
+
`servingConfiguration` records what the numbers were measured against: `targetId`, `runtimeId`, `modelId`, `serverBuild`, `total_slots`, `thinkingLevel`, and `compiledPromptHash`. The build string and slot count are read from the server after the matrix has run while it is still awake, by fetching `/props` and falling back to the model-qualified slots query when `/props` exposes no `total_slots`. The prompt hash is the receipt's static composition hash, so a prompt change is visible as a configuration change rather than as a mysterious metric shift.
|
|
441
|
+
|
|
442
|
+
`eval compare` prints both configurations and refuses outright when they differ:
|
|
443
|
+
|
|
444
|
+
```text
|
|
445
|
+
serving configuration drift; pass --allow-config-drift to compare these runs
|
|
446
|
+
baseline serving: target=mini runtime=llamacpp model=... server_build=b226-2115b73d8 total_slots=1 thinking=off compiled_prompt_hash=...
|
|
447
|
+
candidate serving: ...
|
|
448
|
+
```
|
|
449
|
+
|
|
450
|
+
`--allow-config-drift` proceeds and labels the comparison `config drift: allowed`. There is a second refusal that has no override: a metric whose baseline distribution contains an `estimated` observation and whose candidate does not, or the reverse, raises `EvalTrackedMetricSourceMismatchError` rather than printing a delta, because subtracting a measurement from an estimate produces a number that looks like evidence and is not. `--metric <name>` filters tracked or behavioral rows, accepts `expectedColdReasons`, a specific `expectedColdReasons.<reason>`, a behavioral family, or a behavioral metric, and errors when the name matches nothing.
|
|
451
|
+
|
|
452
|
+
---
|
|
453
|
+
|
|
454
|
+
## Task Outcome Measurement (`verify.measure`)
|
|
205
455
|
|
|
456
|
+
Task outcome commands declared under `verify.measure` are the code grader for whether the model solved the workload and record metrics (`task.solved`, `task.exitCode`). A non-zero exit fails the final result and is named on its verdict as `reason: grader_failed`, while `machinery` remains `ok` when the runner and machinery verifiers succeeded. This keeps the artifact's `pass`, verdict outcome, scenario aggregates, and summary on one pass decision without misreporting a grader failure as broken machinery.
|
package/docs/evals-internal.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# Internal Eval Suites
|
|
2
2
|
|
|
3
3
|
> [!TIP]
|
|
4
|
-
> **Interactive Spec Available:** An interactive blueprint is available at [docs/html/evals_internal_blueprint.html](html/evals_internal_blueprint.html) (Version: 0.3.
|
|
4
|
+
> **Interactive Spec Available:** An interactive blueprint is available at [docs/html/evals_internal_blueprint.html](html/evals_internal_blueprint.html) (Version: 0.3.9).
|
|
5
5
|
|
|
6
6
|
Private suites should live outside this repository. Keep datasets, prompts,
|
|
7
7
|
live fleet coordinates, calibration outputs, and raw run artifacts in a private
|
|
@@ -19,6 +19,77 @@ data directory. Product eval artifacts and external benchmark campaigns are
|
|
|
19
19
|
separate: public benchmark adapters live under `benchmarks/community/` and do
|
|
20
20
|
not use the eval runner.
|
|
21
21
|
|
|
22
|
+
The public behavioral corpus is the deliberate exception to the otherwise
|
|
23
|
+
private Suite v2 data policy. Its reviewable, synthetic suites live under
|
|
24
|
+
`benchmarks/eval/`: a model-free positive/adversarial authority pair for every
|
|
25
|
+
built-in worker recipe, four tiny main-agent model scenarios covering all eight
|
|
26
|
+
behavioral categories with event- and grader-derived facts, and an intentional
|
|
27
|
+
decoy negative control. The model-free driver uses the shipped recipe catalog,
|
|
28
|
+
real dispatch admission, scripted workers, and sealed receipts rather than
|
|
29
|
+
frontmatter inspection. See
|
|
30
|
+
[eval-runner.md](eval-runner.md#public-built-in-behavioral-corpus) for the
|
|
31
|
+
focused commands. Private prompts, calibration cases, fleet coordinates, and
|
|
32
|
+
campaign artifacts still belong outside this repository and must not be copied
|
|
33
|
+
into the public corpus.
|
|
34
|
+
|
|
35
|
+
## Running a private suite as a measurement
|
|
36
|
+
|
|
37
|
+
A private suite is usually run to answer whether a harness change moved
|
|
38
|
+
something, which makes it a measurement rather than a pass or fail. Three
|
|
39
|
+
mechanics matter for that, all documented in full in
|
|
40
|
+
[eval-runner.md](eval-runner.md#the-verdict-envelope).
|
|
41
|
+
|
|
42
|
+
Run repeated trials with `--trials N` rather than by editing `matrix.repeats`.
|
|
43
|
+
The flag overrides the suite's repeat count and asks for an isolated workspace
|
|
44
|
+
per matrix item, so a `local` workspace is copied for the run instead of being
|
|
45
|
+
mutated across trials. Each result's verdict carries its `trialIndex`, and the
|
|
46
|
+
artifact's `aggregates` reduce them per scenario: `k`, `passAtK` (any trial
|
|
47
|
+
passed), `passPowK` (every trial passed), and a mean and nearest-rank p90 for
|
|
48
|
+
every tracked metric. A single trial produces a `k: 1` aggregate whose mean and
|
|
49
|
+
p90 are the same observed value, which is a fact to state in a report rather
|
|
50
|
+
than a distribution to reason about.
|
|
51
|
+
|
|
52
|
+
Read the tracked metrics with their sources attached. A private suite on a
|
|
53
|
+
local target is measuring prefill economics as much as correctness, so
|
|
54
|
+
`uncachedPrefillTokens`, `cacheReadTokens`, `ttftMsFirstCall`, and the
|
|
55
|
+
`expectedColdReasons` histogram are the interesting columns, and each one says
|
|
56
|
+
whether it came from the ledger, from the receipt, or was `estimated`. A metric
|
|
57
|
+
marked `estimated` on one side of a comparison and measured on the other is
|
|
58
|
+
refused rather than differenced.
|
|
59
|
+
|
|
60
|
+
Behavioral suites add a second projection beside those tracked performance
|
|
61
|
+
metrics. Compare it per scenario, role, and target/model envelope rather than
|
|
62
|
+
reducing unlike roles into one pass rate. Correctness and safety are hard
|
|
63
|
+
regression gates; tool efficiency, unnecessary exploration, delegation
|
|
64
|
+
quality, unsupported claims, tokens, latency, and receipt cost remain separate
|
|
65
|
+
families with their own measured coverage and repeat variance. A missing value
|
|
66
|
+
is null and makes that row incomparable, never zero.
|
|
67
|
+
|
|
68
|
+
Put release-blocking assertions under `thresholds.fail` and non-blocking spend
|
|
69
|
+
or latency budgets under `thresholds.informational`. Informational findings are
|
|
70
|
+
printed in every gate run but do not change its exit status. Do not put a cost
|
|
71
|
+
budget in the hard list to compensate for weak correctness, and do not turn a
|
|
72
|
+
correctness rule into an informational budget; the comparison gate evaluates
|
|
73
|
+
correctness and safety before either kind of operator-authored threshold.
|
|
74
|
+
|
|
75
|
+
Record the serving configuration or the comparison is not one. The artifact
|
|
76
|
+
captures `targetId`, `runtimeId`, `modelId`, `serverBuild`, `total_slots`,
|
|
77
|
+
`thinkingLevel`, and `compiledPromptHash`, read from the server after the matrix
|
|
78
|
+
has run while it is still awake. `eval compare` refuses two artifacts whose
|
|
79
|
+
configurations differ unless `--allow-config-drift` is passed, and prints both
|
|
80
|
+
either way. Treat that refusal as the useful behavior it is: a private suite
|
|
81
|
+
compared across a server restart that changed a flag, a quantization, or the
|
|
82
|
+
thinking level is measuring the server, not the change under test.
|
|
83
|
+
|
|
84
|
+
The verdict envelope keeps its original `behavioral: null` field for compatibility.
|
|
85
|
+
A suite that declares a versioned behavioral scenario records the result as a
|
|
86
|
+
separate `clio.eval.behavior.v1` document on the Artifact v4 result, cross-linked
|
|
87
|
+
to the unchanged verdict identity. Its labels come only from bounded transcript,
|
|
88
|
+
tool, receipt, or grader facts, never from an ungrounded judge paragraph. A run whose harness broke records
|
|
89
|
+
`machinery: "infrastructure_failure"`, which the parser refuses to pair with a
|
|
90
|
+
`pass`, so a private suite cannot report a passing rate that includes runs
|
|
91
|
+
nothing measured.
|
|
92
|
+
|
|
22
93
|
## Context Regression Seed
|
|
23
94
|
|
|
24
95
|
```yaml
|
|
@@ -267,4 +338,3 @@ thresholds:
|
|
|
267
338
|
op: gt
|
|
268
339
|
value: 0
|
|
269
340
|
```
|
|
270
|
-
|