@duckcodeailabs/dql-cli 1.14.0 → 1.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/args.d.ts +15 -0
- package/dist/args.d.ts.map +1 -1
- package/dist/args.js +25 -0
- package/dist/args.js.map +1 -1
- package/dist/assets/dql-notebook/assets/{AgentLogPage-Ch7VK20X.js → AgentLogPage-DKbGpRQS.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildDialog-DBr5TmyM.js → AiBuildDialog-DPSu0Mly.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildResult-jLPzQO7O.js → AiBuildResult-1uaGnpi1.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiSidePanel-CdlGsiVC.js → AiSidePanel-BwMREwa7.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AnalyticsHome-C0DXbOwY.js → AnalyticsHome-BGfey_ve.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AppsView-CcpwjApv.js → AppsView-CM1tPywy.js} +4 -4
- package/dist/assets/dql-notebook/assets/{BlockStudio-CFYxafw-.js → BlockStudio--S6WFVO4.js} +1 -1
- package/dist/assets/dql-notebook/assets/{BusinessArtifactView-C0kLYg2p.js → BusinessArtifactView-BEAJ-yNW.js} +1 -1
- package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-CMwElXD_.js → DbtFirstModelingPage-CNyU5MBX.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GitPage-JjhRDeWY.js → GitPage-IcydAai3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GlobalAiRail-DakE4NdR.js → GlobalAiRail-CSKW-5eD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GovernedContextPage-BokDqG6a.js → GovernedContextPage-CFen0eFT.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HelpDocsPage-CjOv6_gz.js → HelpDocsPage-D0D9UCIz.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HomePage-nGaNcwdw.js → HomePage-CmSxapR3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDAG-CO6CFJRg.js → LineageDAG-BGUcIt1B.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDetailView-BnGb7OF7.js → LineageDetailView-Dx48Zfzb.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDrawer-C5Y0Ht0b.js → LineageDrawer-zaBfSLZO.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-Cgh3GR3F.js → LineagePathBreadcrumb-CTZJp4_r.js} +1 -1
- package/dist/assets/dql-notebook/assets/{MiniLineageGraph-Kla9PYuj.js → MiniLineageGraph-CdivNR1S.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewBlockModal-BR2SnmPT.js → NewBlockModal-DMBFB7nE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewNotebookModal-BaCHMYeB.js → NewNotebookModal-DsC9CWlW.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NotebookEditor-CBLqY8cE.js → NotebookEditor-hs-kw8v9.js} +1 -1
- package/dist/assets/dql-notebook/assets/{ReadinessPage-CiN0IWSS.js → ReadinessPage-DJ83cMik.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SetupOnboarding-BhiCsYF-.js → SetupOnboarding-B1Pu_Bvv.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SkillsPage-CCSf8VMm.js → SkillsPage-BHKDay8n.js} +1 -1
- package/dist/assets/dql-notebook/assets/{TrustBadge-BgQmFe_x.js → TrustBadge-BkyGgob2.js} +1 -1
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-BYdrTaEW.js +89 -0
- package/dist/assets/dql-notebook/assets/{answer-to-notebook-AeDYUDla.js → answer-to-notebook-DmNiuLQA.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-left-C55x2hq_.js → arrow-left--1rsrxm8.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-right-C1cJrhOm.js → arrow-right-D5TdqY1G.js} +1 -1
- package/dist/assets/dql-notebook/assets/{book-open-text-CQf_sdv2.js → book-open-text-Bw7nHbzg.js} +1 -1
- package/dist/assets/dql-notebook/assets/{circle-x-X8-Z2yLY.js → circle-x-DLe6NNM4.js} +1 -1
- package/dist/assets/dql-notebook/assets/{dagre.esm-BjjNYKyY.js → dagre.esm-CW5QZdBt.js} +1 -1
- package/dist/assets/dql-notebook/assets/{external-link-C2zrz5DH.js → external-link-C9Q97sA3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{grip-vertical-qGV_PYGU.js → grip-vertical-CztvkIgo.js} +1 -1
- package/dist/assets/dql-notebook/assets/{index-ByTDPDaH.js → index-zHHzDn6l.js} +127 -127
- package/dist/assets/dql-notebook/assets/{link-2-VpyOxQXG.js → link-2-CiKAvumL.js} +1 -1
- package/dist/assets/dql-notebook/assets/{list-tree-CH2Jhwms.js → list-tree-BtnP2nQ5.js} +1 -1
- package/dist/assets/dql-notebook/assets/{minimize-2-B2TZJ8BT.js → minimize-2-TSFGxcCP.js} +1 -1
- package/dist/assets/dql-notebook/assets/{panel-right-open-BunF88lt.js → panel-right-open-BfXIUWy0.js} +1 -1
- package/dist/assets/dql-notebook/assets/{play-DAFVF4_G.js → play-DVbSFJHD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{rotate-ccw-DGCrrqtY.js → rotate-ccw-D_cesDcX.js} +1 -1
- package/dist/assets/dql-notebook/assets/{semantic-fields-Ci-9QL9F.js → semantic-fields-CoVStdYB.js} +1 -1
- package/dist/assets/dql-notebook/assets/{sliders-horizontal-BOAlXXbn.js → sliders-horizontal-l7xV9K5A.js} +1 -1
- package/dist/assets/dql-notebook/assets/{star-CkksZXHt.js → star-CSBS0H3b.js} +1 -1
- package/dist/assets/dql-notebook/assets/{triangle-alert-D3mjyJZE.js → triangle-alert-BTrnyY4q.js} +1 -1
- package/dist/assets/dql-notebook/assets/{upload-CTNOAVEO.js → upload-sySLq9zb.js} +1 -1
- package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-DiQjc7x-.js → usePersistedAgentThreadId-CzwgGdus.js} +1 -1
- package/dist/assets/dql-notebook/assets/{user-round-BWd5tQRg.js → user-round-ChlgXi9j.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wand-sparkles-CqsAv8P-.js → wand-sparkles-CGw0ytyT.js} +1 -1
- package/dist/assets/dql-notebook/assets/{workflow-CSqsj-sC.js → workflow-C_RltiK5.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wrench-DWqzqlX8.js → wrench-D-wLfeu0.js} +1 -1
- package/dist/assets/dql-notebook/assets/{x-65M5rLCE.js → x-B28hIJIC.js} +1 -1
- package/dist/assets/dql-notebook/index.html +1 -1
- package/dist/commands/agent-eval-cassette.d.ts +164 -0
- package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
- package/dist/commands/agent-eval-cassette.js +313 -0
- package/dist/commands/agent-eval-cassette.js.map +1 -0
- package/dist/commands/agent-eval-runtime.d.ts +109 -0
- package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
- package/dist/commands/agent-eval-runtime.js +165 -0
- package/dist/commands/agent-eval-runtime.js.map +1 -0
- package/dist/commands/agent.d.ts +124 -4
- package/dist/commands/agent.d.ts.map +1 -1
- package/dist/commands/agent.js +410 -106
- package/dist/commands/agent.js.map +1 -1
- package/dist/commands/compile.d.ts +13 -1
- package/dist/commands/compile.d.ts.map +1 -1
- package/dist/commands/compile.js +36 -4
- package/dist/commands/compile.js.map +1 -1
- package/dist/commands/sync.d.ts.map +1 -1
- package/dist/commands/sync.js +11 -3
- package/dist/commands/sync.js.map +1 -1
- package/dist/index.js +4 -0
- package/dist/index.js.map +1 -1
- package/dist/llm/analyst-loop-tools.d.ts +19 -0
- package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
- package/dist/llm/analyst-loop-tools.js +56 -0
- package/dist/llm/analyst-loop-tools.js.map +1 -0
- package/dist/llm/providers/dql-agent-provider.d.ts +51 -1
- package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
- package/dist/llm/providers/dql-agent-provider.js +563 -14
- package/dist/llm/providers/dql-agent-provider.js.map +1 -1
- package/dist/llm/types.d.ts +47 -1
- package/dist/llm/types.d.ts.map +1 -1
- package/dist/local-runtime.d.ts +123 -1
- package/dist/local-runtime.d.ts.map +1 -1
- package/dist/local-runtime.js +1437 -163
- package/dist/local-runtime.js.map +1 -1
- package/dist/package.json +10 -10
- package/package.json +10 -10
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-Bw5AXNDB.js +0 -88
package/dist/commands/agent.js
CHANGED
|
@@ -21,6 +21,8 @@
|
|
|
21
21
|
* dql agent feedback <up|down> --block <id> --question "..."
|
|
22
22
|
* Records feedback into the KG. Used by clients without MCP access.
|
|
23
23
|
*/
|
|
24
|
+
import { answerFromRuntimeRun, driveViaRuntime, projectRuntimeRun, } from './agent-eval-runtime.js';
|
|
25
|
+
import { CassetteStore, cassetteEvidenceSummary, cassetteDirFor, evalCassetteCanonicalizationV2, withCassette, } from './agent-eval-cassette.js';
|
|
24
26
|
import { existsSync, readFileSync } from 'node:fs';
|
|
25
27
|
import { join, resolve } from 'node:path';
|
|
26
28
|
import { load as loadYaml } from 'js-yaml';
|
|
@@ -551,7 +553,14 @@ async function runEval(rest, flags) {
|
|
|
551
553
|
if (!existsSync(kgPath))
|
|
552
554
|
await reindexProject(projectRoot, { kgPath });
|
|
553
555
|
const providerName = flags.provider;
|
|
554
|
-
const
|
|
556
|
+
const rawProvider = await pickProvider(providerName);
|
|
557
|
+
// Cassettes make a provider-backed suite repeatable and free to re-run. They
|
|
558
|
+
// apply to the in-process driver here; `--via runtime` needs the SERVER
|
|
559
|
+
// started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
|
|
560
|
+
const cassetteMode = flags.cassette;
|
|
561
|
+
const provider = cassetteMode === 'record' || cassetteMode === 'replay'
|
|
562
|
+
? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode, evalCassetteCanonicalizationV2(projectRoot))
|
|
563
|
+
: rawProvider;
|
|
555
564
|
const reasoningEffort = cliReasoningEffort(flags);
|
|
556
565
|
const requestedDepth = cliAnalysisDepth(flags);
|
|
557
566
|
const kg = new KGStore(kgPath);
|
|
@@ -562,10 +571,22 @@ async function runEval(rest, flags) {
|
|
|
562
571
|
// gracefully when no provider is available so offline eval stays deterministic.
|
|
563
572
|
const judge = Boolean(flags.judge);
|
|
564
573
|
const judgeComplete = async ({ system, user }) => provider.generate([{ role: 'system', content: system }, { role: 'user', content: user }], {});
|
|
574
|
+
// Which half of the stack is under test. `loop` preserves today's behaviour;
|
|
575
|
+
// `runtime` is the one that exercises routing and gates end to end.
|
|
576
|
+
const via = flags.via === 'runtime' ? 'runtime' : 'loop';
|
|
565
577
|
const runtimeBase = flags.runtimeUrl
|
|
566
578
|
?? flags.runtime
|
|
567
579
|
?? process.env.DQL_RUNTIME_URL
|
|
568
580
|
?? 'http://127.0.0.1:3474';
|
|
581
|
+
if (via === 'runtime') {
|
|
582
|
+
// Fail fast and loudly. Without this, an unreachable server turns every case
|
|
583
|
+
// into a transport error and the report reads as a false-refusal spike that
|
|
584
|
+
// no code change caused.
|
|
585
|
+
const probe = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/health`).catch(() => null);
|
|
586
|
+
if (!probe?.ok) {
|
|
587
|
+
throw new Error(`--via runtime needs a running server at ${runtimeBase}. Start one with \`dql serve\`, or use --via loop to score the answer loop in-process.`);
|
|
588
|
+
}
|
|
589
|
+
}
|
|
569
590
|
const semanticLayer = loadAgentSemanticLayer(projectRoot);
|
|
570
591
|
const expandGroundingContext = createGroundingContextExpander(projectRoot);
|
|
571
592
|
const answerLoopTools = buildAnswerLoopTools(projectRoot);
|
|
@@ -605,35 +626,71 @@ async function runEval(rest, flags) {
|
|
|
605
626
|
}
|
|
606
627
|
: undefined,
|
|
607
628
|
}).catch(() => undefined);
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
629
|
+
// `--via runtime` posts to a running `dql serve` so the case exercises the
|
|
630
|
+
// router, engine, plan boundary, and gates. The in-process driver below
|
|
631
|
+
// calls the answer loop directly and cannot observe any of them, which is
|
|
632
|
+
// why a refusal metric taken from it reads cleaner than users experience.
|
|
633
|
+
const runtimeRun = via === 'runtime'
|
|
634
|
+
? await driveViaRuntime({ runtimeBase, question: testCase.question })
|
|
635
|
+
: undefined;
|
|
636
|
+
// Runtime mode is scored from the persisted AgentRun. The transport
|
|
637
|
+
// adapter intentionally has no AgentAnswer.contextPack, so borrowing the
|
|
638
|
+
// local preflight pack here would fabricate retrieval/route evidence for a
|
|
639
|
+
// different execution path.
|
|
640
|
+
const runtimeProjection = runtimeRun ? projectRuntimeRun(runtimeRun) : undefined;
|
|
641
|
+
const result = runtimeRun
|
|
642
|
+
? answerFromRuntimeRun(runtimeRun)
|
|
643
|
+
: await answer({
|
|
644
|
+
question: testCase.question,
|
|
645
|
+
domain: testCase.domain,
|
|
646
|
+
domainContext: testCase.domain && manifest
|
|
647
|
+
? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
|
|
648
|
+
: undefined,
|
|
649
|
+
provider,
|
|
650
|
+
kg,
|
|
651
|
+
manifest: manifest ?? undefined,
|
|
652
|
+
skills,
|
|
653
|
+
memoryContext,
|
|
654
|
+
followUp: testCase.followUp,
|
|
655
|
+
semanticLayer,
|
|
656
|
+
schemaContext,
|
|
657
|
+
contextPack,
|
|
658
|
+
reasoningEffort,
|
|
659
|
+
analysisDepth: contextBudget.analysisDepth,
|
|
660
|
+
expandGroundingContext,
|
|
661
|
+
answerLoopTools,
|
|
662
|
+
executeCertifiedBlock: execute && manifest
|
|
663
|
+
? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
|
|
664
|
+
: undefined,
|
|
665
|
+
executeGeneratedSql: execute
|
|
666
|
+
? createGeneratedSqlExecutor(runtimeBase)
|
|
667
|
+
: undefined,
|
|
668
|
+
captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
|
|
669
|
+
const slug = deriveGeneratedDraftSlug(draftQuestion);
|
|
670
|
+
const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
|
|
671
|
+
if (dqlArtifact?.kind === 'semantic_block') {
|
|
672
|
+
if (!flags.save) {
|
|
673
|
+
return {
|
|
674
|
+
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
675
|
+
askedTimes: 0,
|
|
676
|
+
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
677
|
+
};
|
|
678
|
+
}
|
|
679
|
+
return upsertGeneratedDqlArtifactDraft(projectRoot, {
|
|
680
|
+
slug,
|
|
681
|
+
question: draftQuestion,
|
|
682
|
+
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
683
|
+
proposedDomain,
|
|
684
|
+
dqlArtifact,
|
|
685
|
+
sourceQuestion: followUp?.sourceQuestion,
|
|
686
|
+
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
687
|
+
followupKind: followUp?.kind,
|
|
688
|
+
outputs,
|
|
689
|
+
contextPackId: draftContextPack?.id,
|
|
690
|
+
routeIntent: String(intent),
|
|
691
|
+
validationWarnings,
|
|
692
|
+
});
|
|
693
|
+
}
|
|
637
694
|
if (!flags.save) {
|
|
638
695
|
return {
|
|
639
696
|
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
@@ -641,51 +698,30 @@ async function runEval(rest, flags) {
|
|
|
641
698
|
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
642
699
|
};
|
|
643
700
|
}
|
|
644
|
-
return
|
|
701
|
+
return upsertGeneratedDraft(projectRoot, {
|
|
645
702
|
slug,
|
|
646
703
|
question: draftQuestion,
|
|
704
|
+
proposedSql: sql,
|
|
647
705
|
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
648
706
|
proposedDomain,
|
|
649
|
-
|
|
707
|
+
proposedEntity,
|
|
708
|
+
sourceDqlArtifact,
|
|
650
709
|
sourceQuestion: followUp?.sourceQuestion,
|
|
651
710
|
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
652
711
|
followupKind: followUp?.kind,
|
|
712
|
+
requestedFilters,
|
|
713
|
+
requestedDimensions,
|
|
653
714
|
outputs,
|
|
654
715
|
contextPackId: draftContextPack?.id,
|
|
655
716
|
routeIntent: String(intent),
|
|
656
717
|
validationWarnings,
|
|
657
718
|
});
|
|
658
|
-
}
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
662
|
-
askedTimes: 0,
|
|
663
|
-
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
664
|
-
};
|
|
665
|
-
}
|
|
666
|
-
return upsertGeneratedDraft(projectRoot, {
|
|
667
|
-
slug,
|
|
668
|
-
question: draftQuestion,
|
|
669
|
-
proposedSql: sql,
|
|
670
|
-
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
671
|
-
proposedDomain,
|
|
672
|
-
proposedEntity,
|
|
673
|
-
sourceDqlArtifact,
|
|
674
|
-
sourceQuestion: followUp?.sourceQuestion,
|
|
675
|
-
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
676
|
-
followupKind: followUp?.kind,
|
|
677
|
-
requestedFilters,
|
|
678
|
-
requestedDimensions,
|
|
679
|
-
outputs,
|
|
680
|
-
contextPackId: draftContextPack?.id,
|
|
681
|
-
routeIntent: String(intent),
|
|
682
|
-
validationWarnings,
|
|
683
|
-
});
|
|
684
|
-
},
|
|
685
|
-
});
|
|
686
|
-
const evaluation = evaluateCase(testCase, result);
|
|
719
|
+
},
|
|
720
|
+
});
|
|
721
|
+
const evaluation = evaluateCase(testCase, result, runtimeProjection);
|
|
687
722
|
const durationMs = Date.now() - startedAt;
|
|
688
723
|
const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
|
|
724
|
+
const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
|
|
689
725
|
const judgeVerdict = judge
|
|
690
726
|
? await judgeAnswer({
|
|
691
727
|
question: testCase.question,
|
|
@@ -700,17 +736,33 @@ async function runEval(rest, flags) {
|
|
|
700
736
|
passed: evaluation.failures.length === 0,
|
|
701
737
|
failures: evaluation.failures,
|
|
702
738
|
durationMs,
|
|
739
|
+
// The persisted receipt is the only evidence for this metric. Rows and
|
|
740
|
+
// reader prose are intentionally ignored: a row-bearing answer can be
|
|
741
|
+
// skipped, while a deterministic fallback can render different wording.
|
|
742
|
+
narrationAttempted: narration.narrationAttempted,
|
|
743
|
+
narrationFallback: narration.narrationFallback,
|
|
703
744
|
executionMs: result.result?.executionTime,
|
|
704
745
|
executionMatched: evaluation.executionMatched,
|
|
705
746
|
...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
|
|
706
747
|
kind: result.kind,
|
|
707
|
-
route: result.contextPack?.routeDecision.route,
|
|
708
|
-
|
|
748
|
+
route: runtimeProjection?.route ?? result.contextPack?.routeDecision.route,
|
|
749
|
+
// Only the runtime driver can see the router's clarification options.
|
|
750
|
+
// In-process runs leave this undefined, so a clarify there scores as a
|
|
751
|
+
// dead end — the conservative reading, and another reason `--via runtime`
|
|
752
|
+
// is the truthful one.
|
|
753
|
+
...(runtimeRun ? {
|
|
754
|
+
clarificationOptionCount: runtimeRun.clarificationOptions?.length ?? 0,
|
|
755
|
+
conversational: runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational',
|
|
756
|
+
conversationalAnswer: (runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational')
|
|
757
|
+
&& Boolean(runtimeRun.answer?.trim()),
|
|
758
|
+
meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
|
|
759
|
+
} : {}),
|
|
760
|
+
intent: runtimeRun?.routeDecision?.category ?? result.contextPack?.routeDecision.intent,
|
|
709
761
|
reviewStatus: result.reviewStatus,
|
|
710
|
-
contextObjects: result.contextPack?.objects.length
|
|
762
|
+
contextObjects: runtimeProjection?.retrievalCandidateCount ?? result.contextPack?.objects.length,
|
|
711
763
|
followUp: Boolean(testCase.followUp),
|
|
712
764
|
draftSaved,
|
|
713
|
-
toolCalls: result.evidence?.toolCalls?.length ?? 0,
|
|
765
|
+
toolCalls: runtimeProjection?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0,
|
|
714
766
|
expected: testCase.expected,
|
|
715
767
|
validationCode: evaluation.validationCode,
|
|
716
768
|
trace: buildEvalTrace({
|
|
@@ -719,6 +771,7 @@ async function runEval(rest, flags) {
|
|
|
719
771
|
evaluation,
|
|
720
772
|
durationMs,
|
|
721
773
|
draftSaved,
|
|
774
|
+
runtime: runtimeProjection,
|
|
722
775
|
}),
|
|
723
776
|
});
|
|
724
777
|
}
|
|
@@ -729,16 +782,39 @@ async function runEval(rest, flags) {
|
|
|
729
782
|
}
|
|
730
783
|
const passed = results.filter((r) => r.passed).length;
|
|
731
784
|
const metrics = computeEvalMetrics(results);
|
|
785
|
+
// Runtime evals execute in a separate host, so its cassette directory is
|
|
786
|
+
// supplied explicitly by the eval workflow for reporting. The summary does
|
|
787
|
+
// not inspect prompts or provider credentials; it only classifies the
|
|
788
|
+
// checked-in response provenance.
|
|
789
|
+
const cassetteDirectory = via === 'runtime'
|
|
790
|
+
? process.env.DQL_EVAL_CASSETTE_DIR
|
|
791
|
+
: cassetteMode === 'record' || cassetteMode === 'replay'
|
|
792
|
+
? cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')
|
|
793
|
+
: undefined;
|
|
794
|
+
const cassetteEvidence = cassetteDirectory
|
|
795
|
+
? cassetteEvidenceSummary(new CassetteStore(cassetteDirectory))
|
|
796
|
+
: undefined;
|
|
732
797
|
const thresholds = {
|
|
733
798
|
minToolRequirement: flags.minToolRequirement ?? null,
|
|
734
799
|
minExecutionMatch: flags.minExecutionMatch ?? null,
|
|
735
800
|
minJudgePass: flags.minJudgePass ?? null,
|
|
736
801
|
maxWrongCertified: flags.maxWrongCertified ?? null,
|
|
802
|
+
maxFalseRefusal: flags.maxFalseRefusal ?? null,
|
|
803
|
+
minRefusalRecall: flags.minRefusalRecall ?? null,
|
|
804
|
+
minGroundedNarration: flags.minGroundedNarration ?? null,
|
|
737
805
|
};
|
|
738
806
|
const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
|
|
739
807
|
const ok = passed === results.length && thresholdsPassed;
|
|
740
808
|
if (flags.format === 'json') {
|
|
741
|
-
console.log(JSON.stringify({
|
|
809
|
+
console.log(JSON.stringify({
|
|
810
|
+
ok,
|
|
811
|
+
passed,
|
|
812
|
+
total: results.length,
|
|
813
|
+
thresholds,
|
|
814
|
+
metrics,
|
|
815
|
+
...(cassetteEvidence ? { cassetteEvidence } : {}),
|
|
816
|
+
results,
|
|
817
|
+
}, null, 2));
|
|
742
818
|
if (!ok)
|
|
743
819
|
process.exitCode = 1;
|
|
744
820
|
return;
|
|
@@ -752,6 +828,22 @@ async function runEval(rest, flags) {
|
|
|
752
828
|
console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
|
|
753
829
|
console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
|
|
754
830
|
console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
|
|
831
|
+
if (cassetteEvidence) {
|
|
832
|
+
console.log(`Cassette replay entries: ${cassetteEvidence.totalEntries} `
|
|
833
|
+
+ `(${cassetteEvidence.migratedLegacyDeterministicFixtureEntries} migrated legacy deterministic fixture, `
|
|
834
|
+
+ `${cassetteEvidence.syntheticDeterministicOrchestrationFixtureEntries} synthetic deterministic orchestration fixture).`);
|
|
835
|
+
console.log(cassetteEvidence.realProviderQualityEligible
|
|
836
|
+
? 'Real-provider quality evidence: eligible.'
|
|
837
|
+
: `Real-provider quality evidence: excluded (${cassetteEvidence.realProviderQualityExclusionReasons.join(', ')}).`);
|
|
838
|
+
}
|
|
839
|
+
console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
|
|
840
|
+
console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
|
|
841
|
+
if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
|
|
842
|
+
console.log(` ! Semantic judgment ran for only ${formatRate(metrics.meaning_resolved_rate)} of cases. `
|
|
843
|
+
+ 'Without a reachable provider DQL will not settle a reading by lexical rank (AGT-017), so ambiguous '
|
|
844
|
+
+ 'questions clarify by design — treat the clarification rate above as an artifact, not a product signal.');
|
|
845
|
+
}
|
|
846
|
+
console.log(`Refusal recall: ${formatRate(metrics.refusal_recall)} (${metrics.refusal_required_case_count} case(s) that must refuse)`);
|
|
755
847
|
console.log(`Execution match rate: ${formatRate(metrics.execution_match_rate)}`);
|
|
756
848
|
console.log(`Tool requirement pass rate: ${formatRate(metrics.tool_requirement_pass_rate)}`);
|
|
757
849
|
console.log(`Tool-observed case count: ${metrics.tool_observed_case_count}`);
|
|
@@ -767,6 +859,15 @@ async function runEval(rest, flags) {
|
|
|
767
859
|
if (thresholds.minJudgePass !== null) {
|
|
768
860
|
console.log(`Judge-pass threshold: ${thresholds.minJudgePass} (actual ${formatRate(metrics.judge_pass_rate)})`);
|
|
769
861
|
}
|
|
862
|
+
if (thresholds.maxFalseRefusal !== null) {
|
|
863
|
+
console.log(`False-refusal ceiling: ${thresholds.maxFalseRefusal} (actual ${formatRate(metrics.false_refusal_rate)})`);
|
|
864
|
+
}
|
|
865
|
+
if (thresholds.minRefusalRecall !== null) {
|
|
866
|
+
console.log(`Refusal-recall threshold: ${thresholds.minRefusalRecall} (actual ${formatRate(metrics.refusal_recall)})`);
|
|
867
|
+
}
|
|
868
|
+
if (thresholds.minGroundedNarration !== null && thresholds.minGroundedNarration !== undefined) {
|
|
869
|
+
console.log(`Grounded-narration threshold: ${thresholds.minGroundedNarration} (actual ${formatRate(metrics.grounded_narration_rate)} over ${metrics.grounded_narration_attempted} attempted)`);
|
|
870
|
+
}
|
|
770
871
|
if (thresholds.maxWrongCertified !== null) {
|
|
771
872
|
console.log(`Wrong-certified ceiling: ${thresholds.maxWrongCertified} (actual ${metrics.wrong_certified_count})`);
|
|
772
873
|
}
|
|
@@ -784,13 +885,39 @@ function previewGeneratedDraftPath(projectRoot, domain, slug) {
|
|
|
784
885
|
}
|
|
785
886
|
return `blocks/_drafts/${slug}.dql`;
|
|
786
887
|
}
|
|
787
|
-
function evaluateCase(testCase, result) {
|
|
888
|
+
function evaluateCase(testCase, result, runtime) {
|
|
788
889
|
const expected = testCase.expected;
|
|
789
890
|
if (!expected)
|
|
790
891
|
return { failures: [] };
|
|
791
892
|
const failures = [];
|
|
792
893
|
let validationCode;
|
|
793
894
|
let executionMatched;
|
|
895
|
+
// Answerability is asserted per case as well as aggregated into
|
|
896
|
+
// false_refusal_rate, so a single dead-end fails its own case instead of only
|
|
897
|
+
// nudging a rate someone has to notice.
|
|
898
|
+
const answerable = evalCaseIsAnswerable(expected);
|
|
899
|
+
// Three distinct outcomes, not two. An option-bearing clarification neither
|
|
900
|
+
// answers nor dead-ends: it must not fail an answerable case (its cost is
|
|
901
|
+
// tracked by `clarification_rate`), and it must not fail a must-refuse case
|
|
902
|
+
// either, because it did not assert anything about the data.
|
|
903
|
+
const clarifiedWithOptions = (result.clarificationOptions?.length ?? 0) > 0;
|
|
904
|
+
// A conversational reply ("I'm here to help you explore your data…") asserts
|
|
905
|
+
// nothing about the warehouse. For an out-of-scope question that is the CORRECT
|
|
906
|
+
// outcome — declining politely — so it must not be scored as an answer.
|
|
907
|
+
const conversational = result.answerKind === 'conversational';
|
|
908
|
+
const answerText = typeof result.text === 'string' ? result.text.trim() : '';
|
|
909
|
+
// Conversational replies split two ways, and the case's own expectation says
|
|
910
|
+
// which is right: for an answerable question a substantive reply IS the answer
|
|
911
|
+
// (a definition), and for an out-of-scope one it is the correct decline.
|
|
912
|
+
const conversationalAnswer = conversational && answerText.length > 0;
|
|
913
|
+
const producedDataAnswer = result.kind !== 'no_answer' && !conversational;
|
|
914
|
+
const deadEnded = !producedDataAnswer && !clarifiedWithOptions && !conversationalAnswer;
|
|
915
|
+
if (answerable === true && deadEnded) {
|
|
916
|
+
failures.push(`FALSE REFUSAL: this question is answerable, but the run dead-ended with no answer and no options${result.refusalCode ? ` (${result.refusalCode})` : ''}`);
|
|
917
|
+
}
|
|
918
|
+
if (answerable === false && producedDataAnswer) {
|
|
919
|
+
failures.push(`expected a refusal (question is out of scope / unanswerable), but the run answered with kind ${result.kind}`);
|
|
920
|
+
}
|
|
794
921
|
if (expected.kind && result.kind !== expected.kind)
|
|
795
922
|
failures.push(`kind expected ${expected.kind}, got ${result.kind}`);
|
|
796
923
|
if (expected.sourceTier && result.sourceTier !== expected.sourceTier)
|
|
@@ -799,12 +926,27 @@ function evaluateCase(testCase, result) {
|
|
|
799
926
|
failures.push(`certification expected ${expected.certification}, got ${result.certification}`);
|
|
800
927
|
if (expected.reviewStatus && result.reviewStatus !== expected.reviewStatus)
|
|
801
928
|
failures.push(`reviewStatus expected ${expected.reviewStatus}, got ${result.reviewStatus}`);
|
|
802
|
-
|
|
803
|
-
|
|
929
|
+
const observedRoute = runtime?.route ?? result.contextPack?.routeDecision.route;
|
|
930
|
+
if (expected.route && observedRoute !== expected.route)
|
|
931
|
+
failures.push(`route expected ${expected.route}, got ${observedRoute ?? 'none'}`);
|
|
804
932
|
if (expected.intent && result.contextPack?.routeDecision.intent !== expected.intent)
|
|
805
933
|
failures.push(`intent expected ${expected.intent}, got ${result.contextPack?.routeDecision.intent ?? 'none'}`);
|
|
806
|
-
|
|
934
|
+
// `modeling_gap` is a broad terminal kind. A relationship expectation is
|
|
935
|
+
// satisfied only by the router's persisted relationship-specific witness;
|
|
936
|
+
// otherwise a missing metric/dimension tuple would be misreported as a
|
|
937
|
+
// relationship repair opportunity.
|
|
938
|
+
const runtimeReportsMissingContext = expected.missingContextKind === 'relationship'
|
|
939
|
+
? runtime?.terminalOutcome?.gap?.code === 'MISSING_RELATIONSHIP'
|
|
940
|
+
: runtime?.terminalOutcome?.kind === 'modeling_gap'
|
|
941
|
+
&& expected.missingContextKind === 'modeling_gap';
|
|
942
|
+
if (expected.missingContextKind
|
|
943
|
+
&& !runtimeReportsMissingContext
|
|
944
|
+
&& !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind)) {
|
|
807
945
|
failures.push(`missing context kind ${expected.missingContextKind} was not reported`);
|
|
946
|
+
}
|
|
947
|
+
if (expected.terminalOutcomeKind && runtime?.terminalOutcome?.kind !== expected.terminalOutcomeKind) {
|
|
948
|
+
failures.push(`terminal outcome expected ${expected.terminalOutcomeKind}, got ${runtime?.terminalOutcome?.kind ?? 'none'}`);
|
|
949
|
+
}
|
|
808
950
|
for (const token of stringList(expected.sqlContains)) {
|
|
809
951
|
if (!result.proposedSql?.toLowerCase().includes(token.toLowerCase()))
|
|
810
952
|
failures.push(`SQL did not contain "${token}"`);
|
|
@@ -825,7 +967,7 @@ function evaluateCase(testCase, result) {
|
|
|
825
967
|
failures.push(`draftSaved expected ${expected.draftSaved}, got ${saved}`);
|
|
826
968
|
}
|
|
827
969
|
if (typeof expected.minToolCalls === 'number') {
|
|
828
|
-
const actualToolCalls = result.evidence?.toolCalls?.length ?? 0;
|
|
970
|
+
const actualToolCalls = runtime?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0;
|
|
829
971
|
if (actualToolCalls < expected.minToolCalls) {
|
|
830
972
|
failures.push(`toolCalls expected at least ${expected.minToolCalls}, got ${actualToolCalls}`);
|
|
831
973
|
}
|
|
@@ -849,7 +991,75 @@ function evaluateCase(testCase, result) {
|
|
|
849
991
|
}
|
|
850
992
|
return { failures, validationCode, executionMatched };
|
|
851
993
|
}
|
|
994
|
+
/**
|
|
995
|
+
* Is the case answerable? Explicit `expected.answerable` wins; otherwise infer
|
|
996
|
+
* from the expectations already present, so the metric covers legacy case files.
|
|
997
|
+
* A case with no expectations at all is excluded — it asserts nothing, so it can
|
|
998
|
+
* neither prove nor disprove a false refusal.
|
|
999
|
+
*/
|
|
1000
|
+
export function evalCaseIsAnswerable(expected) {
|
|
1001
|
+
if (!expected)
|
|
1002
|
+
return undefined;
|
|
1003
|
+
if (expected.answerable !== undefined)
|
|
1004
|
+
return expected.answerable;
|
|
1005
|
+
if (expected.kind === 'no_answer')
|
|
1006
|
+
return false;
|
|
1007
|
+
if (expected.sourceTier === 'no_answer')
|
|
1008
|
+
return false;
|
|
1009
|
+
if (expected.route === 'clarify' || expected.route === 'blocked')
|
|
1010
|
+
return false;
|
|
1011
|
+
if (Object.keys(expected).length === 0)
|
|
1012
|
+
return undefined;
|
|
1013
|
+
return true;
|
|
1014
|
+
}
|
|
1015
|
+
/**
|
|
1016
|
+
* Did the run leave the user with NO way forward?
|
|
1017
|
+
*
|
|
1018
|
+
* Deliberately narrower than "did not answer". A clarification that offers
|
|
1019
|
+
* selectable options is answerable on the next turn — worth minimising, tracked
|
|
1020
|
+
* separately as `clarification_rate`, but not the defect. A clarification with
|
|
1021
|
+
* ZERO options is a true dead end: the reported production loop was exactly
|
|
1022
|
+
* this, and a free-text reply to it reproduced the same question forever.
|
|
1023
|
+
*/
|
|
1024
|
+
export function evalResultRefused(result) {
|
|
1025
|
+
// A substantive conversational reply is an ANSWER, not a dead end. A governed
|
|
1026
|
+
// definition ("**top_customers** — Top 10 customers by lifetime spend…") is
|
|
1027
|
+
// exactly what a "what does X mean?" turn should return, and scoring it as a
|
|
1028
|
+
// refusal would report the feature working as the feature failing.
|
|
1029
|
+
if (result.conversationalAnswer)
|
|
1030
|
+
return false;
|
|
1031
|
+
// Order matters: the drivers collapse every clarification to `no_answer`
|
|
1032
|
+
// (it is not an answer), so the option check has to run FIRST or an
|
|
1033
|
+
// option-bearing clarification is miscounted as a dead end.
|
|
1034
|
+
if (result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0)
|
|
1035
|
+
return false;
|
|
1036
|
+
return result.kind === 'no_answer' || result.route === 'clarify';
|
|
1037
|
+
}
|
|
1038
|
+
/** Did the run ask an answerable clarification rather than answering outright? */
|
|
1039
|
+
export function evalResultClarified(result) {
|
|
1040
|
+
return result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0;
|
|
1041
|
+
}
|
|
1042
|
+
/**
|
|
1043
|
+
* Translate only the durable narration receipt into evaluation fields.
|
|
1044
|
+
*
|
|
1045
|
+
* Reader prose, row count, and result shape are deliberately absent: a skipped
|
|
1046
|
+
* narration can have rows, and a deterministic fallback can use any wording.
|
|
1047
|
+
*/
|
|
1048
|
+
export function narrationOutcomeForEval(receipt) {
|
|
1049
|
+
if (receipt?.mode !== 'verified_facts' || !receipt.attempted)
|
|
1050
|
+
return {};
|
|
1051
|
+
return {
|
|
1052
|
+
// A durable receipt is the only source of this metric. An infrastructure
|
|
1053
|
+
// error started a verified narration but did not produce a grounded answer,
|
|
1054
|
+
// so it belongs in the denominator just like the deterministic floor. The
|
|
1055
|
+
// old mapping treated it as a success because only fallback was negative.
|
|
1056
|
+
narrationAttempted: true,
|
|
1057
|
+
narrationFallback: receipt.outcome !== 'success',
|
|
1058
|
+
};
|
|
1059
|
+
}
|
|
852
1060
|
function computeEvalMetrics(results) {
|
|
1061
|
+
const answerableCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === true);
|
|
1062
|
+
const refusalRequiredCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === false);
|
|
853
1063
|
const certifiedCases = results.filter((result) => result.expected?.kind === 'certified' ||
|
|
854
1064
|
result.expected?.certification === 'certified' ||
|
|
855
1065
|
result.expected?.route === 'certified');
|
|
@@ -874,10 +1084,61 @@ function computeEvalMetrics(results) {
|
|
|
874
1084
|
wrong_certified_count: results.filter((result) => result.kind === 'certified' &&
|
|
875
1085
|
(result.expected?.kind ? result.expected.kind !== 'certified' : result.followUp)).length,
|
|
876
1086
|
outside_context_rejection_count: results.filter((result) => result.validationCode === 'unknown_relation' || result.validationCode === 'unknown_column').length,
|
|
1087
|
+
/**
|
|
1088
|
+
* THE headline number: how often an answerable question was refused.
|
|
1089
|
+
* Bounds every other quality metric — a run that refuses cannot be wrong,
|
|
1090
|
+
* so a falling false-refusal rate must be read together with
|
|
1091
|
+
* `execution_match_rate` to be sure refusals were replaced by CORRECT answers.
|
|
1092
|
+
*/
|
|
1093
|
+
false_refusal_rate: ratio(answerableCases.filter(evalResultRefused).length, answerableCases.length),
|
|
1094
|
+
false_refusal_count: answerableCases.filter(evalResultRefused).length,
|
|
1095
|
+
answerable_case_count: answerableCases.length,
|
|
1096
|
+
/**
|
|
1097
|
+
* Answerable cases that asked an option-bearing clarification instead of
|
|
1098
|
+
* answering. Not a defect, but a direct cost in turns — read it next to
|
|
1099
|
+
* false_refusal_rate so a fall in refusals is not just a rise in questions.
|
|
1100
|
+
*/
|
|
1101
|
+
clarification_rate: ratio(answerableCases.filter(evalResultClarified).length, answerableCases.length),
|
|
1102
|
+
/**
|
|
1103
|
+
* Cases where semantic judgment ran. Without a provider `mayAssumeInterpretation`
|
|
1104
|
+
* is false (AGT-017), so every ambiguous question clarifies by design and
|
|
1105
|
+
* `clarification_rate` says nothing about product quality.
|
|
1106
|
+
*/
|
|
1107
|
+
meaning_resolved_rate: ratio(results.filter((result) => result.meaningResolved === true).length, results.length),
|
|
1108
|
+
/**
|
|
1109
|
+
* Latency, which the acceptance matrix asked for and nothing measured. A
|
|
1110
|
+
* quality gain paid for entirely in wall clock is not a gain: the plan's
|
|
1111
|
+
* two-tier target is certified/semantic under 5s while research takes
|
|
1112
|
+
* minutes, and only a per-class p95 can tell those apart from a regression.
|
|
1113
|
+
*/
|
|
1114
|
+
latency_p50_ms: percentileMs(results, 0.5),
|
|
1115
|
+
latency_p95_ms: percentileMs(results, 0.95),
|
|
1116
|
+
latency_p95_answerable_ms: percentileMs(answerableCases, 0.95),
|
|
1117
|
+
/**
|
|
1118
|
+
* How often verified narration survived. When the drafted narration fails
|
|
1119
|
+
* its fact check the reader gets the deterministic record under a
|
|
1120
|
+
* disclaimer — correct, but visibly worse. A silent fall here is exactly
|
|
1121
|
+
* the truncation defect that shipped unnoticed, so it is measured.
|
|
1122
|
+
*/
|
|
1123
|
+
grounded_narration_rate: ratio(results.filter((result) => result.narrationAttempted && result.narrationFallback === false).length, results.filter((result) => result.narrationAttempted).length),
|
|
1124
|
+
grounded_narration_attempted: results.filter((result) => result.narrationAttempted).length,
|
|
1125
|
+
/**
|
|
1126
|
+
* The guard on the above: cases that must NOT produce a data answer.
|
|
1127
|
+
* Scored on "did not answer" rather than "dead-ended", because declining via
|
|
1128
|
+
* a clarification is still declining — what would be wrong is asserting
|
|
1129
|
+
* something about data the project does not have.
|
|
1130
|
+
*/
|
|
1131
|
+
refusal_recall: ratio(refusalRequiredCases.filter((result) => result.kind === 'no_answer' || result.conversational === true).length, refusalRequiredCases.length),
|
|
1132
|
+
refusal_required_case_count: refusalRequiredCases.length,
|
|
877
1133
|
draft_saved_count: results.filter((result) => result.draftSaved).length,
|
|
878
1134
|
tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
|
|
879
1135
|
avg_tool_calls: average(toolCallCounts),
|
|
880
|
-
|
|
1136
|
+
// Runtime runs only report this when the persisted router recorded an
|
|
1137
|
+
// explicit retrieval count. Treat absent evidence as unknown, not as an
|
|
1138
|
+
// invented empty context pack.
|
|
1139
|
+
avg_context_objects: average(results
|
|
1140
|
+
.map((result) => result.contextObjects)
|
|
1141
|
+
.filter((count) => typeof count === 'number')),
|
|
881
1142
|
avg_execution_ms: executionTimes.length ? average(executionTimes) : null,
|
|
882
1143
|
};
|
|
883
1144
|
}
|
|
@@ -885,20 +1146,27 @@ function agentEvalThresholdsPass(metrics, thresholds) {
|
|
|
885
1146
|
// A rate threshold with no applicable cases (metric === null) is vacuously
|
|
886
1147
|
// satisfied — you only fail when the metric exists and falls below the bar.
|
|
887
1148
|
const rateOk = (metric, min) => min === null || min === undefined || metric === null || metric >= min;
|
|
888
|
-
|
|
1149
|
+
// A ceiling is only meaningful when the metric has data; `null` means no
|
|
1150
|
+
// answerable case was scored, which is "unknown", not "perfect".
|
|
1151
|
+
const ceilingOk = (metric, max) => max === null || max === undefined || metric === null || metric <= max;
|
|
1152
|
+
return rateOk(metrics.grounded_narration_rate, thresholds.minGroundedNarration)
|
|
1153
|
+
&& rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
|
|
889
1154
|
&& rateOk(metrics.execution_match_rate, thresholds.minExecutionMatch)
|
|
890
1155
|
&& rateOk(metrics.judge_pass_rate, thresholds.minJudgePass)
|
|
1156
|
+
&& ceilingOk(metrics.false_refusal_rate, thresholds.maxFalseRefusal)
|
|
1157
|
+
&& rateOk(metrics.refusal_recall, thresholds.minRefusalRecall)
|
|
891
1158
|
&& (thresholds.maxWrongCertified === null
|
|
892
1159
|
|| thresholds.maxWrongCertified === undefined
|
|
893
1160
|
|| metrics.wrong_certified_count <= thresholds.maxWrongCertified);
|
|
894
1161
|
}
|
|
895
1162
|
function buildEvalTrace(input) {
|
|
896
|
-
const { testCase, result, evaluation, durationMs, draftSaved } = input;
|
|
1163
|
+
const { testCase, result, evaluation, durationMs, draftSaved, runtime } = input;
|
|
897
1164
|
const routeDecision = result.contextPack?.routeDecision;
|
|
898
1165
|
const selectedRelations = result.contextPack?.retrievalDiagnostics.selectedRelations ?? [];
|
|
899
1166
|
const allowedRelations = result.contextPack?.allowedSqlContext?.relations ?? [];
|
|
900
1167
|
const followUp = testCase.followUp;
|
|
901
1168
|
const toolCalls = result.evidence?.toolCalls ?? [];
|
|
1169
|
+
const observedToolCallCount = runtime?.toolCallCount ?? toolCalls.length;
|
|
902
1170
|
const routeEvidence = result.evidence?.route ?? [];
|
|
903
1171
|
const executionStatus = result.executionError
|
|
904
1172
|
? 'failed'
|
|
@@ -914,33 +1182,44 @@ function buildEvalTrace(input) {
|
|
|
914
1182
|
const rowsExpected = testCase.expected?.rows !== undefined;
|
|
915
1183
|
const expectedMinToolCalls = testCase.expected?.minToolCalls;
|
|
916
1184
|
const toolStatus = typeof expectedMinToolCalls === 'number'
|
|
917
|
-
?
|
|
918
|
-
:
|
|
1185
|
+
? observedToolCallCount >= expectedMinToolCalls ? 'passed' : 'failed'
|
|
1186
|
+
: observedToolCallCount > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
|
|
919
1187
|
const toolMessage = typeof expectedMinToolCalls === 'number'
|
|
920
|
-
?
|
|
921
|
-
? `Observed ${
|
|
922
|
-
: `Observed ${
|
|
923
|
-
:
|
|
924
|
-
? `Observed ${
|
|
1188
|
+
? observedToolCallCount >= expectedMinToolCalls
|
|
1189
|
+
? `Observed ${observedToolCallCount} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
|
|
1190
|
+
: `Observed ${observedToolCallCount} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
|
|
1191
|
+
: observedToolCallCount > 0
|
|
1192
|
+
? `Observed ${observedToolCallCount} provider tool call(s).`
|
|
925
1193
|
: routeEvidence.length > 0
|
|
926
1194
|
? `Captured ${routeEvidence.length} deterministic route evidence step(s).`
|
|
927
1195
|
: 'No provider tool calls were observed for this answer.';
|
|
928
1196
|
return [
|
|
929
1197
|
{
|
|
930
1198
|
stage: 'context',
|
|
931
|
-
status:
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
1199
|
+
status: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
1200
|
+
? 'passed'
|
|
1201
|
+
: result.contextPack ? 'passed' : 'not_run',
|
|
1202
|
+
message: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
1203
|
+
? `Persisted route evidence recorded ${runtime.retrievalCandidateCount ?? 'an unspecified number of'} retrieved candidate(s).`
|
|
1204
|
+
: result.contextPack
|
|
1205
|
+
? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
|
|
1206
|
+
: 'No context pack was attached to the answer.',
|
|
1207
|
+
payload: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
936
1208
|
? {
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
missingContext: result.contextPack.missingContext,
|
|
1209
|
+
evidenceSource: 'persisted_agent_run',
|
|
1210
|
+
retrievalCandidateCount: runtime.retrievalCandidateCount,
|
|
1211
|
+
sourceCoverage: runtime.sourceCoverage,
|
|
1212
|
+
terminalOutcome: runtime.terminalOutcome,
|
|
942
1213
|
}
|
|
943
|
-
:
|
|
1214
|
+
: result.contextPack
|
|
1215
|
+
? {
|
|
1216
|
+
contextPackId: result.contextPack.id,
|
|
1217
|
+
selectedObjectCount: result.contextPack.objects.length,
|
|
1218
|
+
allowedRelationCount: allowedRelations.length,
|
|
1219
|
+
selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
|
|
1220
|
+
missingContext: result.contextPack.missingContext,
|
|
1221
|
+
}
|
|
1222
|
+
: undefined,
|
|
944
1223
|
},
|
|
945
1224
|
{
|
|
946
1225
|
stage: 'rewrite',
|
|
@@ -952,29 +1231,40 @@ function buildEvalTrace(input) {
|
|
|
952
1231
|
},
|
|
953
1232
|
{
|
|
954
1233
|
stage: 'lane',
|
|
955
|
-
status: routeDecision ? 'passed' : 'not_run',
|
|
956
|
-
message:
|
|
957
|
-
? `
|
|
958
|
-
:
|
|
959
|
-
|
|
1234
|
+
status: runtime ? 'passed' : routeDecision ? 'passed' : 'not_run',
|
|
1235
|
+
message: runtime
|
|
1236
|
+
? `Persisted engine route ${runtime.runRoute}${runtime.route ? ` evaluated as ${runtime.route}` : ''}.`
|
|
1237
|
+
: routeDecision
|
|
1238
|
+
? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
|
|
1239
|
+
: 'No lane decision was attached to the answer.',
|
|
1240
|
+
payload: runtime
|
|
960
1241
|
? {
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
exactObjectKey: routeDecision.exactObjectKey,
|
|
1242
|
+
engineRoute: runtime.runRoute,
|
|
1243
|
+
evalRoute: runtime.route,
|
|
1244
|
+
status: runtime.status,
|
|
1245
|
+
trustState: runtime.trustState,
|
|
1246
|
+
terminalOutcome: runtime.terminalOutcome,
|
|
967
1247
|
}
|
|
968
|
-
:
|
|
1248
|
+
: routeDecision
|
|
1249
|
+
? {
|
|
1250
|
+
route: routeDecision.route,
|
|
1251
|
+
intent: routeDecision.intent,
|
|
1252
|
+
reason: routeDecision.reason,
|
|
1253
|
+
trustLabel: routeDecision.trustLabel,
|
|
1254
|
+
reviewStatus: routeDecision.reviewStatus,
|
|
1255
|
+
exactObjectKey: routeDecision.exactObjectKey,
|
|
1256
|
+
}
|
|
1257
|
+
: undefined,
|
|
969
1258
|
},
|
|
970
1259
|
{
|
|
971
1260
|
stage: 'tools',
|
|
972
1261
|
status: toolStatus,
|
|
973
1262
|
message: toolMessage,
|
|
974
1263
|
payload: {
|
|
975
|
-
observedToolCalls:
|
|
1264
|
+
observedToolCalls: observedToolCallCount,
|
|
976
1265
|
expectedMinToolCalls,
|
|
977
|
-
|
|
1266
|
+
...(runtime ? { evidenceSource: 'persisted_agent_run.telemetry' } : {}),
|
|
1267
|
+
providerToolCalls: runtime ? [] : toolCalls.slice(0, 12).map((call) => ({
|
|
978
1268
|
order: call.order,
|
|
979
1269
|
name: call.name,
|
|
980
1270
|
status: call.status,
|
|
@@ -1207,6 +1497,20 @@ export const __test__ = {
|
|
|
1207
1497
|
cliAnalysisDepth,
|
|
1208
1498
|
cliReasoningEffort,
|
|
1209
1499
|
computeEvalMetrics,
|
|
1500
|
+
narrationOutcomeForEval,
|
|
1210
1501
|
evaluateCase,
|
|
1211
1502
|
};
|
|
1503
|
+
/** Percentile over observed case durations. Returns null when nothing timed. */
|
|
1504
|
+
function percentileMs(results, q) {
|
|
1505
|
+
const observed = results
|
|
1506
|
+
.map((result) => result.durationMs)
|
|
1507
|
+
.filter((value) => typeof value === 'number' && Number.isFinite(value) && value >= 0)
|
|
1508
|
+
.sort((left, right) => left - right);
|
|
1509
|
+
if (observed.length === 0)
|
|
1510
|
+
return null;
|
|
1511
|
+
// Nearest-rank: with a handful of cases an interpolated percentile invents a
|
|
1512
|
+
// duration nothing actually took.
|
|
1513
|
+
const rank = Math.min(observed.length - 1, Math.max(0, Math.ceil(q * observed.length) - 1));
|
|
1514
|
+
return observed[rank] ?? null;
|
|
1515
|
+
}
|
|
1212
1516
|
//# sourceMappingURL=agent.js.map
|