@duckcodeailabs/dql-cli 1.14.0 → 1.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/args.d.ts +15 -0
- package/dist/args.d.ts.map +1 -1
- package/dist/args.js +25 -0
- package/dist/args.js.map +1 -1
- package/dist/assets/dql-notebook/assets/{AgentLogPage-Ch7VK20X.js → AgentLogPage-BPz-UWFh.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildDialog-DBr5TmyM.js → AiBuildDialog-BEl53WA_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildResult-jLPzQO7O.js → AiBuildResult-B4yfGTTZ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiSidePanel-CdlGsiVC.js → AiSidePanel-CSZAAvuD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AnalyticsHome-C0DXbOwY.js → AnalyticsHome-D5P6Ujwi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AppsView-CcpwjApv.js → AppsView-DQOwU9Cg.js} +4 -4
- package/dist/assets/dql-notebook/assets/{BlockStudio-CFYxafw-.js → BlockStudio-B4ap0GdY.js} +1 -1
- package/dist/assets/dql-notebook/assets/{BusinessArtifactView-C0kLYg2p.js → BusinessArtifactView-BIfNI-0S.js} +1 -1
- package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-CMwElXD_.js → DbtFirstModelingPage-D72byb2g.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GitPage-JjhRDeWY.js → GitPage-lQb1uXH2.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GlobalAiRail-DakE4NdR.js → GlobalAiRail-CVECf6Xj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GovernedContextPage-BokDqG6a.js → GovernedContextPage-trOyMCY6.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HelpDocsPage-CjOv6_gz.js → HelpDocsPage-D8hLS5lE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HomePage-nGaNcwdw.js → HomePage-eIkfBIep.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDAG-CO6CFJRg.js → LineageDAG-CSqcbDrE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDetailView-BnGb7OF7.js → LineageDetailView-DJZZjZu-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDrawer-C5Y0Ht0b.js → LineageDrawer-BcIipIc3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-Cgh3GR3F.js → LineagePathBreadcrumb-CviIf8PN.js} +1 -1
- package/dist/assets/dql-notebook/assets/{MiniLineageGraph-Kla9PYuj.js → MiniLineageGraph-vH_MY_Ju.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewBlockModal-BR2SnmPT.js → NewBlockModal-DbCQg-pj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewNotebookModal-BaCHMYeB.js → NewNotebookModal-5Vp6XiuK.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NotebookEditor-CBLqY8cE.js → NotebookEditor-DjqJS44s.js} +1 -1
- package/dist/assets/dql-notebook/assets/{ReadinessPage-CiN0IWSS.js → ReadinessPage-BHGrC5ho.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SetupOnboarding-BhiCsYF-.js → SetupOnboarding-BfS9Tdx-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SkillsPage-CCSf8VMm.js → SkillsPage-BFH21wSj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{TrustBadge-BgQmFe_x.js → TrustBadge-zm6g_SxZ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{UnifiedAgentRunPanel-Bw5AXNDB.js → UnifiedAgentRunPanel--oxmjlgr.js} +21 -21
- package/dist/assets/dql-notebook/assets/{answer-to-notebook-AeDYUDla.js → answer-to-notebook-DPhxIEzF.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-left-C55x2hq_.js → arrow-left-DNEb86Xc.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-right-C1cJrhOm.js → arrow-right-DpPWbwaD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{book-open-text-CQf_sdv2.js → book-open-text-D7s5jo4X.js} +1 -1
- package/dist/assets/dql-notebook/assets/{circle-x-X8-Z2yLY.js → circle-x-Db3dXKg7.js} +1 -1
- package/dist/assets/dql-notebook/assets/{dagre.esm-BjjNYKyY.js → dagre.esm-C7pppQ1a.js} +1 -1
- package/dist/assets/dql-notebook/assets/{external-link-C2zrz5DH.js → external-link-BxwXitO_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{grip-vertical-qGV_PYGU.js → grip-vertical-Dt4lkWRi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{index-ByTDPDaH.js → index-DKo-bwNw.js} +2 -2
- package/dist/assets/dql-notebook/assets/{link-2-VpyOxQXG.js → link-2-Dfo2P6wi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{list-tree-CH2Jhwms.js → list-tree-DiTmIWAL.js} +1 -1
- package/dist/assets/dql-notebook/assets/{minimize-2-B2TZJ8BT.js → minimize-2-CMTAkPzL.js} +1 -1
- package/dist/assets/dql-notebook/assets/{panel-right-open-BunF88lt.js → panel-right-open-DwYr7FW4.js} +1 -1
- package/dist/assets/dql-notebook/assets/{play-DAFVF4_G.js → play-BXhHYQ4x.js} +1 -1
- package/dist/assets/dql-notebook/assets/{rotate-ccw-DGCrrqtY.js → rotate-ccw-BNi6F8pl.js} +1 -1
- package/dist/assets/dql-notebook/assets/{semantic-fields-Ci-9QL9F.js → semantic-fields-CNOGysAy.js} +1 -1
- package/dist/assets/dql-notebook/assets/{sliders-horizontal-BOAlXXbn.js → sliders-horizontal-Ec5MUMUW.js} +1 -1
- package/dist/assets/dql-notebook/assets/{star-CkksZXHt.js → star-B9leDkp_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{triangle-alert-D3mjyJZE.js → triangle-alert-BefTYCzx.js} +1 -1
- package/dist/assets/dql-notebook/assets/{upload-CTNOAVEO.js → upload-SPiOM2tQ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-DiQjc7x-.js → usePersistedAgentThreadId-C4foXeiQ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{user-round-BWd5tQRg.js → user-round-comGmyw-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wand-sparkles-CqsAv8P-.js → wand-sparkles-BffR4dF8.js} +1 -1
- package/dist/assets/dql-notebook/assets/{workflow-CSqsj-sC.js → workflow-ChmPTEzH.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wrench-DWqzqlX8.js → wrench-DovaG_ze.js} +1 -1
- package/dist/assets/dql-notebook/assets/{x-65M5rLCE.js → x-nRx91AgW.js} +1 -1
- package/dist/assets/dql-notebook/index.html +1 -1
- package/dist/commands/agent-eval-cassette.d.ts +73 -0
- package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
- package/dist/commands/agent-eval-cassette.js +170 -0
- package/dist/commands/agent-eval-cassette.js.map +1 -0
- package/dist/commands/agent-eval-runtime.d.ts +97 -0
- package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
- package/dist/commands/agent-eval-runtime.js +155 -0
- package/dist/commands/agent-eval-runtime.js.map +1 -0
- package/dist/commands/agent.d.ts +115 -1
- package/dist/commands/agent.d.ts.map +1 -1
- package/dist/commands/agent.js +289 -62
- package/dist/commands/agent.js.map +1 -1
- package/dist/index.js +4 -0
- package/dist/index.js.map +1 -1
- package/dist/llm/analyst-loop-tools.d.ts +19 -0
- package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
- package/dist/llm/analyst-loop-tools.js +56 -0
- package/dist/llm/analyst-loop-tools.js.map +1 -0
- package/dist/llm/providers/dql-agent-provider.d.ts +24 -1
- package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
- package/dist/llm/providers/dql-agent-provider.js +306 -7
- package/dist/llm/providers/dql-agent-provider.js.map +1 -1
- package/dist/llm/types.d.ts +19 -1
- package/dist/llm/types.d.ts.map +1 -1
- package/dist/local-runtime.d.ts +108 -1
- package/dist/local-runtime.d.ts.map +1 -1
- package/dist/local-runtime.js +626 -110
- package/dist/local-runtime.js.map +1 -1
- package/dist/package.json +10 -10
- package/package.json +10 -10
package/dist/commands/agent.js
CHANGED
|
@@ -21,6 +21,8 @@
|
|
|
21
21
|
* dql agent feedback <up|down> --block <id> --question "..."
|
|
22
22
|
* Records feedback into the KG. Used by clients without MCP access.
|
|
23
23
|
*/
|
|
24
|
+
import { answerFromRuntimeRun, driveViaRuntime, evalRouteForRun } from './agent-eval-runtime.js';
|
|
25
|
+
import { CassetteStore, cassetteDirFor, withCassette } from './agent-eval-cassette.js';
|
|
24
26
|
import { existsSync, readFileSync } from 'node:fs';
|
|
25
27
|
import { join, resolve } from 'node:path';
|
|
26
28
|
import { load as loadYaml } from 'js-yaml';
|
|
@@ -551,7 +553,14 @@ async function runEval(rest, flags) {
|
|
|
551
553
|
if (!existsSync(kgPath))
|
|
552
554
|
await reindexProject(projectRoot, { kgPath });
|
|
553
555
|
const providerName = flags.provider;
|
|
554
|
-
const
|
|
556
|
+
const rawProvider = await pickProvider(providerName);
|
|
557
|
+
// Cassettes make a provider-backed suite repeatable and free to re-run. They
|
|
558
|
+
// apply to the in-process driver here; `--via runtime` needs the SERVER
|
|
559
|
+
// started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
|
|
560
|
+
const cassetteMode = flags.cassette;
|
|
561
|
+
const provider = cassetteMode === 'record' || cassetteMode === 'replay'
|
|
562
|
+
? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode)
|
|
563
|
+
: rawProvider;
|
|
555
564
|
const reasoningEffort = cliReasoningEffort(flags);
|
|
556
565
|
const requestedDepth = cliAnalysisDepth(flags);
|
|
557
566
|
const kg = new KGStore(kgPath);
|
|
@@ -562,10 +571,22 @@ async function runEval(rest, flags) {
|
|
|
562
571
|
// gracefully when no provider is available so offline eval stays deterministic.
|
|
563
572
|
const judge = Boolean(flags.judge);
|
|
564
573
|
const judgeComplete = async ({ system, user }) => provider.generate([{ role: 'system', content: system }, { role: 'user', content: user }], {});
|
|
574
|
+
// Which half of the stack is under test. `loop` preserves today's behaviour;
|
|
575
|
+
// `runtime` is the one that exercises routing and gates end to end.
|
|
576
|
+
const via = flags.via === 'runtime' ? 'runtime' : 'loop';
|
|
565
577
|
const runtimeBase = flags.runtimeUrl
|
|
566
578
|
?? flags.runtime
|
|
567
579
|
?? process.env.DQL_RUNTIME_URL
|
|
568
580
|
?? 'http://127.0.0.1:3474';
|
|
581
|
+
if (via === 'runtime') {
|
|
582
|
+
// Fail fast and loudly. Without this, an unreachable server turns every case
|
|
583
|
+
// into a transport error and the report reads as a false-refusal spike that
|
|
584
|
+
// no code change caused.
|
|
585
|
+
const probe = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/health`).catch(() => null);
|
|
586
|
+
if (!probe?.ok) {
|
|
587
|
+
throw new Error(`--via runtime needs a running server at ${runtimeBase}. Start one with \`dql serve\`, or use --via loop to score the answer loop in-process.`);
|
|
588
|
+
}
|
|
589
|
+
}
|
|
569
590
|
const semanticLayer = loadAgentSemanticLayer(projectRoot);
|
|
570
591
|
const expandGroundingContext = createGroundingContextExpander(projectRoot);
|
|
571
592
|
const answerLoopTools = buildAnswerLoopTools(projectRoot);
|
|
@@ -605,35 +626,66 @@ async function runEval(rest, flags) {
|
|
|
605
626
|
}
|
|
606
627
|
: undefined,
|
|
607
628
|
}).catch(() => undefined);
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
629
|
+
// `--via runtime` posts to a running `dql serve` so the case exercises the
|
|
630
|
+
// router, engine, plan boundary, and gates. The in-process driver below
|
|
631
|
+
// calls the answer loop directly and cannot observe any of them, which is
|
|
632
|
+
// why a refusal metric taken from it reads cleaner than users experience.
|
|
633
|
+
const runtimeRun = via === 'runtime'
|
|
634
|
+
? await driveViaRuntime({ runtimeBase, question: testCase.question })
|
|
635
|
+
: undefined;
|
|
636
|
+
const result = runtimeRun
|
|
637
|
+
? answerFromRuntimeRun(runtimeRun)
|
|
638
|
+
: await answer({
|
|
639
|
+
question: testCase.question,
|
|
640
|
+
domain: testCase.domain,
|
|
641
|
+
domainContext: testCase.domain && manifest
|
|
642
|
+
? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
|
|
643
|
+
: undefined,
|
|
644
|
+
provider,
|
|
645
|
+
kg,
|
|
646
|
+
manifest: manifest ?? undefined,
|
|
647
|
+
skills,
|
|
648
|
+
memoryContext,
|
|
649
|
+
followUp: testCase.followUp,
|
|
650
|
+
semanticLayer,
|
|
651
|
+
schemaContext,
|
|
652
|
+
contextPack,
|
|
653
|
+
reasoningEffort,
|
|
654
|
+
analysisDepth: contextBudget.analysisDepth,
|
|
655
|
+
expandGroundingContext,
|
|
656
|
+
answerLoopTools,
|
|
657
|
+
executeCertifiedBlock: execute && manifest
|
|
658
|
+
? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
|
|
659
|
+
: undefined,
|
|
660
|
+
executeGeneratedSql: execute
|
|
661
|
+
? createGeneratedSqlExecutor(runtimeBase)
|
|
662
|
+
: undefined,
|
|
663
|
+
captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
|
|
664
|
+
const slug = deriveGeneratedDraftSlug(draftQuestion);
|
|
665
|
+
const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
|
|
666
|
+
if (dqlArtifact?.kind === 'semantic_block') {
|
|
667
|
+
if (!flags.save) {
|
|
668
|
+
return {
|
|
669
|
+
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
670
|
+
askedTimes: 0,
|
|
671
|
+
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
672
|
+
};
|
|
673
|
+
}
|
|
674
|
+
return upsertGeneratedDqlArtifactDraft(projectRoot, {
|
|
675
|
+
slug,
|
|
676
|
+
question: draftQuestion,
|
|
677
|
+
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
678
|
+
proposedDomain,
|
|
679
|
+
dqlArtifact,
|
|
680
|
+
sourceQuestion: followUp?.sourceQuestion,
|
|
681
|
+
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
682
|
+
followupKind: followUp?.kind,
|
|
683
|
+
outputs,
|
|
684
|
+
contextPackId: draftContextPack?.id,
|
|
685
|
+
routeIntent: String(intent),
|
|
686
|
+
validationWarnings,
|
|
687
|
+
});
|
|
688
|
+
}
|
|
637
689
|
if (!flags.save) {
|
|
638
690
|
return {
|
|
639
691
|
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
@@ -641,51 +693,30 @@ async function runEval(rest, flags) {
|
|
|
641
693
|
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
642
694
|
};
|
|
643
695
|
}
|
|
644
|
-
return
|
|
696
|
+
return upsertGeneratedDraft(projectRoot, {
|
|
645
697
|
slug,
|
|
646
698
|
question: draftQuestion,
|
|
699
|
+
proposedSql: sql,
|
|
647
700
|
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
648
701
|
proposedDomain,
|
|
649
|
-
|
|
702
|
+
proposedEntity,
|
|
703
|
+
sourceDqlArtifact,
|
|
650
704
|
sourceQuestion: followUp?.sourceQuestion,
|
|
651
705
|
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
652
706
|
followupKind: followUp?.kind,
|
|
707
|
+
requestedFilters,
|
|
708
|
+
requestedDimensions,
|
|
653
709
|
outputs,
|
|
654
710
|
contextPackId: draftContextPack?.id,
|
|
655
711
|
routeIntent: String(intent),
|
|
656
712
|
validationWarnings,
|
|
657
713
|
});
|
|
658
|
-
}
|
|
659
|
-
|
|
660
|
-
return {
|
|
661
|
-
path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
|
|
662
|
-
askedTimes: 0,
|
|
663
|
-
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
664
|
-
};
|
|
665
|
-
}
|
|
666
|
-
return upsertGeneratedDraft(projectRoot, {
|
|
667
|
-
slug,
|
|
668
|
-
question: draftQuestion,
|
|
669
|
-
proposedSql: sql,
|
|
670
|
-
proposedContractId: `${proposedDomain}.Unknown.${slug}`,
|
|
671
|
-
proposedDomain,
|
|
672
|
-
proposedEntity,
|
|
673
|
-
sourceDqlArtifact,
|
|
674
|
-
sourceQuestion: followUp?.sourceQuestion,
|
|
675
|
-
sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
|
|
676
|
-
followupKind: followUp?.kind,
|
|
677
|
-
requestedFilters,
|
|
678
|
-
requestedDimensions,
|
|
679
|
-
outputs,
|
|
680
|
-
contextPackId: draftContextPack?.id,
|
|
681
|
-
routeIntent: String(intent),
|
|
682
|
-
validationWarnings,
|
|
683
|
-
});
|
|
684
|
-
},
|
|
685
|
-
});
|
|
714
|
+
},
|
|
715
|
+
});
|
|
686
716
|
const evaluation = evaluateCase(testCase, result);
|
|
687
717
|
const durationMs = Date.now() - startedAt;
|
|
688
718
|
const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
|
|
719
|
+
const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
|
|
689
720
|
const judgeVerdict = judge
|
|
690
721
|
? await judgeAnswer({
|
|
691
722
|
question: testCase.question,
|
|
@@ -700,11 +731,27 @@ async function runEval(rest, flags) {
|
|
|
700
731
|
passed: evaluation.failures.length === 0,
|
|
701
732
|
failures: evaluation.failures,
|
|
702
733
|
durationMs,
|
|
734
|
+
// The persisted receipt is the only evidence for this metric. Rows and
|
|
735
|
+
// reader prose are intentionally ignored: a row-bearing answer can be
|
|
736
|
+
// skipped, while a deterministic fallback can render different wording.
|
|
737
|
+
narrationAttempted: narration.narrationAttempted,
|
|
738
|
+
narrationFallback: narration.narrationFallback,
|
|
703
739
|
executionMs: result.result?.executionTime,
|
|
704
740
|
executionMatched: evaluation.executionMatched,
|
|
705
741
|
...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
|
|
706
742
|
kind: result.kind,
|
|
707
|
-
route: result.contextPack?.routeDecision.route,
|
|
743
|
+
route: runtimeRun ? evalRouteForRun(runtimeRun.route) : result.contextPack?.routeDecision.route,
|
|
744
|
+
// Only the runtime driver can see the router's clarification options.
|
|
745
|
+
// In-process runs leave this undefined, so a clarify there scores as a
|
|
746
|
+
// dead end — the conservative reading, and another reason `--via runtime`
|
|
747
|
+
// is the truthful one.
|
|
748
|
+
...(runtimeRun ? {
|
|
749
|
+
clarificationOptionCount: runtimeRun.clarificationOptions?.length ?? 0,
|
|
750
|
+
conversational: runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational',
|
|
751
|
+
conversationalAnswer: (runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational')
|
|
752
|
+
&& Boolean(runtimeRun.answer?.trim()),
|
|
753
|
+
meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
|
|
754
|
+
} : {}),
|
|
708
755
|
intent: result.contextPack?.routeDecision.intent,
|
|
709
756
|
reviewStatus: result.reviewStatus,
|
|
710
757
|
contextObjects: result.contextPack?.objects.length ?? 0,
|
|
@@ -734,6 +781,9 @@ async function runEval(rest, flags) {
|
|
|
734
781
|
minExecutionMatch: flags.minExecutionMatch ?? null,
|
|
735
782
|
minJudgePass: flags.minJudgePass ?? null,
|
|
736
783
|
maxWrongCertified: flags.maxWrongCertified ?? null,
|
|
784
|
+
maxFalseRefusal: flags.maxFalseRefusal ?? null,
|
|
785
|
+
minRefusalRecall: flags.minRefusalRecall ?? null,
|
|
786
|
+
minGroundedNarration: flags.minGroundedNarration ?? null,
|
|
737
787
|
};
|
|
738
788
|
const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
|
|
739
789
|
const ok = passed === results.length && thresholdsPassed;
|
|
@@ -752,6 +802,14 @@ async function runEval(rest, flags) {
|
|
|
752
802
|
console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
|
|
753
803
|
console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
|
|
754
804
|
console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
|
|
805
|
+
console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
|
|
806
|
+
console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
|
|
807
|
+
if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
|
|
808
|
+
console.log(` ! Semantic judgment ran for only ${formatRate(metrics.meaning_resolved_rate)} of cases. `
|
|
809
|
+
+ 'Without a reachable provider DQL will not settle a reading by lexical rank (AGT-017), so ambiguous '
|
|
810
|
+
+ 'questions clarify by design — treat the clarification rate above as an artifact, not a product signal.');
|
|
811
|
+
}
|
|
812
|
+
console.log(`Refusal recall: ${formatRate(metrics.refusal_recall)} (${metrics.refusal_required_case_count} case(s) that must refuse)`);
|
|
755
813
|
console.log(`Execution match rate: ${formatRate(metrics.execution_match_rate)}`);
|
|
756
814
|
console.log(`Tool requirement pass rate: ${formatRate(metrics.tool_requirement_pass_rate)}`);
|
|
757
815
|
console.log(`Tool-observed case count: ${metrics.tool_observed_case_count}`);
|
|
@@ -767,6 +825,15 @@ async function runEval(rest, flags) {
|
|
|
767
825
|
if (thresholds.minJudgePass !== null) {
|
|
768
826
|
console.log(`Judge-pass threshold: ${thresholds.minJudgePass} (actual ${formatRate(metrics.judge_pass_rate)})`);
|
|
769
827
|
}
|
|
828
|
+
if (thresholds.maxFalseRefusal !== null) {
|
|
829
|
+
console.log(`False-refusal ceiling: ${thresholds.maxFalseRefusal} (actual ${formatRate(metrics.false_refusal_rate)})`);
|
|
830
|
+
}
|
|
831
|
+
if (thresholds.minRefusalRecall !== null) {
|
|
832
|
+
console.log(`Refusal-recall threshold: ${thresholds.minRefusalRecall} (actual ${formatRate(metrics.refusal_recall)})`);
|
|
833
|
+
}
|
|
834
|
+
if (thresholds.minGroundedNarration !== null && thresholds.minGroundedNarration !== undefined) {
|
|
835
|
+
console.log(`Grounded-narration threshold: ${thresholds.minGroundedNarration} (actual ${formatRate(metrics.grounded_narration_rate)} over ${metrics.grounded_narration_attempted} attempted)`);
|
|
836
|
+
}
|
|
770
837
|
if (thresholds.maxWrongCertified !== null) {
|
|
771
838
|
console.log(`Wrong-certified ceiling: ${thresholds.maxWrongCertified} (actual ${metrics.wrong_certified_count})`);
|
|
772
839
|
}
|
|
@@ -791,6 +858,32 @@ function evaluateCase(testCase, result) {
|
|
|
791
858
|
const failures = [];
|
|
792
859
|
let validationCode;
|
|
793
860
|
let executionMatched;
|
|
861
|
+
// Answerability is asserted per case as well as aggregated into
|
|
862
|
+
// false_refusal_rate, so a single dead-end fails its own case instead of only
|
|
863
|
+
// nudging a rate someone has to notice.
|
|
864
|
+
const answerable = evalCaseIsAnswerable(expected);
|
|
865
|
+
// Three distinct outcomes, not two. An option-bearing clarification neither
|
|
866
|
+
// answers nor dead-ends: it must not fail an answerable case (its cost is
|
|
867
|
+
// tracked by `clarification_rate`), and it must not fail a must-refuse case
|
|
868
|
+
// either, because it did not assert anything about the data.
|
|
869
|
+
const clarifiedWithOptions = (result.clarificationOptions?.length ?? 0) > 0;
|
|
870
|
+
// A conversational reply ("I'm here to help you explore your data…") asserts
|
|
871
|
+
// nothing about the warehouse. For an out-of-scope question that is the CORRECT
|
|
872
|
+
// outcome — declining politely — so it must not be scored as an answer.
|
|
873
|
+
const conversational = result.answerKind === 'conversational';
|
|
874
|
+
const answerText = typeof result.text === 'string' ? result.text.trim() : '';
|
|
875
|
+
// Conversational replies split two ways, and the case's own expectation says
|
|
876
|
+
// which is right: for an answerable question a substantive reply IS the answer
|
|
877
|
+
// (a definition), and for an out-of-scope one it is the correct decline.
|
|
878
|
+
const conversationalAnswer = conversational && answerText.length > 0;
|
|
879
|
+
const producedDataAnswer = result.kind !== 'no_answer' && !conversational;
|
|
880
|
+
const deadEnded = !producedDataAnswer && !clarifiedWithOptions && !conversationalAnswer;
|
|
881
|
+
if (answerable === true && deadEnded) {
|
|
882
|
+
failures.push(`FALSE REFUSAL: this question is answerable, but the run dead-ended with no answer and no options${result.refusalCode ? ` (${result.refusalCode})` : ''}`);
|
|
883
|
+
}
|
|
884
|
+
if (answerable === false && producedDataAnswer) {
|
|
885
|
+
failures.push(`expected a refusal (question is out of scope / unanswerable), but the run answered with kind ${result.kind}`);
|
|
886
|
+
}
|
|
794
887
|
if (expected.kind && result.kind !== expected.kind)
|
|
795
888
|
failures.push(`kind expected ${expected.kind}, got ${result.kind}`);
|
|
796
889
|
if (expected.sourceTier && result.sourceTier !== expected.sourceTier)
|
|
@@ -849,7 +942,75 @@ function evaluateCase(testCase, result) {
|
|
|
849
942
|
}
|
|
850
943
|
return { failures, validationCode, executionMatched };
|
|
851
944
|
}
|
|
945
|
+
/**
|
|
946
|
+
* Is the case answerable? Explicit `expected.answerable` wins; otherwise infer
|
|
947
|
+
* from the expectations already present, so the metric covers legacy case files.
|
|
948
|
+
* A case with no expectations at all is excluded — it asserts nothing, so it can
|
|
949
|
+
* neither prove nor disprove a false refusal.
|
|
950
|
+
*/
|
|
951
|
+
export function evalCaseIsAnswerable(expected) {
|
|
952
|
+
if (!expected)
|
|
953
|
+
return undefined;
|
|
954
|
+
if (expected.answerable !== undefined)
|
|
955
|
+
return expected.answerable;
|
|
956
|
+
if (expected.kind === 'no_answer')
|
|
957
|
+
return false;
|
|
958
|
+
if (expected.sourceTier === 'no_answer')
|
|
959
|
+
return false;
|
|
960
|
+
if (expected.route === 'clarify')
|
|
961
|
+
return false;
|
|
962
|
+
if (Object.keys(expected).length === 0)
|
|
963
|
+
return undefined;
|
|
964
|
+
return true;
|
|
965
|
+
}
|
|
966
|
+
/**
|
|
967
|
+
* Did the run leave the user with NO way forward?
|
|
968
|
+
*
|
|
969
|
+
* Deliberately narrower than "did not answer". A clarification that offers
|
|
970
|
+
* selectable options is answerable on the next turn — worth minimising, tracked
|
|
971
|
+
* separately as `clarification_rate`, but not the defect. A clarification with
|
|
972
|
+
* ZERO options is a true dead end: the reported production loop was exactly
|
|
973
|
+
* this, and a free-text reply to it reproduced the same question forever.
|
|
974
|
+
*/
|
|
975
|
+
export function evalResultRefused(result) {
|
|
976
|
+
// A substantive conversational reply is an ANSWER, not a dead end. A governed
|
|
977
|
+
// definition ("**top_customers** — Top 10 customers by lifetime spend…") is
|
|
978
|
+
// exactly what a "what does X mean?" turn should return, and scoring it as a
|
|
979
|
+
// refusal would report the feature working as the feature failing.
|
|
980
|
+
if (result.conversationalAnswer)
|
|
981
|
+
return false;
|
|
982
|
+
// Order matters: the drivers collapse every clarification to `no_answer`
|
|
983
|
+
// (it is not an answer), so the option check has to run FIRST or an
|
|
984
|
+
// option-bearing clarification is miscounted as a dead end.
|
|
985
|
+
if (result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0)
|
|
986
|
+
return false;
|
|
987
|
+
return result.kind === 'no_answer' || result.route === 'clarify';
|
|
988
|
+
}
|
|
989
|
+
/** Did the run ask an answerable clarification rather than answering outright? */
|
|
990
|
+
export function evalResultClarified(result) {
|
|
991
|
+
return result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0;
|
|
992
|
+
}
|
|
993
|
+
/**
|
|
994
|
+
* Translate only the durable narration receipt into evaluation fields.
|
|
995
|
+
*
|
|
996
|
+
* Reader prose, row count, and result shape are deliberately absent: a skipped
|
|
997
|
+
* narration can have rows, and a deterministic fallback can use any wording.
|
|
998
|
+
*/
|
|
999
|
+
export function narrationOutcomeForEval(receipt) {
|
|
1000
|
+
if (receipt?.mode !== 'verified_facts' || !receipt.attempted)
|
|
1001
|
+
return {};
|
|
1002
|
+
return {
|
|
1003
|
+
// A durable receipt is the only source of this metric. An infrastructure
|
|
1004
|
+
// error started a verified narration but did not produce a grounded answer,
|
|
1005
|
+
// so it belongs in the denominator just like the deterministic floor. The
|
|
1006
|
+
// old mapping treated it as a success because only fallback was negative.
|
|
1007
|
+
narrationAttempted: true,
|
|
1008
|
+
narrationFallback: receipt.outcome !== 'success',
|
|
1009
|
+
};
|
|
1010
|
+
}
|
|
852
1011
|
function computeEvalMetrics(results) {
|
|
1012
|
+
const answerableCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === true);
|
|
1013
|
+
const refusalRequiredCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === false);
|
|
853
1014
|
const certifiedCases = results.filter((result) => result.expected?.kind === 'certified' ||
|
|
854
1015
|
result.expected?.certification === 'certified' ||
|
|
855
1016
|
result.expected?.route === 'certified');
|
|
@@ -874,6 +1035,52 @@ function computeEvalMetrics(results) {
|
|
|
874
1035
|
wrong_certified_count: results.filter((result) => result.kind === 'certified' &&
|
|
875
1036
|
(result.expected?.kind ? result.expected.kind !== 'certified' : result.followUp)).length,
|
|
876
1037
|
outside_context_rejection_count: results.filter((result) => result.validationCode === 'unknown_relation' || result.validationCode === 'unknown_column').length,
|
|
1038
|
+
/**
|
|
1039
|
+
* THE headline number: how often an answerable question was refused.
|
|
1040
|
+
* Bounds every other quality metric — a run that refuses cannot be wrong,
|
|
1041
|
+
* so a falling false-refusal rate must be read together with
|
|
1042
|
+
* `execution_match_rate` to be sure refusals were replaced by CORRECT answers.
|
|
1043
|
+
*/
|
|
1044
|
+
false_refusal_rate: ratio(answerableCases.filter(evalResultRefused).length, answerableCases.length),
|
|
1045
|
+
false_refusal_count: answerableCases.filter(evalResultRefused).length,
|
|
1046
|
+
answerable_case_count: answerableCases.length,
|
|
1047
|
+
/**
|
|
1048
|
+
* Answerable cases that asked an option-bearing clarification instead of
|
|
1049
|
+
* answering. Not a defect, but a direct cost in turns — read it next to
|
|
1050
|
+
* false_refusal_rate so a fall in refusals is not just a rise in questions.
|
|
1051
|
+
*/
|
|
1052
|
+
clarification_rate: ratio(answerableCases.filter(evalResultClarified).length, answerableCases.length),
|
|
1053
|
+
/**
|
|
1054
|
+
* Cases where semantic judgment ran. Without a provider `mayAssumeInterpretation`
|
|
1055
|
+
* is false (AGT-017), so every ambiguous question clarifies by design and
|
|
1056
|
+
* `clarification_rate` says nothing about product quality.
|
|
1057
|
+
*/
|
|
1058
|
+
meaning_resolved_rate: ratio(results.filter((result) => result.meaningResolved === true).length, results.length),
|
|
1059
|
+
/**
|
|
1060
|
+
* Latency, which the acceptance matrix asked for and nothing measured. A
|
|
1061
|
+
* quality gain paid for entirely in wall clock is not a gain: the plan's
|
|
1062
|
+
* two-tier target is certified/semantic under 5s while research takes
|
|
1063
|
+
* minutes, and only a per-class p95 can tell those apart from a regression.
|
|
1064
|
+
*/
|
|
1065
|
+
latency_p50_ms: percentileMs(results, 0.5),
|
|
1066
|
+
latency_p95_ms: percentileMs(results, 0.95),
|
|
1067
|
+
latency_p95_answerable_ms: percentileMs(answerableCases, 0.95),
|
|
1068
|
+
/**
|
|
1069
|
+
* How often verified narration survived. When the drafted narration fails
|
|
1070
|
+
* its fact check the reader gets the deterministic record under a
|
|
1071
|
+
* disclaimer — correct, but visibly worse. A silent fall here is exactly
|
|
1072
|
+
* the truncation defect that shipped unnoticed, so it is measured.
|
|
1073
|
+
*/
|
|
1074
|
+
grounded_narration_rate: ratio(results.filter((result) => result.narrationAttempted && result.narrationFallback === false).length, results.filter((result) => result.narrationAttempted).length),
|
|
1075
|
+
grounded_narration_attempted: results.filter((result) => result.narrationAttempted).length,
|
|
1076
|
+
/**
|
|
1077
|
+
* The guard on the above: cases that must NOT produce a data answer.
|
|
1078
|
+
* Scored on "did not answer" rather than "dead-ended", because declining via
|
|
1079
|
+
* a clarification is still declining — what would be wrong is asserting
|
|
1080
|
+
* something about data the project does not have.
|
|
1081
|
+
*/
|
|
1082
|
+
refusal_recall: ratio(refusalRequiredCases.filter((result) => result.kind === 'no_answer' || result.conversational === true).length, refusalRequiredCases.length),
|
|
1083
|
+
refusal_required_case_count: refusalRequiredCases.length,
|
|
877
1084
|
draft_saved_count: results.filter((result) => result.draftSaved).length,
|
|
878
1085
|
tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
|
|
879
1086
|
avg_tool_calls: average(toolCallCounts),
|
|
@@ -885,9 +1092,15 @@ function agentEvalThresholdsPass(metrics, thresholds) {
|
|
|
885
1092
|
// A rate threshold with no applicable cases (metric === null) is vacuously
|
|
886
1093
|
// satisfied — you only fail when the metric exists and falls below the bar.
|
|
887
1094
|
const rateOk = (metric, min) => min === null || min === undefined || metric === null || metric >= min;
|
|
888
|
-
|
|
1095
|
+
// A ceiling is only meaningful when the metric has data; `null` means no
|
|
1096
|
+
// answerable case was scored, which is "unknown", not "perfect".
|
|
1097
|
+
const ceilingOk = (metric, max) => max === null || max === undefined || metric === null || metric <= max;
|
|
1098
|
+
return rateOk(metrics.grounded_narration_rate, thresholds.minGroundedNarration)
|
|
1099
|
+
&& rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
|
|
889
1100
|
&& rateOk(metrics.execution_match_rate, thresholds.minExecutionMatch)
|
|
890
1101
|
&& rateOk(metrics.judge_pass_rate, thresholds.minJudgePass)
|
|
1102
|
+
&& ceilingOk(metrics.false_refusal_rate, thresholds.maxFalseRefusal)
|
|
1103
|
+
&& rateOk(metrics.refusal_recall, thresholds.minRefusalRecall)
|
|
891
1104
|
&& (thresholds.maxWrongCertified === null
|
|
892
1105
|
|| thresholds.maxWrongCertified === undefined
|
|
893
1106
|
|| metrics.wrong_certified_count <= thresholds.maxWrongCertified);
|
|
@@ -1207,6 +1420,20 @@ export const __test__ = {
|
|
|
1207
1420
|
cliAnalysisDepth,
|
|
1208
1421
|
cliReasoningEffort,
|
|
1209
1422
|
computeEvalMetrics,
|
|
1423
|
+
narrationOutcomeForEval,
|
|
1210
1424
|
evaluateCase,
|
|
1211
1425
|
};
|
|
1426
|
+
/** Percentile over observed case durations. Returns null when nothing timed. */
|
|
1427
|
+
function percentileMs(results, q) {
|
|
1428
|
+
const observed = results
|
|
1429
|
+
.map((result) => result.durationMs)
|
|
1430
|
+
.filter((value) => typeof value === 'number' && Number.isFinite(value) && value >= 0)
|
|
1431
|
+
.sort((left, right) => left - right);
|
|
1432
|
+
if (observed.length === 0)
|
|
1433
|
+
return null;
|
|
1434
|
+
// Nearest-rank: with a handful of cases an interpolated percentile invents a
|
|
1435
|
+
// duration nothing actually took.
|
|
1436
|
+
const rank = Math.min(observed.length - 1, Math.max(0, Math.ceil(q * observed.length) - 1));
|
|
1437
|
+
return observed[rank] ?? null;
|
|
1438
|
+
}
|
|
1212
1439
|
//# sourceMappingURL=agent.js.map
|