@duckcodeailabs/dql-cli 1.13.5 → 1.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/args.d.ts +17 -0
  2. package/dist/args.d.ts.map +1 -1
  3. package/dist/args.js +30 -0
  4. package/dist/args.js.map +1 -1
  5. package/dist/assets/dql-notebook/assets/{AgentLogPage-BzQOKjyV.js → AgentLogPage-BPz-UWFh.js} +1 -1
  6. package/dist/assets/dql-notebook/assets/{AiBuildDialog-CTxha499.js → AiBuildDialog-BEl53WA_.js} +1 -1
  7. package/dist/assets/dql-notebook/assets/{AiBuildResult--MuD_I6g.js → AiBuildResult-B4yfGTTZ.js} +1 -1
  8. package/dist/assets/dql-notebook/assets/{AiSidePanel-DcI4PMJ1.js → AiSidePanel-CSZAAvuD.js} +1 -1
  9. package/dist/assets/dql-notebook/assets/{AnalyticsHome-Dq54vjNz.js → AnalyticsHome-D5P6Ujwi.js} +1 -1
  10. package/dist/assets/dql-notebook/assets/{AppsView-0SlwWYex.js → AppsView-DQOwU9Cg.js} +4 -4
  11. package/dist/assets/dql-notebook/assets/{BlockStudio-CxxXZD3M.js → BlockStudio-B4ap0GdY.js} +1 -1
  12. package/dist/assets/dql-notebook/assets/{BusinessArtifactView-DxdAmeUN.js → BusinessArtifactView-BIfNI-0S.js} +1 -1
  13. package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-y81gFg_b.js → DbtFirstModelingPage-D72byb2g.js} +1 -1
  14. package/dist/assets/dql-notebook/assets/{GitPage-GQtcwncb.js → GitPage-lQb1uXH2.js} +1 -1
  15. package/dist/assets/dql-notebook/assets/{GlobalAiRail-DM4wkxR_.js → GlobalAiRail-CVECf6Xj.js} +1 -1
  16. package/dist/assets/dql-notebook/assets/{GovernedContextPage-B8Ft6JES.js → GovernedContextPage-trOyMCY6.js} +1 -1
  17. package/dist/assets/dql-notebook/assets/{HelpDocsPage-DnD4nMuF.js → HelpDocsPage-D8hLS5lE.js} +1 -1
  18. package/dist/assets/dql-notebook/assets/{HomePage-Ehb0ITj-.js → HomePage-eIkfBIep.js} +1 -1
  19. package/dist/assets/dql-notebook/assets/{LineageDAG-Lvjc2AQX.js → LineageDAG-CSqcbDrE.js} +1 -1
  20. package/dist/assets/dql-notebook/assets/{LineageDetailView-DX32pfbp.js → LineageDetailView-DJZZjZu-.js} +1 -1
  21. package/dist/assets/dql-notebook/assets/{LineageDrawer-CFq3jPfk.js → LineageDrawer-BcIipIc3.js} +1 -1
  22. package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-CVHXuHls.js → LineagePathBreadcrumb-CviIf8PN.js} +1 -1
  23. package/dist/assets/dql-notebook/assets/{MiniLineageGraph-BKaeT5Kt.js → MiniLineageGraph-vH_MY_Ju.js} +1 -1
  24. package/dist/assets/dql-notebook/assets/{NewBlockModal-DI-JDzFH.js → NewBlockModal-DbCQg-pj.js} +1 -1
  25. package/dist/assets/dql-notebook/assets/{NewNotebookModal-Csw6eF74.js → NewNotebookModal-5Vp6XiuK.js} +1 -1
  26. package/dist/assets/dql-notebook/assets/{NotebookEditor-FEqs3789.js → NotebookEditor-DjqJS44s.js} +1 -1
  27. package/dist/assets/dql-notebook/assets/{ReadinessPage-Bk0S_MEA.js → ReadinessPage-BHGrC5ho.js} +1 -1
  28. package/dist/assets/dql-notebook/assets/{SetupOnboarding-DyLaPFGn.js → SetupOnboarding-BfS9Tdx-.js} +1 -1
  29. package/dist/assets/dql-notebook/assets/{SkillsPage-DLHyvhci.js → SkillsPage-BFH21wSj.js} +1 -1
  30. package/dist/assets/dql-notebook/assets/{TrustBadge-CIDLj2A6.js → TrustBadge-zm6g_SxZ.js} +1 -1
  31. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel--oxmjlgr.js +88 -0
  32. package/dist/assets/dql-notebook/assets/{answer-to-notebook-CFEJHLvs.js → answer-to-notebook-DPhxIEzF.js} +1 -1
  33. package/dist/assets/dql-notebook/assets/{arrow-left-B-Zdcyvm.js → arrow-left-DNEb86Xc.js} +1 -1
  34. package/dist/assets/dql-notebook/assets/{arrow-right-Cn8TM7cp.js → arrow-right-DpPWbwaD.js} +1 -1
  35. package/dist/assets/dql-notebook/assets/{book-open-text-Bko2NNUs.js → book-open-text-D7s5jo4X.js} +1 -1
  36. package/dist/assets/dql-notebook/assets/{circle-x-AWJAmUBB.js → circle-x-Db3dXKg7.js} +1 -1
  37. package/dist/assets/dql-notebook/assets/{dagre.esm-Cl_ucrRh.js → dagre.esm-C7pppQ1a.js} +1 -1
  38. package/dist/assets/dql-notebook/assets/{external-link-DN57tb5f.js → external-link-BxwXitO_.js} +1 -1
  39. package/dist/assets/dql-notebook/assets/{grip-vertical-BOWwFFva.js → grip-vertical-Dt4lkWRi.js} +1 -1
  40. package/dist/assets/dql-notebook/assets/{index-Ck-wqvV2.js → index-DKo-bwNw.js} +4 -4
  41. package/dist/assets/dql-notebook/assets/{link-2-xUhbfBs8.js → link-2-Dfo2P6wi.js} +1 -1
  42. package/dist/assets/dql-notebook/assets/{list-tree-BDWcBr67.js → list-tree-DiTmIWAL.js} +1 -1
  43. package/dist/assets/dql-notebook/assets/{minimize-2-CEeNoMTS.js → minimize-2-CMTAkPzL.js} +1 -1
  44. package/dist/assets/dql-notebook/assets/{panel-right-open-DnSeUvOY.js → panel-right-open-DwYr7FW4.js} +1 -1
  45. package/dist/assets/dql-notebook/assets/{play-EoDj8-fa.js → play-BXhHYQ4x.js} +1 -1
  46. package/dist/assets/dql-notebook/assets/{rotate-ccw-BAMxWT9A.js → rotate-ccw-BNi6F8pl.js} +1 -1
  47. package/dist/assets/dql-notebook/assets/{semantic-fields-BwLOs1kl.js → semantic-fields-CNOGysAy.js} +1 -1
  48. package/dist/assets/dql-notebook/assets/{sliders-horizontal-BYkm8SWW.js → sliders-horizontal-Ec5MUMUW.js} +1 -1
  49. package/dist/assets/dql-notebook/assets/{star-sVexqNCs.js → star-B9leDkp_.js} +1 -1
  50. package/dist/assets/dql-notebook/assets/{triangle-alert-D2njv4o4.js → triangle-alert-BefTYCzx.js} +1 -1
  51. package/dist/assets/dql-notebook/assets/{upload-CXTMxIA8.js → upload-SPiOM2tQ.js} +1 -1
  52. package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-CPDeaAgL.js → usePersistedAgentThreadId-C4foXeiQ.js} +4 -4
  53. package/dist/assets/dql-notebook/assets/{user-round-zVsI_uud.js → user-round-comGmyw-.js} +1 -1
  54. package/dist/assets/dql-notebook/assets/{wand-sparkles-BMIwIzXT.js → wand-sparkles-BffR4dF8.js} +1 -1
  55. package/dist/assets/dql-notebook/assets/{workflow-Dj9y1sNk.js → workflow-ChmPTEzH.js} +1 -1
  56. package/dist/assets/dql-notebook/assets/{wrench-DPQi6zrs.js → wrench-DovaG_ze.js} +1 -1
  57. package/dist/assets/dql-notebook/assets/{x-XhhbtinL.js → x-nRx91AgW.js} +1 -1
  58. package/dist/assets/dql-notebook/index.html +1 -1
  59. package/dist/commands/agent-eval-cassette.d.ts +73 -0
  60. package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
  61. package/dist/commands/agent-eval-cassette.js +170 -0
  62. package/dist/commands/agent-eval-cassette.js.map +1 -0
  63. package/dist/commands/agent-eval-runtime.d.ts +97 -0
  64. package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
  65. package/dist/commands/agent-eval-runtime.js +155 -0
  66. package/dist/commands/agent-eval-runtime.js.map +1 -0
  67. package/dist/commands/agent.d.ts +115 -1
  68. package/dist/commands/agent.d.ts.map +1 -1
  69. package/dist/commands/agent.js +289 -62
  70. package/dist/commands/agent.js.map +1 -1
  71. package/dist/commands/eval.d.ts +15 -1
  72. package/dist/commands/eval.d.ts.map +1 -1
  73. package/dist/commands/eval.js +32 -3
  74. package/dist/commands/eval.js.map +1 -1
  75. package/dist/index.js +5 -1
  76. package/dist/index.js.map +1 -1
  77. package/dist/llm/analyst-loop-tools.d.ts +19 -0
  78. package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
  79. package/dist/llm/analyst-loop-tools.js +56 -0
  80. package/dist/llm/analyst-loop-tools.js.map +1 -0
  81. package/dist/llm/providers/dql-agent-provider.d.ts +24 -1
  82. package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
  83. package/dist/llm/providers/dql-agent-provider.js +343 -10
  84. package/dist/llm/providers/dql-agent-provider.js.map +1 -1
  85. package/dist/llm/types.d.ts +19 -1
  86. package/dist/llm/types.d.ts.map +1 -1
  87. package/dist/local-runtime.d.ts +132 -2
  88. package/dist/local-runtime.d.ts.map +1 -1
  89. package/dist/local-runtime.js +1120 -184
  90. package/dist/local-runtime.js.map +1 -1
  91. package/dist/package.json +10 -10
  92. package/package.json +10 -10
  93. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-C0oKTU6G.js +0 -88
@@ -21,6 +21,8 @@
21
21
  * dql agent feedback <up|down> --block <id> --question "..."
22
22
  * Records feedback into the KG. Used by clients without MCP access.
23
23
  */
24
+ import { answerFromRuntimeRun, driveViaRuntime, evalRouteForRun } from './agent-eval-runtime.js';
25
+ import { CassetteStore, cassetteDirFor, withCassette } from './agent-eval-cassette.js';
24
26
  import { existsSync, readFileSync } from 'node:fs';
25
27
  import { join, resolve } from 'node:path';
26
28
  import { load as loadYaml } from 'js-yaml';
@@ -551,7 +553,14 @@ async function runEval(rest, flags) {
551
553
  if (!existsSync(kgPath))
552
554
  await reindexProject(projectRoot, { kgPath });
553
555
  const providerName = flags.provider;
554
- const provider = await pickProvider(providerName);
556
+ const rawProvider = await pickProvider(providerName);
557
+ // Cassettes make a provider-backed suite repeatable and free to re-run. They
558
+ // apply to the in-process driver here; `--via runtime` needs the SERVER
559
+ // started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
560
+ const cassetteMode = flags.cassette;
561
+ const provider = cassetteMode === 'record' || cassetteMode === 'replay'
562
+ ? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode)
563
+ : rawProvider;
555
564
  const reasoningEffort = cliReasoningEffort(flags);
556
565
  const requestedDepth = cliAnalysisDepth(flags);
557
566
  const kg = new KGStore(kgPath);
@@ -562,10 +571,22 @@ async function runEval(rest, flags) {
562
571
  // gracefully when no provider is available so offline eval stays deterministic.
563
572
  const judge = Boolean(flags.judge);
564
573
  const judgeComplete = async ({ system, user }) => provider.generate([{ role: 'system', content: system }, { role: 'user', content: user }], {});
574
+ // Which half of the stack is under test. `loop` preserves today's behaviour;
575
+ // `runtime` is the one that exercises routing and gates end to end.
576
+ const via = flags.via === 'runtime' ? 'runtime' : 'loop';
565
577
  const runtimeBase = flags.runtimeUrl
566
578
  ?? flags.runtime
567
579
  ?? process.env.DQL_RUNTIME_URL
568
580
  ?? 'http://127.0.0.1:3474';
581
+ if (via === 'runtime') {
582
+ // Fail fast and loudly. Without this, an unreachable server turns every case
583
+ // into a transport error and the report reads as a false-refusal spike that
584
+ // no code change caused.
585
+ const probe = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/health`).catch(() => null);
586
+ if (!probe?.ok) {
587
+ throw new Error(`--via runtime needs a running server at ${runtimeBase}. Start one with \`dql serve\`, or use --via loop to score the answer loop in-process.`);
588
+ }
589
+ }
569
590
  const semanticLayer = loadAgentSemanticLayer(projectRoot);
570
591
  const expandGroundingContext = createGroundingContextExpander(projectRoot);
571
592
  const answerLoopTools = buildAnswerLoopTools(projectRoot);
@@ -605,35 +626,66 @@ async function runEval(rest, flags) {
605
626
  }
606
627
  : undefined,
607
628
  }).catch(() => undefined);
608
- const result = await answer({
609
- question: testCase.question,
610
- domain: testCase.domain,
611
- domainContext: testCase.domain && manifest
612
- ? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
613
- : undefined,
614
- provider,
615
- kg,
616
- manifest: manifest ?? undefined,
617
- skills,
618
- memoryContext,
619
- followUp: testCase.followUp,
620
- semanticLayer,
621
- schemaContext,
622
- contextPack,
623
- reasoningEffort,
624
- analysisDepth: contextBudget.analysisDepth,
625
- expandGroundingContext,
626
- answerLoopTools,
627
- executeCertifiedBlock: execute && manifest
628
- ? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
629
- : undefined,
630
- executeGeneratedSql: execute
631
- ? createGeneratedSqlExecutor(runtimeBase)
632
- : undefined,
633
- captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
634
- const slug = deriveGeneratedDraftSlug(draftQuestion);
635
- const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
636
- if (dqlArtifact?.kind === 'semantic_block') {
629
+ // `--via runtime` posts to a running `dql serve` so the case exercises the
630
+ // router, engine, plan boundary, and gates. The in-process driver below
631
+ // calls the answer loop directly and cannot observe any of them, which is
632
+ // why a refusal metric taken from it reads cleaner than users experience.
633
+ const runtimeRun = via === 'runtime'
634
+ ? await driveViaRuntime({ runtimeBase, question: testCase.question })
635
+ : undefined;
636
+ const result = runtimeRun
637
+ ? answerFromRuntimeRun(runtimeRun)
638
+ : await answer({
639
+ question: testCase.question,
640
+ domain: testCase.domain,
641
+ domainContext: testCase.domain && manifest
642
+ ? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
643
+ : undefined,
644
+ provider,
645
+ kg,
646
+ manifest: manifest ?? undefined,
647
+ skills,
648
+ memoryContext,
649
+ followUp: testCase.followUp,
650
+ semanticLayer,
651
+ schemaContext,
652
+ contextPack,
653
+ reasoningEffort,
654
+ analysisDepth: contextBudget.analysisDepth,
655
+ expandGroundingContext,
656
+ answerLoopTools,
657
+ executeCertifiedBlock: execute && manifest
658
+ ? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
659
+ : undefined,
660
+ executeGeneratedSql: execute
661
+ ? createGeneratedSqlExecutor(runtimeBase)
662
+ : undefined,
663
+ captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
664
+ const slug = deriveGeneratedDraftSlug(draftQuestion);
665
+ const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
666
+ if (dqlArtifact?.kind === 'semantic_block') {
667
+ if (!flags.save) {
668
+ return {
669
+ path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
670
+ askedTimes: 0,
671
+ proposedContractId: `${proposedDomain}.Unknown.${slug}`,
672
+ };
673
+ }
674
+ return upsertGeneratedDqlArtifactDraft(projectRoot, {
675
+ slug,
676
+ question: draftQuestion,
677
+ proposedContractId: `${proposedDomain}.Unknown.${slug}`,
678
+ proposedDomain,
679
+ dqlArtifact,
680
+ sourceQuestion: followUp?.sourceQuestion,
681
+ sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
682
+ followupKind: followUp?.kind,
683
+ outputs,
684
+ contextPackId: draftContextPack?.id,
685
+ routeIntent: String(intent),
686
+ validationWarnings,
687
+ });
688
+ }
637
689
  if (!flags.save) {
638
690
  return {
639
691
  path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
@@ -641,51 +693,30 @@ async function runEval(rest, flags) {
641
693
  proposedContractId: `${proposedDomain}.Unknown.${slug}`,
642
694
  };
643
695
  }
644
- return upsertGeneratedDqlArtifactDraft(projectRoot, {
696
+ return upsertGeneratedDraft(projectRoot, {
645
697
  slug,
646
698
  question: draftQuestion,
699
+ proposedSql: sql,
647
700
  proposedContractId: `${proposedDomain}.Unknown.${slug}`,
648
701
  proposedDomain,
649
- dqlArtifact,
702
+ proposedEntity,
703
+ sourceDqlArtifact,
650
704
  sourceQuestion: followUp?.sourceQuestion,
651
705
  sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
652
706
  followupKind: followUp?.kind,
707
+ requestedFilters,
708
+ requestedDimensions,
653
709
  outputs,
654
710
  contextPackId: draftContextPack?.id,
655
711
  routeIntent: String(intent),
656
712
  validationWarnings,
657
713
  });
658
- }
659
- if (!flags.save) {
660
- return {
661
- path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
662
- askedTimes: 0,
663
- proposedContractId: `${proposedDomain}.Unknown.${slug}`,
664
- };
665
- }
666
- return upsertGeneratedDraft(projectRoot, {
667
- slug,
668
- question: draftQuestion,
669
- proposedSql: sql,
670
- proposedContractId: `${proposedDomain}.Unknown.${slug}`,
671
- proposedDomain,
672
- proposedEntity,
673
- sourceDqlArtifact,
674
- sourceQuestion: followUp?.sourceQuestion,
675
- sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
676
- followupKind: followUp?.kind,
677
- requestedFilters,
678
- requestedDimensions,
679
- outputs,
680
- contextPackId: draftContextPack?.id,
681
- routeIntent: String(intent),
682
- validationWarnings,
683
- });
684
- },
685
- });
714
+ },
715
+ });
686
716
  const evaluation = evaluateCase(testCase, result);
687
717
  const durationMs = Date.now() - startedAt;
688
718
  const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
719
+ const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
689
720
  const judgeVerdict = judge
690
721
  ? await judgeAnswer({
691
722
  question: testCase.question,
@@ -700,11 +731,27 @@ async function runEval(rest, flags) {
700
731
  passed: evaluation.failures.length === 0,
701
732
  failures: evaluation.failures,
702
733
  durationMs,
734
+ // The persisted receipt is the only evidence for this metric. Rows and
735
+ // reader prose are intentionally ignored: a row-bearing answer can be
736
+ // skipped, while a deterministic fallback can render different wording.
737
+ narrationAttempted: narration.narrationAttempted,
738
+ narrationFallback: narration.narrationFallback,
703
739
  executionMs: result.result?.executionTime,
704
740
  executionMatched: evaluation.executionMatched,
705
741
  ...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
706
742
  kind: result.kind,
707
- route: result.contextPack?.routeDecision.route,
743
+ route: runtimeRun ? evalRouteForRun(runtimeRun.route) : result.contextPack?.routeDecision.route,
744
+ // Only the runtime driver can see the router's clarification options.
745
+ // In-process runs leave this undefined, so a clarify there scores as a
746
+ // dead end — the conservative reading, and another reason `--via runtime`
747
+ // is the truthful one.
748
+ ...(runtimeRun ? {
749
+ clarificationOptionCount: runtimeRun.clarificationOptions?.length ?? 0,
750
+ conversational: runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational',
751
+ conversationalAnswer: (runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational')
752
+ && Boolean(runtimeRun.answer?.trim()),
753
+ meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
754
+ } : {}),
708
755
  intent: result.contextPack?.routeDecision.intent,
709
756
  reviewStatus: result.reviewStatus,
710
757
  contextObjects: result.contextPack?.objects.length ?? 0,
@@ -734,6 +781,9 @@ async function runEval(rest, flags) {
734
781
  minExecutionMatch: flags.minExecutionMatch ?? null,
735
782
  minJudgePass: flags.minJudgePass ?? null,
736
783
  maxWrongCertified: flags.maxWrongCertified ?? null,
784
+ maxFalseRefusal: flags.maxFalseRefusal ?? null,
785
+ minRefusalRecall: flags.minRefusalRecall ?? null,
786
+ minGroundedNarration: flags.minGroundedNarration ?? null,
737
787
  };
738
788
  const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
739
789
  const ok = passed === results.length && thresholdsPassed;
@@ -752,6 +802,14 @@ async function runEval(rest, flags) {
752
802
  console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
753
803
  console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
754
804
  console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
805
+ console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
806
+ console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
807
+ if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
808
+ console.log(` ! Semantic judgment ran for only ${formatRate(metrics.meaning_resolved_rate)} of cases. `
809
+ + 'Without a reachable provider DQL will not settle a reading by lexical rank (AGT-017), so ambiguous '
810
+ + 'questions clarify by design — treat the clarification rate above as an artifact, not a product signal.');
811
+ }
812
+ console.log(`Refusal recall: ${formatRate(metrics.refusal_recall)} (${metrics.refusal_required_case_count} case(s) that must refuse)`);
755
813
  console.log(`Execution match rate: ${formatRate(metrics.execution_match_rate)}`);
756
814
  console.log(`Tool requirement pass rate: ${formatRate(metrics.tool_requirement_pass_rate)}`);
757
815
  console.log(`Tool-observed case count: ${metrics.tool_observed_case_count}`);
@@ -767,6 +825,15 @@ async function runEval(rest, flags) {
767
825
  if (thresholds.minJudgePass !== null) {
768
826
  console.log(`Judge-pass threshold: ${thresholds.minJudgePass} (actual ${formatRate(metrics.judge_pass_rate)})`);
769
827
  }
828
+ if (thresholds.maxFalseRefusal !== null) {
829
+ console.log(`False-refusal ceiling: ${thresholds.maxFalseRefusal} (actual ${formatRate(metrics.false_refusal_rate)})`);
830
+ }
831
+ if (thresholds.minRefusalRecall !== null) {
832
+ console.log(`Refusal-recall threshold: ${thresholds.minRefusalRecall} (actual ${formatRate(metrics.refusal_recall)})`);
833
+ }
834
+ if (thresholds.minGroundedNarration !== null && thresholds.minGroundedNarration !== undefined) {
835
+ console.log(`Grounded-narration threshold: ${thresholds.minGroundedNarration} (actual ${formatRate(metrics.grounded_narration_rate)} over ${metrics.grounded_narration_attempted} attempted)`);
836
+ }
770
837
  if (thresholds.maxWrongCertified !== null) {
771
838
  console.log(`Wrong-certified ceiling: ${thresholds.maxWrongCertified} (actual ${metrics.wrong_certified_count})`);
772
839
  }
@@ -791,6 +858,32 @@ function evaluateCase(testCase, result) {
791
858
  const failures = [];
792
859
  let validationCode;
793
860
  let executionMatched;
861
+ // Answerability is asserted per case as well as aggregated into
862
+ // false_refusal_rate, so a single dead-end fails its own case instead of only
863
+ // nudging a rate someone has to notice.
864
+ const answerable = evalCaseIsAnswerable(expected);
865
+ // Three distinct outcomes, not two. An option-bearing clarification neither
866
+ // answers nor dead-ends: it must not fail an answerable case (its cost is
867
+ // tracked by `clarification_rate`), and it must not fail a must-refuse case
868
+ // either, because it did not assert anything about the data.
869
+ const clarifiedWithOptions = (result.clarificationOptions?.length ?? 0) > 0;
870
+ // A conversational reply ("I'm here to help you explore your data…") asserts
871
+ // nothing about the warehouse. For an out-of-scope question that is the CORRECT
872
+ // outcome — declining politely — so it must not be scored as an answer.
873
+ const conversational = result.answerKind === 'conversational';
874
+ const answerText = typeof result.text === 'string' ? result.text.trim() : '';
875
+ // Conversational replies split two ways, and the case's own expectation says
876
+ // which is right: for an answerable question a substantive reply IS the answer
877
+ // (a definition), and for an out-of-scope one it is the correct decline.
878
+ const conversationalAnswer = conversational && answerText.length > 0;
879
+ const producedDataAnswer = result.kind !== 'no_answer' && !conversational;
880
+ const deadEnded = !producedDataAnswer && !clarifiedWithOptions && !conversationalAnswer;
881
+ if (answerable === true && deadEnded) {
882
+ failures.push(`FALSE REFUSAL: this question is answerable, but the run dead-ended with no answer and no options${result.refusalCode ? ` (${result.refusalCode})` : ''}`);
883
+ }
884
+ if (answerable === false && producedDataAnswer) {
885
+ failures.push(`expected a refusal (question is out of scope / unanswerable), but the run answered with kind ${result.kind}`);
886
+ }
794
887
  if (expected.kind && result.kind !== expected.kind)
795
888
  failures.push(`kind expected ${expected.kind}, got ${result.kind}`);
796
889
  if (expected.sourceTier && result.sourceTier !== expected.sourceTier)
@@ -849,7 +942,75 @@ function evaluateCase(testCase, result) {
849
942
  }
850
943
  return { failures, validationCode, executionMatched };
851
944
  }
945
+ /**
946
+ * Is the case answerable? Explicit `expected.answerable` wins; otherwise infer
947
+ * from the expectations already present, so the metric covers legacy case files.
948
+ * A case with no expectations at all is excluded — it asserts nothing, so it can
949
+ * neither prove nor disprove a false refusal.
950
+ */
951
+ export function evalCaseIsAnswerable(expected) {
952
+ if (!expected)
953
+ return undefined;
954
+ if (expected.answerable !== undefined)
955
+ return expected.answerable;
956
+ if (expected.kind === 'no_answer')
957
+ return false;
958
+ if (expected.sourceTier === 'no_answer')
959
+ return false;
960
+ if (expected.route === 'clarify')
961
+ return false;
962
+ if (Object.keys(expected).length === 0)
963
+ return undefined;
964
+ return true;
965
+ }
966
+ /**
967
+ * Did the run leave the user with NO way forward?
968
+ *
969
+ * Deliberately narrower than "did not answer". A clarification that offers
970
+ * selectable options is answerable on the next turn — worth minimising, tracked
971
+ * separately as `clarification_rate`, but not the defect. A clarification with
972
+ * ZERO options is a true dead end: the reported production loop was exactly
973
+ * this, and a free-text reply to it reproduced the same question forever.
974
+ */
975
+ export function evalResultRefused(result) {
976
+ // A substantive conversational reply is an ANSWER, not a dead end. A governed
977
+ // definition ("**top_customers** — Top 10 customers by lifetime spend…") is
978
+ // exactly what a "what does X mean?" turn should return, and scoring it as a
979
+ // refusal would report the feature working as the feature failing.
980
+ if (result.conversationalAnswer)
981
+ return false;
982
+ // Order matters: the drivers collapse every clarification to `no_answer`
983
+ // (it is not an answer), so the option check has to run FIRST or an
984
+ // option-bearing clarification is miscounted as a dead end.
985
+ if (result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0)
986
+ return false;
987
+ return result.kind === 'no_answer' || result.route === 'clarify';
988
+ }
989
+ /** Did the run ask an answerable clarification rather than answering outright? */
990
+ export function evalResultClarified(result) {
991
+ return result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0;
992
+ }
993
+ /**
994
+ * Translate only the durable narration receipt into evaluation fields.
995
+ *
996
+ * Reader prose, row count, and result shape are deliberately absent: a skipped
997
+ * narration can have rows, and a deterministic fallback can use any wording.
998
+ */
999
+ export function narrationOutcomeForEval(receipt) {
1000
+ if (receipt?.mode !== 'verified_facts' || !receipt.attempted)
1001
+ return {};
1002
+ return {
1003
+ // A durable receipt is the only source of this metric. An infrastructure
1004
+ // error started a verified narration but did not produce a grounded answer,
1005
+ // so it belongs in the denominator just like the deterministic floor. The
1006
+ // old mapping treated it as a success because only fallback was negative.
1007
+ narrationAttempted: true,
1008
+ narrationFallback: receipt.outcome !== 'success',
1009
+ };
1010
+ }
852
1011
  function computeEvalMetrics(results) {
1012
+ const answerableCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === true);
1013
+ const refusalRequiredCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === false);
853
1014
  const certifiedCases = results.filter((result) => result.expected?.kind === 'certified' ||
854
1015
  result.expected?.certification === 'certified' ||
855
1016
  result.expected?.route === 'certified');
@@ -874,6 +1035,52 @@ function computeEvalMetrics(results) {
874
1035
  wrong_certified_count: results.filter((result) => result.kind === 'certified' &&
875
1036
  (result.expected?.kind ? result.expected.kind !== 'certified' : result.followUp)).length,
876
1037
  outside_context_rejection_count: results.filter((result) => result.validationCode === 'unknown_relation' || result.validationCode === 'unknown_column').length,
1038
+ /**
1039
+ * THE headline number: how often an answerable question was refused.
1040
+ * Bounds every other quality metric — a run that refuses cannot be wrong,
1041
+ * so a falling false-refusal rate must be read together with
1042
+ * `execution_match_rate` to be sure refusals were replaced by CORRECT answers.
1043
+ */
1044
+ false_refusal_rate: ratio(answerableCases.filter(evalResultRefused).length, answerableCases.length),
1045
+ false_refusal_count: answerableCases.filter(evalResultRefused).length,
1046
+ answerable_case_count: answerableCases.length,
1047
+ /**
1048
+ * Answerable cases that asked an option-bearing clarification instead of
1049
+ * answering. Not a defect, but a direct cost in turns — read it next to
1050
+ * false_refusal_rate so a fall in refusals is not just a rise in questions.
1051
+ */
1052
+ clarification_rate: ratio(answerableCases.filter(evalResultClarified).length, answerableCases.length),
1053
+ /**
1054
+ * Cases where semantic judgment ran. Without a provider `mayAssumeInterpretation`
1055
+ * is false (AGT-017), so every ambiguous question clarifies by design and
1056
+ * `clarification_rate` says nothing about product quality.
1057
+ */
1058
+ meaning_resolved_rate: ratio(results.filter((result) => result.meaningResolved === true).length, results.length),
1059
+ /**
1060
+ * Latency, which the acceptance matrix asked for and nothing measured. A
1061
+ * quality gain paid for entirely in wall clock is not a gain: the plan's
1062
+ * two-tier target is certified/semantic under 5s while research takes
1063
+ * minutes, and only a per-class p95 can tell those apart from a regression.
1064
+ */
1065
+ latency_p50_ms: percentileMs(results, 0.5),
1066
+ latency_p95_ms: percentileMs(results, 0.95),
1067
+ latency_p95_answerable_ms: percentileMs(answerableCases, 0.95),
1068
+ /**
1069
+ * How often verified narration survived. When the drafted narration fails
1070
+ * its fact check the reader gets the deterministic record under a
1071
+ * disclaimer — correct, but visibly worse. A silent fall here is exactly
1072
+ * the truncation defect that shipped unnoticed, so it is measured.
1073
+ */
1074
+ grounded_narration_rate: ratio(results.filter((result) => result.narrationAttempted && result.narrationFallback === false).length, results.filter((result) => result.narrationAttempted).length),
1075
+ grounded_narration_attempted: results.filter((result) => result.narrationAttempted).length,
1076
+ /**
1077
+ * The guard on the above: cases that must NOT produce a data answer.
1078
+ * Scored on "did not answer" rather than "dead-ended", because declining via
1079
+ * a clarification is still declining — what would be wrong is asserting
1080
+ * something about data the project does not have.
1081
+ */
1082
+ refusal_recall: ratio(refusalRequiredCases.filter((result) => result.kind === 'no_answer' || result.conversational === true).length, refusalRequiredCases.length),
1083
+ refusal_required_case_count: refusalRequiredCases.length,
877
1084
  draft_saved_count: results.filter((result) => result.draftSaved).length,
878
1085
  tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
879
1086
  avg_tool_calls: average(toolCallCounts),
@@ -885,9 +1092,15 @@ function agentEvalThresholdsPass(metrics, thresholds) {
885
1092
  // A rate threshold with no applicable cases (metric === null) is vacuously
886
1093
  // satisfied — you only fail when the metric exists and falls below the bar.
887
1094
  const rateOk = (metric, min) => min === null || min === undefined || metric === null || metric >= min;
888
- return rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
1095
+ // A ceiling is only meaningful when the metric has data; `null` means no
1096
+ // answerable case was scored, which is "unknown", not "perfect".
1097
+ const ceilingOk = (metric, max) => max === null || max === undefined || metric === null || metric <= max;
1098
+ return rateOk(metrics.grounded_narration_rate, thresholds.minGroundedNarration)
1099
+ && rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
889
1100
  && rateOk(metrics.execution_match_rate, thresholds.minExecutionMatch)
890
1101
  && rateOk(metrics.judge_pass_rate, thresholds.minJudgePass)
1102
+ && ceilingOk(metrics.false_refusal_rate, thresholds.maxFalseRefusal)
1103
+ && rateOk(metrics.refusal_recall, thresholds.minRefusalRecall)
891
1104
  && (thresholds.maxWrongCertified === null
892
1105
  || thresholds.maxWrongCertified === undefined
893
1106
  || metrics.wrong_certified_count <= thresholds.maxWrongCertified);
@@ -1207,6 +1420,20 @@ export const __test__ = {
1207
1420
  cliAnalysisDepth,
1208
1421
  cliReasoningEffort,
1209
1422
  computeEvalMetrics,
1423
+ narrationOutcomeForEval,
1210
1424
  evaluateCase,
1211
1425
  };
1426
+ /** Percentile over observed case durations. Returns null when nothing timed. */
1427
+ function percentileMs(results, q) {
1428
+ const observed = results
1429
+ .map((result) => result.durationMs)
1430
+ .filter((value) => typeof value === 'number' && Number.isFinite(value) && value >= 0)
1431
+ .sort((left, right) => left - right);
1432
+ if (observed.length === 0)
1433
+ return null;
1434
+ // Nearest-rank: with a handful of cases an interpolated percentile invents a
1435
+ // duration nothing actually took.
1436
+ const rank = Math.min(observed.length - 1, Math.max(0, Math.ceil(q * observed.length) - 1));
1437
+ return observed[rank] ?? null;
1438
+ }
1212
1439
  //# sourceMappingURL=agent.js.map