@duckcodeailabs/dql-cli 1.14.0 → 1.14.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/dist/args.d.ts +15 -0
  2. package/dist/args.d.ts.map +1 -1
  3. package/dist/args.js +25 -0
  4. package/dist/args.js.map +1 -1
  5. package/dist/assets/dql-notebook/assets/{AgentLogPage-Ch7VK20X.js → AgentLogPage-DKbGpRQS.js} +1 -1
  6. package/dist/assets/dql-notebook/assets/{AiBuildDialog-DBr5TmyM.js → AiBuildDialog-DPSu0Mly.js} +1 -1
  7. package/dist/assets/dql-notebook/assets/{AiBuildResult-jLPzQO7O.js → AiBuildResult-1uaGnpi1.js} +1 -1
  8. package/dist/assets/dql-notebook/assets/{AiSidePanel-CdlGsiVC.js → AiSidePanel-BwMREwa7.js} +1 -1
  9. package/dist/assets/dql-notebook/assets/{AnalyticsHome-C0DXbOwY.js → AnalyticsHome-BGfey_ve.js} +1 -1
  10. package/dist/assets/dql-notebook/assets/{AppsView-CcpwjApv.js → AppsView-CM1tPywy.js} +4 -4
  11. package/dist/assets/dql-notebook/assets/{BlockStudio-CFYxafw-.js → BlockStudio--S6WFVO4.js} +1 -1
  12. package/dist/assets/dql-notebook/assets/{BusinessArtifactView-C0kLYg2p.js → BusinessArtifactView-BEAJ-yNW.js} +1 -1
  13. package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-CMwElXD_.js → DbtFirstModelingPage-CNyU5MBX.js} +1 -1
  14. package/dist/assets/dql-notebook/assets/{GitPage-JjhRDeWY.js → GitPage-IcydAai3.js} +1 -1
  15. package/dist/assets/dql-notebook/assets/{GlobalAiRail-DakE4NdR.js → GlobalAiRail-CSKW-5eD.js} +1 -1
  16. package/dist/assets/dql-notebook/assets/{GovernedContextPage-BokDqG6a.js → GovernedContextPage-CFen0eFT.js} +1 -1
  17. package/dist/assets/dql-notebook/assets/{HelpDocsPage-CjOv6_gz.js → HelpDocsPage-D0D9UCIz.js} +1 -1
  18. package/dist/assets/dql-notebook/assets/{HomePage-nGaNcwdw.js → HomePage-CmSxapR3.js} +1 -1
  19. package/dist/assets/dql-notebook/assets/{LineageDAG-CO6CFJRg.js → LineageDAG-BGUcIt1B.js} +1 -1
  20. package/dist/assets/dql-notebook/assets/{LineageDetailView-BnGb7OF7.js → LineageDetailView-Dx48Zfzb.js} +1 -1
  21. package/dist/assets/dql-notebook/assets/{LineageDrawer-C5Y0Ht0b.js → LineageDrawer-zaBfSLZO.js} +1 -1
  22. package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-Cgh3GR3F.js → LineagePathBreadcrumb-CTZJp4_r.js} +1 -1
  23. package/dist/assets/dql-notebook/assets/{MiniLineageGraph-Kla9PYuj.js → MiniLineageGraph-CdivNR1S.js} +1 -1
  24. package/dist/assets/dql-notebook/assets/{NewBlockModal-BR2SnmPT.js → NewBlockModal-DMBFB7nE.js} +1 -1
  25. package/dist/assets/dql-notebook/assets/{NewNotebookModal-BaCHMYeB.js → NewNotebookModal-DsC9CWlW.js} +1 -1
  26. package/dist/assets/dql-notebook/assets/{NotebookEditor-CBLqY8cE.js → NotebookEditor-hs-kw8v9.js} +1 -1
  27. package/dist/assets/dql-notebook/assets/{ReadinessPage-CiN0IWSS.js → ReadinessPage-DJ83cMik.js} +1 -1
  28. package/dist/assets/dql-notebook/assets/{SetupOnboarding-BhiCsYF-.js → SetupOnboarding-B1Pu_Bvv.js} +1 -1
  29. package/dist/assets/dql-notebook/assets/{SkillsPage-CCSf8VMm.js → SkillsPage-BHKDay8n.js} +1 -1
  30. package/dist/assets/dql-notebook/assets/{TrustBadge-BgQmFe_x.js → TrustBadge-BkyGgob2.js} +1 -1
  31. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-BYdrTaEW.js +89 -0
  32. package/dist/assets/dql-notebook/assets/{answer-to-notebook-AeDYUDla.js → answer-to-notebook-DmNiuLQA.js} +1 -1
  33. package/dist/assets/dql-notebook/assets/{arrow-left-C55x2hq_.js → arrow-left--1rsrxm8.js} +1 -1
  34. package/dist/assets/dql-notebook/assets/{arrow-right-C1cJrhOm.js → arrow-right-D5TdqY1G.js} +1 -1
  35. package/dist/assets/dql-notebook/assets/{book-open-text-CQf_sdv2.js → book-open-text-Bw7nHbzg.js} +1 -1
  36. package/dist/assets/dql-notebook/assets/{circle-x-X8-Z2yLY.js → circle-x-DLe6NNM4.js} +1 -1
  37. package/dist/assets/dql-notebook/assets/{dagre.esm-BjjNYKyY.js → dagre.esm-CW5QZdBt.js} +1 -1
  38. package/dist/assets/dql-notebook/assets/{external-link-C2zrz5DH.js → external-link-C9Q97sA3.js} +1 -1
  39. package/dist/assets/dql-notebook/assets/{grip-vertical-qGV_PYGU.js → grip-vertical-CztvkIgo.js} +1 -1
  40. package/dist/assets/dql-notebook/assets/{index-ByTDPDaH.js → index-zHHzDn6l.js} +127 -127
  41. package/dist/assets/dql-notebook/assets/{link-2-VpyOxQXG.js → link-2-CiKAvumL.js} +1 -1
  42. package/dist/assets/dql-notebook/assets/{list-tree-CH2Jhwms.js → list-tree-BtnP2nQ5.js} +1 -1
  43. package/dist/assets/dql-notebook/assets/{minimize-2-B2TZJ8BT.js → minimize-2-TSFGxcCP.js} +1 -1
  44. package/dist/assets/dql-notebook/assets/{panel-right-open-BunF88lt.js → panel-right-open-BfXIUWy0.js} +1 -1
  45. package/dist/assets/dql-notebook/assets/{play-DAFVF4_G.js → play-DVbSFJHD.js} +1 -1
  46. package/dist/assets/dql-notebook/assets/{rotate-ccw-DGCrrqtY.js → rotate-ccw-D_cesDcX.js} +1 -1
  47. package/dist/assets/dql-notebook/assets/{semantic-fields-Ci-9QL9F.js → semantic-fields-CoVStdYB.js} +1 -1
  48. package/dist/assets/dql-notebook/assets/{sliders-horizontal-BOAlXXbn.js → sliders-horizontal-l7xV9K5A.js} +1 -1
  49. package/dist/assets/dql-notebook/assets/{star-CkksZXHt.js → star-CSBS0H3b.js} +1 -1
  50. package/dist/assets/dql-notebook/assets/{triangle-alert-D3mjyJZE.js → triangle-alert-BTrnyY4q.js} +1 -1
  51. package/dist/assets/dql-notebook/assets/{upload-CTNOAVEO.js → upload-sySLq9zb.js} +1 -1
  52. package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-DiQjc7x-.js → usePersistedAgentThreadId-CzwgGdus.js} +1 -1
  53. package/dist/assets/dql-notebook/assets/{user-round-BWd5tQRg.js → user-round-ChlgXi9j.js} +1 -1
  54. package/dist/assets/dql-notebook/assets/{wand-sparkles-CqsAv8P-.js → wand-sparkles-CGw0ytyT.js} +1 -1
  55. package/dist/assets/dql-notebook/assets/{workflow-CSqsj-sC.js → workflow-C_RltiK5.js} +1 -1
  56. package/dist/assets/dql-notebook/assets/{wrench-DWqzqlX8.js → wrench-D-wLfeu0.js} +1 -1
  57. package/dist/assets/dql-notebook/assets/{x-65M5rLCE.js → x-B28hIJIC.js} +1 -1
  58. package/dist/assets/dql-notebook/index.html +1 -1
  59. package/dist/commands/agent-eval-cassette.d.ts +164 -0
  60. package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
  61. package/dist/commands/agent-eval-cassette.js +313 -0
  62. package/dist/commands/agent-eval-cassette.js.map +1 -0
  63. package/dist/commands/agent-eval-runtime.d.ts +109 -0
  64. package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
  65. package/dist/commands/agent-eval-runtime.js +165 -0
  66. package/dist/commands/agent-eval-runtime.js.map +1 -0
  67. package/dist/commands/agent.d.ts +124 -4
  68. package/dist/commands/agent.d.ts.map +1 -1
  69. package/dist/commands/agent.js +410 -106
  70. package/dist/commands/agent.js.map +1 -1
  71. package/dist/commands/compile.d.ts +13 -1
  72. package/dist/commands/compile.d.ts.map +1 -1
  73. package/dist/commands/compile.js +36 -4
  74. package/dist/commands/compile.js.map +1 -1
  75. package/dist/commands/sync.d.ts.map +1 -1
  76. package/dist/commands/sync.js +11 -3
  77. package/dist/commands/sync.js.map +1 -1
  78. package/dist/index.js +4 -0
  79. package/dist/index.js.map +1 -1
  80. package/dist/llm/analyst-loop-tools.d.ts +19 -0
  81. package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
  82. package/dist/llm/analyst-loop-tools.js +56 -0
  83. package/dist/llm/analyst-loop-tools.js.map +1 -0
  84. package/dist/llm/providers/dql-agent-provider.d.ts +51 -1
  85. package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
  86. package/dist/llm/providers/dql-agent-provider.js +563 -14
  87. package/dist/llm/providers/dql-agent-provider.js.map +1 -1
  88. package/dist/llm/types.d.ts +47 -1
  89. package/dist/llm/types.d.ts.map +1 -1
  90. package/dist/local-runtime.d.ts +123 -1
  91. package/dist/local-runtime.d.ts.map +1 -1
  92. package/dist/local-runtime.js +1437 -163
  93. package/dist/local-runtime.js.map +1 -1
  94. package/dist/package.json +10 -10
  95. package/package.json +10 -10
  96. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-Bw5AXNDB.js +0 -88
@@ -21,6 +21,8 @@
21
21
  * dql agent feedback <up|down> --block <id> --question "..."
22
22
  * Records feedback into the KG. Used by clients without MCP access.
23
23
  */
24
+ import { answerFromRuntimeRun, driveViaRuntime, projectRuntimeRun, } from './agent-eval-runtime.js';
25
+ import { CassetteStore, cassetteEvidenceSummary, cassetteDirFor, evalCassetteCanonicalizationV2, withCassette, } from './agent-eval-cassette.js';
24
26
  import { existsSync, readFileSync } from 'node:fs';
25
27
  import { join, resolve } from 'node:path';
26
28
  import { load as loadYaml } from 'js-yaml';
@@ -551,7 +553,14 @@ async function runEval(rest, flags) {
551
553
  if (!existsSync(kgPath))
552
554
  await reindexProject(projectRoot, { kgPath });
553
555
  const providerName = flags.provider;
554
- const provider = await pickProvider(providerName);
556
+ const rawProvider = await pickProvider(providerName);
557
+ // Cassettes make a provider-backed suite repeatable and free to re-run. They
558
+ // apply to the in-process driver here; `--via runtime` needs the SERVER
559
+ // started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
560
+ const cassetteMode = flags.cassette;
561
+ const provider = cassetteMode === 'record' || cassetteMode === 'replay'
562
+ ? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode, evalCassetteCanonicalizationV2(projectRoot))
563
+ : rawProvider;
555
564
  const reasoningEffort = cliReasoningEffort(flags);
556
565
  const requestedDepth = cliAnalysisDepth(flags);
557
566
  const kg = new KGStore(kgPath);
@@ -562,10 +571,22 @@ async function runEval(rest, flags) {
562
571
  // gracefully when no provider is available so offline eval stays deterministic.
563
572
  const judge = Boolean(flags.judge);
564
573
  const judgeComplete = async ({ system, user }) => provider.generate([{ role: 'system', content: system }, { role: 'user', content: user }], {});
574
+ // Which half of the stack is under test. `loop` preserves today's behaviour;
575
+ // `runtime` is the one that exercises routing and gates end to end.
576
+ const via = flags.via === 'runtime' ? 'runtime' : 'loop';
565
577
  const runtimeBase = flags.runtimeUrl
566
578
  ?? flags.runtime
567
579
  ?? process.env.DQL_RUNTIME_URL
568
580
  ?? 'http://127.0.0.1:3474';
581
+ if (via === 'runtime') {
582
+ // Fail fast and loudly. Without this, an unreachable server turns every case
583
+ // into a transport error and the report reads as a false-refusal spike that
584
+ // no code change caused.
585
+ const probe = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/health`).catch(() => null);
586
+ if (!probe?.ok) {
587
+ throw new Error(`--via runtime needs a running server at ${runtimeBase}. Start one with \`dql serve\`, or use --via loop to score the answer loop in-process.`);
588
+ }
589
+ }
569
590
  const semanticLayer = loadAgentSemanticLayer(projectRoot);
570
591
  const expandGroundingContext = createGroundingContextExpander(projectRoot);
571
592
  const answerLoopTools = buildAnswerLoopTools(projectRoot);
@@ -605,35 +626,71 @@ async function runEval(rest, flags) {
605
626
  }
606
627
  : undefined,
607
628
  }).catch(() => undefined);
608
- const result = await answer({
609
- question: testCase.question,
610
- domain: testCase.domain,
611
- domainContext: testCase.domain && manifest
612
- ? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
613
- : undefined,
614
- provider,
615
- kg,
616
- manifest: manifest ?? undefined,
617
- skills,
618
- memoryContext,
619
- followUp: testCase.followUp,
620
- semanticLayer,
621
- schemaContext,
622
- contextPack,
623
- reasoningEffort,
624
- analysisDepth: contextBudget.analysisDepth,
625
- expandGroundingContext,
626
- answerLoopTools,
627
- executeCertifiedBlock: execute && manifest
628
- ? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
629
- : undefined,
630
- executeGeneratedSql: execute
631
- ? createGeneratedSqlExecutor(runtimeBase)
632
- : undefined,
633
- captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
634
- const slug = deriveGeneratedDraftSlug(draftQuestion);
635
- const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
636
- if (dqlArtifact?.kind === 'semantic_block') {
629
+ // `--via runtime` posts to a running `dql serve` so the case exercises the
630
+ // router, engine, plan boundary, and gates. The in-process driver below
631
+ // calls the answer loop directly and cannot observe any of them, which is
632
+ // why a refusal metric taken from it reads cleaner than users experience.
633
+ const runtimeRun = via === 'runtime'
634
+ ? await driveViaRuntime({ runtimeBase, question: testCase.question })
635
+ : undefined;
636
+ // Runtime mode is scored from the persisted AgentRun. The transport
637
+ // adapter intentionally has no AgentAnswer.contextPack, so borrowing the
638
+ // local preflight pack here would fabricate retrieval/route evidence for a
639
+ // different execution path.
640
+ const runtimeProjection = runtimeRun ? projectRuntimeRun(runtimeRun) : undefined;
641
+ const result = runtimeRun
642
+ ? answerFromRuntimeRun(runtimeRun)
643
+ : await answer({
644
+ question: testCase.question,
645
+ domain: testCase.domain,
646
+ domainContext: testCase.domain && manifest
647
+ ? resolveDomainContextEnvelope({ manifest, activeDomain: testCase.domain, source: 'explicit_api' })
648
+ : undefined,
649
+ provider,
650
+ kg,
651
+ manifest: manifest ?? undefined,
652
+ skills,
653
+ memoryContext,
654
+ followUp: testCase.followUp,
655
+ semanticLayer,
656
+ schemaContext,
657
+ contextPack,
658
+ reasoningEffort,
659
+ analysisDepth: contextBudget.analysisDepth,
660
+ expandGroundingContext,
661
+ answerLoopTools,
662
+ executeCertifiedBlock: execute && manifest
663
+ ? createCertifiedBlockExecutor(projectRoot, manifest, runtimeBase)
664
+ : undefined,
665
+ executeGeneratedSql: execute
666
+ ? createGeneratedSqlExecutor(runtimeBase)
667
+ : undefined,
668
+ captureGeneratedDraft: ({ question: draftQuestion, sql, intent, followUp, contextPack: draftContextPack, sourceBlock, sourceDqlArtifact, dqlArtifact, proposedEntity, requestedFilters, requestedDimensions, validationWarnings, outputs }) => {
669
+ const slug = deriveGeneratedDraftSlug(draftQuestion);
670
+ const proposedDomain = sourceBlock?.domain ?? draftContextPack?.objects.find((object) => object.domain)?.domain ?? testCase.domain ?? 'misc';
671
+ if (dqlArtifact?.kind === 'semantic_block') {
672
+ if (!flags.save) {
673
+ return {
674
+ path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
675
+ askedTimes: 0,
676
+ proposedContractId: `${proposedDomain}.Unknown.${slug}`,
677
+ };
678
+ }
679
+ return upsertGeneratedDqlArtifactDraft(projectRoot, {
680
+ slug,
681
+ question: draftQuestion,
682
+ proposedContractId: `${proposedDomain}.Unknown.${slug}`,
683
+ proposedDomain,
684
+ dqlArtifact,
685
+ sourceQuestion: followUp?.sourceQuestion,
686
+ sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
687
+ followupKind: followUp?.kind,
688
+ outputs,
689
+ contextPackId: draftContextPack?.id,
690
+ routeIntent: String(intent),
691
+ validationWarnings,
692
+ });
693
+ }
637
694
  if (!flags.save) {
638
695
  return {
639
696
  path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
@@ -641,51 +698,30 @@ async function runEval(rest, flags) {
641
698
  proposedContractId: `${proposedDomain}.Unknown.${slug}`,
642
699
  };
643
700
  }
644
- return upsertGeneratedDqlArtifactDraft(projectRoot, {
701
+ return upsertGeneratedDraft(projectRoot, {
645
702
  slug,
646
703
  question: draftQuestion,
704
+ proposedSql: sql,
647
705
  proposedContractId: `${proposedDomain}.Unknown.${slug}`,
648
706
  proposedDomain,
649
- dqlArtifact,
707
+ proposedEntity,
708
+ sourceDqlArtifact,
650
709
  sourceQuestion: followUp?.sourceQuestion,
651
710
  sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
652
711
  followupKind: followUp?.kind,
712
+ requestedFilters,
713
+ requestedDimensions,
653
714
  outputs,
654
715
  contextPackId: draftContextPack?.id,
655
716
  routeIntent: String(intent),
656
717
  validationWarnings,
657
718
  });
658
- }
659
- if (!flags.save) {
660
- return {
661
- path: previewGeneratedDraftPath(projectRoot, proposedDomain, slug),
662
- askedTimes: 0,
663
- proposedContractId: `${proposedDomain}.Unknown.${slug}`,
664
- };
665
- }
666
- return upsertGeneratedDraft(projectRoot, {
667
- slug,
668
- question: draftQuestion,
669
- proposedSql: sql,
670
- proposedContractId: `${proposedDomain}.Unknown.${slug}`,
671
- proposedDomain,
672
- proposedEntity,
673
- sourceDqlArtifact,
674
- sourceQuestion: followUp?.sourceQuestion,
675
- sourceBlock: followUp?.sourceBlockName ?? sourceBlock?.name,
676
- followupKind: followUp?.kind,
677
- requestedFilters,
678
- requestedDimensions,
679
- outputs,
680
- contextPackId: draftContextPack?.id,
681
- routeIntent: String(intent),
682
- validationWarnings,
683
- });
684
- },
685
- });
686
- const evaluation = evaluateCase(testCase, result);
719
+ },
720
+ });
721
+ const evaluation = evaluateCase(testCase, result, runtimeProjection);
687
722
  const durationMs = Date.now() - startedAt;
688
723
  const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
724
+ const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
689
725
  const judgeVerdict = judge
690
726
  ? await judgeAnswer({
691
727
  question: testCase.question,
@@ -700,17 +736,33 @@ async function runEval(rest, flags) {
700
736
  passed: evaluation.failures.length === 0,
701
737
  failures: evaluation.failures,
702
738
  durationMs,
739
+ // The persisted receipt is the only evidence for this metric. Rows and
740
+ // reader prose are intentionally ignored: a row-bearing answer can be
741
+ // skipped, while a deterministic fallback can render different wording.
742
+ narrationAttempted: narration.narrationAttempted,
743
+ narrationFallback: narration.narrationFallback,
703
744
  executionMs: result.result?.executionTime,
704
745
  executionMatched: evaluation.executionMatched,
705
746
  ...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
706
747
  kind: result.kind,
707
- route: result.contextPack?.routeDecision.route,
708
- intent: result.contextPack?.routeDecision.intent,
748
+ route: runtimeProjection?.route ?? result.contextPack?.routeDecision.route,
749
+ // Only the runtime driver can see the router's clarification options.
750
+ // In-process runs leave this undefined, so a clarify there scores as a
751
+ // dead end — the conservative reading, and another reason `--via runtime`
752
+ // is the truthful one.
753
+ ...(runtimeRun ? {
754
+ clarificationOptionCount: runtimeRun.clarificationOptions?.length ?? 0,
755
+ conversational: runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational',
756
+ conversationalAnswer: (runtimeRun.route === 'conversation' || runtimeRun.answerKind === 'conversational')
757
+ && Boolean(runtimeRun.answer?.trim()),
758
+ meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
759
+ } : {}),
760
+ intent: runtimeRun?.routeDecision?.category ?? result.contextPack?.routeDecision.intent,
709
761
  reviewStatus: result.reviewStatus,
710
- contextObjects: result.contextPack?.objects.length ?? 0,
762
+ contextObjects: runtimeProjection?.retrievalCandidateCount ?? result.contextPack?.objects.length,
711
763
  followUp: Boolean(testCase.followUp),
712
764
  draftSaved,
713
- toolCalls: result.evidence?.toolCalls?.length ?? 0,
765
+ toolCalls: runtimeProjection?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0,
714
766
  expected: testCase.expected,
715
767
  validationCode: evaluation.validationCode,
716
768
  trace: buildEvalTrace({
@@ -719,6 +771,7 @@ async function runEval(rest, flags) {
719
771
  evaluation,
720
772
  durationMs,
721
773
  draftSaved,
774
+ runtime: runtimeProjection,
722
775
  }),
723
776
  });
724
777
  }
@@ -729,16 +782,39 @@ async function runEval(rest, flags) {
729
782
  }
730
783
  const passed = results.filter((r) => r.passed).length;
731
784
  const metrics = computeEvalMetrics(results);
785
+ // Runtime evals execute in a separate host, so its cassette directory is
786
+ // supplied explicitly by the eval workflow for reporting. The summary does
787
+ // not inspect prompts or provider credentials; it only classifies the
788
+ // checked-in response provenance.
789
+ const cassetteDirectory = via === 'runtime'
790
+ ? process.env.DQL_EVAL_CASSETTE_DIR
791
+ : cassetteMode === 'record' || cassetteMode === 'replay'
792
+ ? cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')
793
+ : undefined;
794
+ const cassetteEvidence = cassetteDirectory
795
+ ? cassetteEvidenceSummary(new CassetteStore(cassetteDirectory))
796
+ : undefined;
732
797
  const thresholds = {
733
798
  minToolRequirement: flags.minToolRequirement ?? null,
734
799
  minExecutionMatch: flags.minExecutionMatch ?? null,
735
800
  minJudgePass: flags.minJudgePass ?? null,
736
801
  maxWrongCertified: flags.maxWrongCertified ?? null,
802
+ maxFalseRefusal: flags.maxFalseRefusal ?? null,
803
+ minRefusalRecall: flags.minRefusalRecall ?? null,
804
+ minGroundedNarration: flags.minGroundedNarration ?? null,
737
805
  };
738
806
  const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
739
807
  const ok = passed === results.length && thresholdsPassed;
740
808
  if (flags.format === 'json') {
741
- console.log(JSON.stringify({ ok, passed, total: results.length, thresholds, metrics, results }, null, 2));
809
+ console.log(JSON.stringify({
810
+ ok,
811
+ passed,
812
+ total: results.length,
813
+ thresholds,
814
+ metrics,
815
+ ...(cassetteEvidence ? { cassetteEvidence } : {}),
816
+ results,
817
+ }, null, 2));
742
818
  if (!ok)
743
819
  process.exitCode = 1;
744
820
  return;
@@ -752,6 +828,22 @@ async function runEval(rest, flags) {
752
828
  console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
753
829
  console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
754
830
  console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
831
+ if (cassetteEvidence) {
832
+ console.log(`Cassette replay entries: ${cassetteEvidence.totalEntries} `
833
+ + `(${cassetteEvidence.migratedLegacyDeterministicFixtureEntries} migrated legacy deterministic fixture, `
834
+ + `${cassetteEvidence.syntheticDeterministicOrchestrationFixtureEntries} synthetic deterministic orchestration fixture).`);
835
+ console.log(cassetteEvidence.realProviderQualityEligible
836
+ ? 'Real-provider quality evidence: eligible.'
837
+ : `Real-provider quality evidence: excluded (${cassetteEvidence.realProviderQualityExclusionReasons.join(', ')}).`);
838
+ }
839
+ console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
840
+ console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
841
+ if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
842
+ console.log(` ! Semantic judgment ran for only ${formatRate(metrics.meaning_resolved_rate)} of cases. `
843
+ + 'Without a reachable provider DQL will not settle a reading by lexical rank (AGT-017), so ambiguous '
844
+ + 'questions clarify by design — treat the clarification rate above as an artifact, not a product signal.');
845
+ }
846
+ console.log(`Refusal recall: ${formatRate(metrics.refusal_recall)} (${metrics.refusal_required_case_count} case(s) that must refuse)`);
755
847
  console.log(`Execution match rate: ${formatRate(metrics.execution_match_rate)}`);
756
848
  console.log(`Tool requirement pass rate: ${formatRate(metrics.tool_requirement_pass_rate)}`);
757
849
  console.log(`Tool-observed case count: ${metrics.tool_observed_case_count}`);
@@ -767,6 +859,15 @@ async function runEval(rest, flags) {
767
859
  if (thresholds.minJudgePass !== null) {
768
860
  console.log(`Judge-pass threshold: ${thresholds.minJudgePass} (actual ${formatRate(metrics.judge_pass_rate)})`);
769
861
  }
862
+ if (thresholds.maxFalseRefusal !== null) {
863
+ console.log(`False-refusal ceiling: ${thresholds.maxFalseRefusal} (actual ${formatRate(metrics.false_refusal_rate)})`);
864
+ }
865
+ if (thresholds.minRefusalRecall !== null) {
866
+ console.log(`Refusal-recall threshold: ${thresholds.minRefusalRecall} (actual ${formatRate(metrics.refusal_recall)})`);
867
+ }
868
+ if (thresholds.minGroundedNarration !== null && thresholds.minGroundedNarration !== undefined) {
869
+ console.log(`Grounded-narration threshold: ${thresholds.minGroundedNarration} (actual ${formatRate(metrics.grounded_narration_rate)} over ${metrics.grounded_narration_attempted} attempted)`);
870
+ }
770
871
  if (thresholds.maxWrongCertified !== null) {
771
872
  console.log(`Wrong-certified ceiling: ${thresholds.maxWrongCertified} (actual ${metrics.wrong_certified_count})`);
772
873
  }
@@ -784,13 +885,39 @@ function previewGeneratedDraftPath(projectRoot, domain, slug) {
784
885
  }
785
886
  return `blocks/_drafts/${slug}.dql`;
786
887
  }
787
- function evaluateCase(testCase, result) {
888
+ function evaluateCase(testCase, result, runtime) {
788
889
  const expected = testCase.expected;
789
890
  if (!expected)
790
891
  return { failures: [] };
791
892
  const failures = [];
792
893
  let validationCode;
793
894
  let executionMatched;
895
+ // Answerability is asserted per case as well as aggregated into
896
+ // false_refusal_rate, so a single dead-end fails its own case instead of only
897
+ // nudging a rate someone has to notice.
898
+ const answerable = evalCaseIsAnswerable(expected);
899
+ // Three distinct outcomes, not two. An option-bearing clarification neither
900
+ // answers nor dead-ends: it must not fail an answerable case (its cost is
901
+ // tracked by `clarification_rate`), and it must not fail a must-refuse case
902
+ // either, because it did not assert anything about the data.
903
+ const clarifiedWithOptions = (result.clarificationOptions?.length ?? 0) > 0;
904
+ // A conversational reply ("I'm here to help you explore your data…") asserts
905
+ // nothing about the warehouse. For an out-of-scope question that is the CORRECT
906
+ // outcome — declining politely — so it must not be scored as an answer.
907
+ const conversational = result.answerKind === 'conversational';
908
+ const answerText = typeof result.text === 'string' ? result.text.trim() : '';
909
+ // Conversational replies split two ways, and the case's own expectation says
910
+ // which is right: for an answerable question a substantive reply IS the answer
911
+ // (a definition), and for an out-of-scope one it is the correct decline.
912
+ const conversationalAnswer = conversational && answerText.length > 0;
913
+ const producedDataAnswer = result.kind !== 'no_answer' && !conversational;
914
+ const deadEnded = !producedDataAnswer && !clarifiedWithOptions && !conversationalAnswer;
915
+ if (answerable === true && deadEnded) {
916
+ failures.push(`FALSE REFUSAL: this question is answerable, but the run dead-ended with no answer and no options${result.refusalCode ? ` (${result.refusalCode})` : ''}`);
917
+ }
918
+ if (answerable === false && producedDataAnswer) {
919
+ failures.push(`expected a refusal (question is out of scope / unanswerable), but the run answered with kind ${result.kind}`);
920
+ }
794
921
  if (expected.kind && result.kind !== expected.kind)
795
922
  failures.push(`kind expected ${expected.kind}, got ${result.kind}`);
796
923
  if (expected.sourceTier && result.sourceTier !== expected.sourceTier)
@@ -799,12 +926,27 @@ function evaluateCase(testCase, result) {
799
926
  failures.push(`certification expected ${expected.certification}, got ${result.certification}`);
800
927
  if (expected.reviewStatus && result.reviewStatus !== expected.reviewStatus)
801
928
  failures.push(`reviewStatus expected ${expected.reviewStatus}, got ${result.reviewStatus}`);
802
- if (expected.route && result.contextPack?.routeDecision.route !== expected.route)
803
- failures.push(`route expected ${expected.route}, got ${result.contextPack?.routeDecision.route ?? 'none'}`);
929
+ const observedRoute = runtime?.route ?? result.contextPack?.routeDecision.route;
930
+ if (expected.route && observedRoute !== expected.route)
931
+ failures.push(`route expected ${expected.route}, got ${observedRoute ?? 'none'}`);
804
932
  if (expected.intent && result.contextPack?.routeDecision.intent !== expected.intent)
805
933
  failures.push(`intent expected ${expected.intent}, got ${result.contextPack?.routeDecision.intent ?? 'none'}`);
806
- if (expected.missingContextKind && !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind))
934
+ // `modeling_gap` is a broad terminal kind. A relationship expectation is
935
+ // satisfied only by the router's persisted relationship-specific witness;
936
+ // otherwise a missing metric/dimension tuple would be misreported as a
937
+ // relationship repair opportunity.
938
+ const runtimeReportsMissingContext = expected.missingContextKind === 'relationship'
939
+ ? runtime?.terminalOutcome?.gap?.code === 'MISSING_RELATIONSHIP'
940
+ : runtime?.terminalOutcome?.kind === 'modeling_gap'
941
+ && expected.missingContextKind === 'modeling_gap';
942
+ if (expected.missingContextKind
943
+ && !runtimeReportsMissingContext
944
+ && !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind)) {
807
945
  failures.push(`missing context kind ${expected.missingContextKind} was not reported`);
946
+ }
947
+ if (expected.terminalOutcomeKind && runtime?.terminalOutcome?.kind !== expected.terminalOutcomeKind) {
948
+ failures.push(`terminal outcome expected ${expected.terminalOutcomeKind}, got ${runtime?.terminalOutcome?.kind ?? 'none'}`);
949
+ }
808
950
  for (const token of stringList(expected.sqlContains)) {
809
951
  if (!result.proposedSql?.toLowerCase().includes(token.toLowerCase()))
810
952
  failures.push(`SQL did not contain "${token}"`);
@@ -825,7 +967,7 @@ function evaluateCase(testCase, result) {
825
967
  failures.push(`draftSaved expected ${expected.draftSaved}, got ${saved}`);
826
968
  }
827
969
  if (typeof expected.minToolCalls === 'number') {
828
- const actualToolCalls = result.evidence?.toolCalls?.length ?? 0;
970
+ const actualToolCalls = runtime?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0;
829
971
  if (actualToolCalls < expected.minToolCalls) {
830
972
  failures.push(`toolCalls expected at least ${expected.minToolCalls}, got ${actualToolCalls}`);
831
973
  }
@@ -849,7 +991,75 @@ function evaluateCase(testCase, result) {
849
991
  }
850
992
  return { failures, validationCode, executionMatched };
851
993
  }
994
+ /**
995
+ * Is the case answerable? Explicit `expected.answerable` wins; otherwise infer
996
+ * from the expectations already present, so the metric covers legacy case files.
997
+ * A case with no expectations at all is excluded — it asserts nothing, so it can
998
+ * neither prove nor disprove a false refusal.
999
+ */
1000
+ export function evalCaseIsAnswerable(expected) {
1001
+ if (!expected)
1002
+ return undefined;
1003
+ if (expected.answerable !== undefined)
1004
+ return expected.answerable;
1005
+ if (expected.kind === 'no_answer')
1006
+ return false;
1007
+ if (expected.sourceTier === 'no_answer')
1008
+ return false;
1009
+ if (expected.route === 'clarify' || expected.route === 'blocked')
1010
+ return false;
1011
+ if (Object.keys(expected).length === 0)
1012
+ return undefined;
1013
+ return true;
1014
+ }
1015
+ /**
1016
+ * Did the run leave the user with NO way forward?
1017
+ *
1018
+ * Deliberately narrower than "did not answer". A clarification that offers
1019
+ * selectable options is answerable on the next turn — worth minimising, tracked
1020
+ * separately as `clarification_rate`, but not the defect. A clarification with
1021
+ * ZERO options is a true dead end: the reported production loop was exactly
1022
+ * this, and a free-text reply to it reproduced the same question forever.
1023
+ */
1024
+ export function evalResultRefused(result) {
1025
+ // A substantive conversational reply is an ANSWER, not a dead end. A governed
1026
+ // definition ("**top_customers** — Top 10 customers by lifetime spend…") is
1027
+ // exactly what a "what does X mean?" turn should return, and scoring it as a
1028
+ // refusal would report the feature working as the feature failing.
1029
+ if (result.conversationalAnswer)
1030
+ return false;
1031
+ // Order matters: the drivers collapse every clarification to `no_answer`
1032
+ // (it is not an answer), so the option check has to run FIRST or an
1033
+ // option-bearing clarification is miscounted as a dead end.
1034
+ if (result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0)
1035
+ return false;
1036
+ return result.kind === 'no_answer' || result.route === 'clarify';
1037
+ }
1038
+ /** Did the run ask an answerable clarification rather than answering outright? */
1039
+ export function evalResultClarified(result) {
1040
+ return result.route === 'clarify' && (result.clarificationOptionCount ?? 0) > 0;
1041
+ }
1042
+ /**
1043
+ * Translate only the durable narration receipt into evaluation fields.
1044
+ *
1045
+ * Reader prose, row count, and result shape are deliberately absent: a skipped
1046
+ * narration can have rows, and a deterministic fallback can use any wording.
1047
+ */
1048
+ export function narrationOutcomeForEval(receipt) {
1049
+ if (receipt?.mode !== 'verified_facts' || !receipt.attempted)
1050
+ return {};
1051
+ return {
1052
+ // A durable receipt is the only source of this metric. An infrastructure
1053
+ // error started a verified narration but did not produce a grounded answer,
1054
+ // so it belongs in the denominator just like the deterministic floor. The
1055
+ // old mapping treated it as a success because only fallback was negative.
1056
+ narrationAttempted: true,
1057
+ narrationFallback: receipt.outcome !== 'success',
1058
+ };
1059
+ }
852
1060
  function computeEvalMetrics(results) {
1061
+ const answerableCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === true);
1062
+ const refusalRequiredCases = results.filter((result) => evalCaseIsAnswerable(result.expected) === false);
853
1063
  const certifiedCases = results.filter((result) => result.expected?.kind === 'certified' ||
854
1064
  result.expected?.certification === 'certified' ||
855
1065
  result.expected?.route === 'certified');
@@ -874,10 +1084,61 @@ function computeEvalMetrics(results) {
874
1084
  wrong_certified_count: results.filter((result) => result.kind === 'certified' &&
875
1085
  (result.expected?.kind ? result.expected.kind !== 'certified' : result.followUp)).length,
876
1086
  outside_context_rejection_count: results.filter((result) => result.validationCode === 'unknown_relation' || result.validationCode === 'unknown_column').length,
1087
+ /**
1088
+ * THE headline number: how often an answerable question was refused.
1089
+ * Bounds every other quality metric — a run that refuses cannot be wrong,
1090
+ * so a falling false-refusal rate must be read together with
1091
+ * `execution_match_rate` to be sure refusals were replaced by CORRECT answers.
1092
+ */
1093
+ false_refusal_rate: ratio(answerableCases.filter(evalResultRefused).length, answerableCases.length),
1094
+ false_refusal_count: answerableCases.filter(evalResultRefused).length,
1095
+ answerable_case_count: answerableCases.length,
1096
+ /**
1097
+ * Answerable cases that asked an option-bearing clarification instead of
1098
+ * answering. Not a defect, but a direct cost in turns — read it next to
1099
+ * false_refusal_rate so a fall in refusals is not just a rise in questions.
1100
+ */
1101
+ clarification_rate: ratio(answerableCases.filter(evalResultClarified).length, answerableCases.length),
1102
+ /**
1103
+ * Cases where semantic judgment ran. Without a provider `mayAssumeInterpretation`
1104
+ * is false (AGT-017), so every ambiguous question clarifies by design and
1105
+ * `clarification_rate` says nothing about product quality.
1106
+ */
1107
+ meaning_resolved_rate: ratio(results.filter((result) => result.meaningResolved === true).length, results.length),
1108
+ /**
1109
+ * Latency, which the acceptance matrix asked for and nothing measured. A
1110
+ * quality gain paid for entirely in wall clock is not a gain: the plan's
1111
+ * two-tier target is certified/semantic under 5s while research takes
1112
+ * minutes, and only a per-class p95 can tell those apart from a regression.
1113
+ */
1114
+ latency_p50_ms: percentileMs(results, 0.5),
1115
+ latency_p95_ms: percentileMs(results, 0.95),
1116
+ latency_p95_answerable_ms: percentileMs(answerableCases, 0.95),
1117
+ /**
1118
+ * How often verified narration survived. When the drafted narration fails
1119
+ * its fact check the reader gets the deterministic record under a
1120
+ * disclaimer — correct, but visibly worse. A silent fall here is exactly
1121
+ * the truncation defect that shipped unnoticed, so it is measured.
1122
+ */
1123
+ grounded_narration_rate: ratio(results.filter((result) => result.narrationAttempted && result.narrationFallback === false).length, results.filter((result) => result.narrationAttempted).length),
1124
+ grounded_narration_attempted: results.filter((result) => result.narrationAttempted).length,
1125
+ /**
1126
+ * The guard on the above: cases that must NOT produce a data answer.
1127
+ * Scored on "did not answer" rather than "dead-ended", because declining via
1128
+ * a clarification is still declining — what would be wrong is asserting
1129
+ * something about data the project does not have.
1130
+ */
1131
+ refusal_recall: ratio(refusalRequiredCases.filter((result) => result.kind === 'no_answer' || result.conversational === true).length, refusalRequiredCases.length),
1132
+ refusal_required_case_count: refusalRequiredCases.length,
877
1133
  draft_saved_count: results.filter((result) => result.draftSaved).length,
878
1134
  tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
879
1135
  avg_tool_calls: average(toolCallCounts),
880
- avg_context_objects: average(results.map((result) => result.contextObjects)),
1136
+ // Runtime runs only report this when the persisted router recorded an
1137
+ // explicit retrieval count. Treat absent evidence as unknown, not as an
1138
+ // invented empty context pack.
1139
+ avg_context_objects: average(results
1140
+ .map((result) => result.contextObjects)
1141
+ .filter((count) => typeof count === 'number')),
881
1142
  avg_execution_ms: executionTimes.length ? average(executionTimes) : null,
882
1143
  };
883
1144
  }
@@ -885,20 +1146,27 @@ function agentEvalThresholdsPass(metrics, thresholds) {
885
1146
  // A rate threshold with no applicable cases (metric === null) is vacuously
886
1147
  // satisfied — you only fail when the metric exists and falls below the bar.
887
1148
  const rateOk = (metric, min) => min === null || min === undefined || metric === null || metric >= min;
888
- return rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
1149
+ // A ceiling is only meaningful when the metric has data; `null` means no
1150
+ // answerable case was scored, which is "unknown", not "perfect".
1151
+ const ceilingOk = (metric, max) => max === null || max === undefined || metric === null || metric <= max;
1152
+ return rateOk(metrics.grounded_narration_rate, thresholds.minGroundedNarration)
1153
+ && rateOk(metrics.tool_requirement_pass_rate, thresholds.minToolRequirement)
889
1154
  && rateOk(metrics.execution_match_rate, thresholds.minExecutionMatch)
890
1155
  && rateOk(metrics.judge_pass_rate, thresholds.minJudgePass)
1156
+ && ceilingOk(metrics.false_refusal_rate, thresholds.maxFalseRefusal)
1157
+ && rateOk(metrics.refusal_recall, thresholds.minRefusalRecall)
891
1158
  && (thresholds.maxWrongCertified === null
892
1159
  || thresholds.maxWrongCertified === undefined
893
1160
  || metrics.wrong_certified_count <= thresholds.maxWrongCertified);
894
1161
  }
895
1162
  function buildEvalTrace(input) {
896
- const { testCase, result, evaluation, durationMs, draftSaved } = input;
1163
+ const { testCase, result, evaluation, durationMs, draftSaved, runtime } = input;
897
1164
  const routeDecision = result.contextPack?.routeDecision;
898
1165
  const selectedRelations = result.contextPack?.retrievalDiagnostics.selectedRelations ?? [];
899
1166
  const allowedRelations = result.contextPack?.allowedSqlContext?.relations ?? [];
900
1167
  const followUp = testCase.followUp;
901
1168
  const toolCalls = result.evidence?.toolCalls ?? [];
1169
+ const observedToolCallCount = runtime?.toolCallCount ?? toolCalls.length;
902
1170
  const routeEvidence = result.evidence?.route ?? [];
903
1171
  const executionStatus = result.executionError
904
1172
  ? 'failed'
@@ -914,33 +1182,44 @@ function buildEvalTrace(input) {
914
1182
  const rowsExpected = testCase.expected?.rows !== undefined;
915
1183
  const expectedMinToolCalls = testCase.expected?.minToolCalls;
916
1184
  const toolStatus = typeof expectedMinToolCalls === 'number'
917
- ? toolCalls.length >= expectedMinToolCalls ? 'passed' : 'failed'
918
- : toolCalls.length > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
1185
+ ? observedToolCallCount >= expectedMinToolCalls ? 'passed' : 'failed'
1186
+ : observedToolCallCount > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
919
1187
  const toolMessage = typeof expectedMinToolCalls === 'number'
920
- ? toolCalls.length >= expectedMinToolCalls
921
- ? `Observed ${toolCalls.length} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
922
- : `Observed ${toolCalls.length} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
923
- : toolCalls.length > 0
924
- ? `Observed ${toolCalls.length} provider tool call(s).`
1188
+ ? observedToolCallCount >= expectedMinToolCalls
1189
+ ? `Observed ${observedToolCallCount} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
1190
+ : `Observed ${observedToolCallCount} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
1191
+ : observedToolCallCount > 0
1192
+ ? `Observed ${observedToolCallCount} provider tool call(s).`
925
1193
  : routeEvidence.length > 0
926
1194
  ? `Captured ${routeEvidence.length} deterministic route evidence step(s).`
927
1195
  : 'No provider tool calls were observed for this answer.';
928
1196
  return [
929
1197
  {
930
1198
  stage: 'context',
931
- status: result.contextPack ? 'passed' : 'not_run',
932
- message: result.contextPack
933
- ? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
934
- : 'No context pack was attached to the answer.',
935
- payload: result.contextPack
1199
+ status: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
1200
+ ? 'passed'
1201
+ : result.contextPack ? 'passed' : 'not_run',
1202
+ message: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
1203
+ ? `Persisted route evidence recorded ${runtime.retrievalCandidateCount ?? 'an unspecified number of'} retrieved candidate(s).`
1204
+ : result.contextPack
1205
+ ? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
1206
+ : 'No context pack was attached to the answer.',
1207
+ payload: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
936
1208
  ? {
937
- contextPackId: result.contextPack.id,
938
- selectedObjectCount: result.contextPack.objects.length,
939
- allowedRelationCount: allowedRelations.length,
940
- selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
941
- missingContext: result.contextPack.missingContext,
1209
+ evidenceSource: 'persisted_agent_run',
1210
+ retrievalCandidateCount: runtime.retrievalCandidateCount,
1211
+ sourceCoverage: runtime.sourceCoverage,
1212
+ terminalOutcome: runtime.terminalOutcome,
942
1213
  }
943
- : undefined,
1214
+ : result.contextPack
1215
+ ? {
1216
+ contextPackId: result.contextPack.id,
1217
+ selectedObjectCount: result.contextPack.objects.length,
1218
+ allowedRelationCount: allowedRelations.length,
1219
+ selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
1220
+ missingContext: result.contextPack.missingContext,
1221
+ }
1222
+ : undefined,
944
1223
  },
945
1224
  {
946
1225
  stage: 'rewrite',
@@ -952,29 +1231,40 @@ function buildEvalTrace(input) {
952
1231
  },
953
1232
  {
954
1233
  stage: 'lane',
955
- status: routeDecision ? 'passed' : 'not_run',
956
- message: routeDecision
957
- ? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
958
- : 'No lane decision was attached to the answer.',
959
- payload: routeDecision
1234
+ status: runtime ? 'passed' : routeDecision ? 'passed' : 'not_run',
1235
+ message: runtime
1236
+ ? `Persisted engine route ${runtime.runRoute}${runtime.route ? ` evaluated as ${runtime.route}` : ''}.`
1237
+ : routeDecision
1238
+ ? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
1239
+ : 'No lane decision was attached to the answer.',
1240
+ payload: runtime
960
1241
  ? {
961
- route: routeDecision.route,
962
- intent: routeDecision.intent,
963
- reason: routeDecision.reason,
964
- trustLabel: routeDecision.trustLabel,
965
- reviewStatus: routeDecision.reviewStatus,
966
- exactObjectKey: routeDecision.exactObjectKey,
1242
+ engineRoute: runtime.runRoute,
1243
+ evalRoute: runtime.route,
1244
+ status: runtime.status,
1245
+ trustState: runtime.trustState,
1246
+ terminalOutcome: runtime.terminalOutcome,
967
1247
  }
968
- : undefined,
1248
+ : routeDecision
1249
+ ? {
1250
+ route: routeDecision.route,
1251
+ intent: routeDecision.intent,
1252
+ reason: routeDecision.reason,
1253
+ trustLabel: routeDecision.trustLabel,
1254
+ reviewStatus: routeDecision.reviewStatus,
1255
+ exactObjectKey: routeDecision.exactObjectKey,
1256
+ }
1257
+ : undefined,
969
1258
  },
970
1259
  {
971
1260
  stage: 'tools',
972
1261
  status: toolStatus,
973
1262
  message: toolMessage,
974
1263
  payload: {
975
- observedToolCalls: toolCalls.length,
1264
+ observedToolCalls: observedToolCallCount,
976
1265
  expectedMinToolCalls,
977
- providerToolCalls: toolCalls.slice(0, 12).map((call) => ({
1266
+ ...(runtime ? { evidenceSource: 'persisted_agent_run.telemetry' } : {}),
1267
+ providerToolCalls: runtime ? [] : toolCalls.slice(0, 12).map((call) => ({
978
1268
  order: call.order,
979
1269
  name: call.name,
980
1270
  status: call.status,
@@ -1207,6 +1497,20 @@ export const __test__ = {
1207
1497
  cliAnalysisDepth,
1208
1498
  cliReasoningEffort,
1209
1499
  computeEvalMetrics,
1500
+ narrationOutcomeForEval,
1210
1501
  evaluateCase,
1211
1502
  };
1503
+ /** Percentile over observed case durations. Returns null when nothing timed. */
1504
+ function percentileMs(results, q) {
1505
+ const observed = results
1506
+ .map((result) => result.durationMs)
1507
+ .filter((value) => typeof value === 'number' && Number.isFinite(value) && value >= 0)
1508
+ .sort((left, right) => left - right);
1509
+ if (observed.length === 0)
1510
+ return null;
1511
+ // Nearest-rank: with a handful of cases an interpolated percentile invents a
1512
+ // duration nothing actually took.
1513
+ const rank = Math.min(observed.length - 1, Math.max(0, Math.ceil(q * observed.length) - 1));
1514
+ return observed[rank] ?? null;
1515
+ }
1212
1516
  //# sourceMappingURL=agent.js.map