@duckcodeailabs/dql-cli 1.14.1 → 1.14.3-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/dist/args.d.ts +4 -0
  2. package/dist/args.d.ts.map +1 -1
  3. package/dist/args.js +16 -0
  4. package/dist/args.js.map +1 -1
  5. package/dist/assets/dql-notebook/assets/{AgentLogPage-BPz-UWFh.js → AgentLogPage-Jbyb4so-.js} +1 -1
  6. package/dist/assets/dql-notebook/assets/{AiBuildDialog-BEl53WA_.js → AiBuildDialog-Du07X1qO.js} +1 -1
  7. package/dist/assets/dql-notebook/assets/{AiBuildResult-B4yfGTTZ.js → AiBuildResult-xKtGthWv.js} +1 -1
  8. package/dist/assets/dql-notebook/assets/{AiSidePanel-CSZAAvuD.js → AiSidePanel-CE_L04dK.js} +1 -1
  9. package/dist/assets/dql-notebook/assets/AnalyticsHome-CiUE10uF.js +6 -0
  10. package/dist/assets/dql-notebook/assets/{AppsView-DQOwU9Cg.js → AppsView-DlNaTuwD.js} +30 -35
  11. package/dist/assets/dql-notebook/assets/AskObservabilityPage-DracZPI9.js +1 -0
  12. package/dist/assets/dql-notebook/assets/AskTracePage-CSMoMw9a.js +8 -0
  13. package/dist/assets/dql-notebook/assets/{BlockStudio-B4ap0GdY.js → BlockStudio-DQS5Hcq9.js} +9 -14
  14. package/dist/assets/dql-notebook/assets/BusinessArtifactView-CqV4i9l_.js +1 -0
  15. package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-D72byb2g.js → DbtFirstModelingPage-JKhB_Y2H.js} +3 -3
  16. package/dist/assets/dql-notebook/assets/{GitPage-lQb1uXH2.js → GitPage-DjaNp0o3.js} +1 -1
  17. package/dist/assets/dql-notebook/assets/{GlobalAiRail-CVECf6Xj.js → GlobalAiRail-CsV3zlc8.js} +1 -1
  18. package/dist/assets/dql-notebook/assets/{GovernedContextPage-trOyMCY6.js → GovernedContextPage-uXUHUOIJ.js} +3 -3
  19. package/dist/assets/dql-notebook/assets/{HelpDocsPage-D8hLS5lE.js → HelpDocsPage-Hbh9IeYB.js} +1 -1
  20. package/dist/assets/dql-notebook/assets/{HomePage-eIkfBIep.js → HomePage-CRRl-R_v.js} +1 -1
  21. package/dist/assets/dql-notebook/assets/LineageDAG-OMUP2sbl.js +1 -0
  22. package/dist/assets/dql-notebook/assets/LineageDetailView-C964BLuV.js +1 -0
  23. package/dist/assets/dql-notebook/assets/LineageDrawer-CzY0nfpD.js +1 -0
  24. package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-CviIf8PN.js → LineagePathBreadcrumb-0Gi-Fm1B.js} +1 -1
  25. package/dist/assets/dql-notebook/assets/MiniLineageGraph-wn0wLc9m.js +1 -0
  26. package/dist/assets/dql-notebook/assets/{NewBlockModal-DbCQg-pj.js → NewBlockModal-kMAzk91d.js} +1 -1
  27. package/dist/assets/dql-notebook/assets/{NewNotebookModal-5Vp6XiuK.js → NewNotebookModal-D_cWd_6M.js} +1 -1
  28. package/dist/assets/dql-notebook/assets/{NotebookEditor-DjqJS44s.js → NotebookEditor-Bax0lxaz.js} +25 -30
  29. package/dist/assets/dql-notebook/assets/{ReadinessPage-BHGrC5ho.js → ReadinessPage-DEUIXQQY.js} +1 -1
  30. package/dist/assets/dql-notebook/assets/{SetupOnboarding-BfS9Tdx-.js → SetupOnboarding-CyE-Ex41.js} +1 -1
  31. package/dist/assets/dql-notebook/assets/{SkillsPage-BFH21wSj.js → SkillsPage-QgvKGPxE.js} +1 -1
  32. package/dist/assets/dql-notebook/assets/{TrustBadge-zm6g_SxZ.js → TrustBadge-D5CGp0cu.js} +1 -1
  33. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-BjsTE6yf.js +89 -0
  34. package/dist/assets/dql-notebook/assets/{answer-to-notebook-DPhxIEzF.js → answer-to-notebook-tACdCsTV.js} +1 -1
  35. package/dist/assets/dql-notebook/assets/{arrow-left-DNEb86Xc.js → arrow-left-CF6wxXU5.js} +1 -1
  36. package/dist/assets/dql-notebook/assets/{arrow-right-DpPWbwaD.js → arrow-right-C7G3M_S8.js} +1 -1
  37. package/dist/assets/dql-notebook/assets/{book-open-text-D7s5jo4X.js → book-open-text-DDBO8G7L.js} +1 -1
  38. package/dist/assets/dql-notebook/assets/chevron-left-DxajrX2v.js +6 -0
  39. package/dist/assets/dql-notebook/assets/{circle-x-Db3dXKg7.js → circle-x-HBvXjnT6.js} +1 -1
  40. package/dist/assets/dql-notebook/assets/clock-3-CJIdsPLt.js +6 -0
  41. package/dist/assets/dql-notebook/assets/dagre.esm-B6nvU4OB.js +1 -0
  42. package/dist/assets/dql-notebook/assets/{external-link-BxwXitO_.js → external-link-WciAU8Bu.js} +1 -1
  43. package/dist/assets/dql-notebook/assets/{grip-vertical-Dt4lkWRi.js → grip-vertical-DVu7ok6x.js} +1 -1
  44. package/dist/assets/dql-notebook/assets/index-B_kaoARS.css +1 -0
  45. package/dist/assets/dql-notebook/assets/{index-DKo-bwNw.js → index-D3wnucQC.js} +133 -128
  46. package/dist/assets/dql-notebook/assets/{link-2-Dfo2P6wi.js → link-2-C6uQx50X.js} +1 -1
  47. package/dist/assets/dql-notebook/assets/{list-tree-DiTmIWAL.js → list-tree-DapsIuQB.js} +1 -1
  48. package/dist/assets/dql-notebook/assets/{minimize-2-CMTAkPzL.js → minimize-2-BlVoipnT.js} +1 -1
  49. package/dist/assets/dql-notebook/assets/{panel-right-open-DwYr7FW4.js → panel-right-open-DcnKWxfx.js} +1 -1
  50. package/dist/assets/dql-notebook/assets/{play-BXhHYQ4x.js → play-CNuRuaAs.js} +1 -1
  51. package/dist/assets/dql-notebook/assets/{rotate-ccw-BNi6F8pl.js → rotate-ccw-B2p953Zh.js} +1 -1
  52. package/dist/assets/dql-notebook/assets/{semantic-fields-CNOGysAy.js → semantic-fields-2fWgkTJc.js} +1 -1
  53. package/dist/assets/dql-notebook/assets/{sliders-horizontal-Ec5MUMUW.js → sliders-horizontal-DwRTu1w4.js} +1 -1
  54. package/dist/assets/dql-notebook/assets/{star-B9leDkp_.js → star-Cbaj4F3i.js} +1 -1
  55. package/dist/assets/dql-notebook/assets/style-NlN9F6JE.js +23 -0
  56. package/dist/assets/dql-notebook/assets/{triangle-alert-BefTYCzx.js → triangle-alert-BPHvH1wA.js} +1 -1
  57. package/dist/assets/dql-notebook/assets/{upload-SPiOM2tQ.js → upload-DIqK0KE1.js} +1 -1
  58. package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-C4foXeiQ.js → usePersistedAgentThreadId-DpQXOEDw.js} +1 -1
  59. package/dist/assets/dql-notebook/assets/{user-round-comGmyw-.js → user-round-DukNupB_.js} +1 -1
  60. package/dist/assets/dql-notebook/assets/{wand-sparkles-BffR4dF8.js → wand-sparkles-CAwLX0b9.js} +1 -1
  61. package/dist/assets/dql-notebook/assets/{workflow-ChmPTEzH.js → workflow-bOT-NqdO.js} +1 -1
  62. package/dist/assets/dql-notebook/assets/{wrench-DovaG_ze.js → wrench-1m0_kcza.js} +1 -1
  63. package/dist/assets/dql-notebook/assets/{x-nRx91AgW.js → x-fp82BETM.js} +1 -1
  64. package/dist/assets/dql-notebook/index.html +2 -2
  65. package/dist/commands/agent-eval-cassette.d.ts +97 -6
  66. package/dist/commands/agent-eval-cassette.d.ts.map +1 -1
  67. package/dist/commands/agent-eval-cassette.js +165 -22
  68. package/dist/commands/agent-eval-cassette.js.map +1 -1
  69. package/dist/commands/agent-eval-runtime.d.ts +23 -2
  70. package/dist/commands/agent-eval-runtime.d.ts.map +1 -1
  71. package/dist/commands/agent-eval-runtime.js +19 -2
  72. package/dist/commands/agent-eval-runtime.js.map +1 -1
  73. package/dist/commands/agent-trace.d.ts +3 -0
  74. package/dist/commands/agent-trace.d.ts.map +1 -0
  75. package/dist/commands/agent-trace.js +169 -0
  76. package/dist/commands/agent-trace.js.map +1 -0
  77. package/dist/commands/agent.d.ts +37 -5
  78. package/dist/commands/agent.d.ts.map +1 -1
  79. package/dist/commands/agent.js +534 -84
  80. package/dist/commands/agent.js.map +1 -1
  81. package/dist/commands/compile.d.ts +13 -1
  82. package/dist/commands/compile.d.ts.map +1 -1
  83. package/dist/commands/compile.js +36 -4
  84. package/dist/commands/compile.js.map +1 -1
  85. package/dist/commands/notebook.d.ts +2 -0
  86. package/dist/commands/notebook.d.ts.map +1 -1
  87. package/dist/commands/notebook.js +4 -0
  88. package/dist/commands/notebook.js.map +1 -1
  89. package/dist/commands/sync.d.ts.map +1 -1
  90. package/dist/commands/sync.js +11 -3
  91. package/dist/commands/sync.js.map +1 -1
  92. package/dist/index.js +2 -0
  93. package/dist/index.js.map +1 -1
  94. package/dist/llm/providers/dql-agent-provider.d.ts +29 -2
  95. package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
  96. package/dist/llm/providers/dql-agent-provider.js +1008 -144
  97. package/dist/llm/providers/dql-agent-provider.js.map +1 -1
  98. package/dist/llm/types.d.ts +48 -1
  99. package/dist/llm/types.d.ts.map +1 -1
  100. package/dist/local-runtime.d.ts +271 -33
  101. package/dist/local-runtime.d.ts.map +1 -1
  102. package/dist/local-runtime.js +3658 -546
  103. package/dist/local-runtime.js.map +1 -1
  104. package/dist/package.json +10 -10
  105. package/dist/providers/oauth/claude-oauth.d.ts.map +1 -1
  106. package/dist/providers/oauth/claude-oauth.js +52 -18
  107. package/dist/providers/oauth/claude-oauth.js.map +1 -1
  108. package/dist/providers/oauth/codex-oauth.d.ts.map +1 -1
  109. package/dist/providers/oauth/codex-oauth.js +72 -44
  110. package/dist/providers/oauth/codex-oauth.js.map +1 -1
  111. package/dist/providers/subscription-cli.d.ts +20 -0
  112. package/dist/providers/subscription-cli.d.ts.map +1 -1
  113. package/dist/providers/subscription-cli.js +69 -11
  114. package/dist/providers/subscription-cli.js.map +1 -1
  115. package/package.json +10 -10
  116. package/dist/assets/dql-notebook/assets/AnalyticsHome-D5P6Ujwi.js +0 -6
  117. package/dist/assets/dql-notebook/assets/BusinessArtifactView-BIfNI-0S.js +0 -1
  118. package/dist/assets/dql-notebook/assets/LineageDAG-CSqcbDrE.js +0 -1
  119. package/dist/assets/dql-notebook/assets/LineageDetailView-DJZZjZu-.js +0 -1
  120. package/dist/assets/dql-notebook/assets/LineageDrawer-BcIipIc3.js +0 -1
  121. package/dist/assets/dql-notebook/assets/MiniLineageGraph-vH_MY_Ju.js +0 -1
  122. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel--oxmjlgr.js +0 -88
  123. package/dist/assets/dql-notebook/assets/dagre.esm-C7pppQ1a.js +0 -23
  124. package/dist/assets/dql-notebook/assets/index-B3shyZsg.css +0 -1
  125. /package/dist/assets/dql-notebook/assets/{dagre-BZV40eAE.css → style-BZV40eAE.css} +0 -0
@@ -21,17 +21,19 @@
21
21
  * dql agent feedback <up|down> --block <id> --question "..."
22
22
  * Records feedback into the KG. Used by clients without MCP access.
23
23
  */
24
- import { answerFromRuntimeRun, driveViaRuntime, evalRouteForRun } from './agent-eval-runtime.js';
25
- import { CassetteStore, cassetteDirFor, withCassette } from './agent-eval-cassette.js';
24
+ import { answerFromRuntimeRun, driveViaRuntime, projectRuntimeRun, } from './agent-eval-runtime.js';
25
+ import { CassetteStore, cassetteEvidenceSummary, cassetteDirFor, evalCassetteCanonicalizationV2, withCassette, } from './agent-eval-cassette.js';
26
26
  import { existsSync, readFileSync } from 'node:fs';
27
27
  import { join, resolve } from 'node:path';
28
28
  import { load as loadYaml } from 'js-yaml';
29
- import { KGStore, MemoryStore, defaultKgPath, defaultMemoryPath, reindexProject, loadSkills, pickProvider, answer, resolveDomainContextEnvelope, buildAnalysisQuestionPlan, buildLocalContextPack, coerceReasoningEffort, contextRetrievalBudgetForQuestion, deriveGeneratedDraftSlug, loadAgentSemanticLayer, recordQueryRun, recordRuntimeSchemaSnapshot, upsertGeneratedDqlArtifactDraft, upsertGeneratedDraft, validateSqlAgainstLocalContext, } from '@duckcodeailabs/dql-agent';
29
+ import { AskTraceSqliteStoreV1, KGStore, MemoryStore, defaultKgPath, defaultMemoryPath, reindexProject, loadSkills, pickProvider, answer, resolveDomainContextEnvelope, buildAnalysisQuestionPlan, buildLocalContextPack, classifyProviderFailure, coerceReasoningEffort, contextRetrievalBudgetForQuestion, deriveGeneratedDraftSlug, loadAgentSemanticLayer, recordQueryRun, recordRuntimeSchemaSnapshot, upsertGeneratedDqlArtifactDraft, upsertGeneratedDraft, validateSqlAgainstLocalContext, createAskTraceObserverV1, defaultAskTraceSqlitePath, } from '@duckcodeailabs/dql-agent';
30
+ import { createHash, randomUUID } from 'node:crypto';
30
31
  import { buildManifest, resolveDbtManifestPath } from '@duckcodeailabs/dql-core';
31
32
  import { findProjectRoot } from '../local-runtime.js';
32
33
  import { buildAnswerLoopTools, createGroundingContextExpander } from '../llm/answer-loop-tools.js';
33
34
  import { judgeAnswer } from './eval-judge.js';
34
35
  import { startProjectRuntime } from './notebook.js';
36
+ import { runAgentTrace } from './agent-trace.js';
35
37
  /**
36
38
  * Resolve the runtime the agent posts certified blocks / generated SQL to.
37
39
  *
@@ -55,7 +57,56 @@ async function resolveAgentRuntime(projectRoot, flags) {
55
57
  return { runtimeBase: base, close: async () => { } };
56
58
  }
57
59
  const handle = await startProjectRuntime(projectRoot, { preferredPort: 0 });
58
- return { runtimeBase: handle.url, close: handle.close };
60
+ return { runtimeBase: handle.url, close: handle.close, askTraceCapability: handle.askTraceCapability };
61
+ }
62
+ function isLoopbackRuntimeUrl(runtimeBase) {
63
+ try {
64
+ const hostname = new URL(runtimeBase).hostname;
65
+ return hostname === '127.0.0.1' || hostname === 'localhost' || hostname === '::1' || hostname === '[::1]';
66
+ }
67
+ catch {
68
+ return false;
69
+ }
70
+ }
71
+ /**
72
+ * An already-running local Notebook runtime mints a one-shot, loopback-only
73
+ * capability for this CLI request. Remote runtimes deliberately receive no
74
+ * capability and therefore cannot be relabelled as CLI by arbitrary text.
75
+ */
76
+ export async function requestLoopbackCliAskTraceCapability(runtimeBase) {
77
+ if (!isLoopbackRuntimeUrl(runtimeBase))
78
+ return undefined;
79
+ try {
80
+ const response = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/ask-traces/cli-capability`, {
81
+ headers: { Accept: 'application/json' },
82
+ });
83
+ if (!response.ok)
84
+ return undefined;
85
+ const payload = await response.json();
86
+ return typeof payload.capability === 'string'
87
+ && typeof payload.expiresAt === 'string'
88
+ && payload.scope === 'agent-runs'
89
+ ? payload.capability
90
+ : undefined;
91
+ }
92
+ catch {
93
+ // A pre-observability runtime remains usable. Its request stays browser
94
+ // attributed rather than fabricating a client-controlled CLI surface.
95
+ return undefined;
96
+ }
97
+ }
98
+ function runtimeProviderForCliFlag(value) {
99
+ if (typeof value !== 'string' || !value.trim())
100
+ return undefined;
101
+ switch (value.trim().toLowerCase()) {
102
+ case 'claude': return 'anthropic';
103
+ case 'openai': return 'openai';
104
+ case 'gemini': return 'gemini';
105
+ case 'ollama': return 'ollama';
106
+ // Preserve an invalid explicit value through the host request so the
107
+ // canonical preflight can return a typed `model_not_found` diagnostic.
108
+ default: return value.trim().toLowerCase();
109
+ }
59
110
  }
60
111
  async function fetchRuntimeSchemaContext(runtimeBase) {
61
112
  try {
@@ -142,6 +193,137 @@ function cliAnalysisDepth(flags) {
142
193
  const value = flags.analysisDepth?.trim().toLowerCase();
143
194
  return value === 'quick' || value === 'deep' ? value : undefined;
144
195
  }
196
+ /**
197
+ * Compatibility test adapter for a standalone provider boundary. Production
198
+ * `dql agent ask` no longer calls this: it uses the runtime AgentRun engine so
199
+ * the router, cascade, freeze, tools, SQL, and provider spans share one
200
+ * canonical trace. Keep this adapter truthful for lower-level provider tests
201
+ * without making it an alternate orchestration authority.
202
+ */
203
+ export function createDirectCliAskTraceProvider(provider, trace) {
204
+ let attemptIndex = 0;
205
+ let lastFailedSpanId;
206
+ const pending = new Map();
207
+ const keyFor = (event) => `${event.provider}:${event.operation}:${event.attemptIndex}`;
208
+ const attempt = (event, retryOfSpanId) => ({
209
+ version: 1,
210
+ phase: 'generation',
211
+ physicalAttemptIndex: ++attemptIndex,
212
+ providerFingerprint: `sha256:${createHash('sha256').update(event.provider).digest('hex')}`,
213
+ ...(event.model ? { modelFingerprint: `sha256:${createHash('sha256').update(event.model).digest('hex')}` } : {}),
214
+ ...(retryOfSpanId ? { retryOfSpanId } : {}),
215
+ admission: 'admitted',
216
+ provenance: 'live',
217
+ });
218
+ const finish = (entry, outcome, error) => {
219
+ if (!entry.spanId)
220
+ return;
221
+ const diagnostic = outcome === 'ok' ? undefined : classifyProviderFailure({
222
+ phase: 'generation',
223
+ code: error && typeof error === 'object' ? String(error.code ?? '') : undefined,
224
+ message: error instanceof Error ? error.message : String(error ?? ''),
225
+ providerFingerprint: entry.attempt.providerFingerprint,
226
+ modelFingerprint: entry.attempt.modelFingerprint,
227
+ });
228
+ const finalAttempt = outcome === 'ok'
229
+ ? entry.attempt
230
+ : {
231
+ ...entry.attempt,
232
+ ...(diagnostic?.httpStatusClass ? { httpStatusClass: diagnostic.httpStatusClass } : {}),
233
+ ...(diagnostic ? { retryable: diagnostic.retryable, safeAction: diagnostic.safeAction } : {}),
234
+ cause: outcome === 'cancelled' ? 'cancelled' : diagnostic?.cause ?? 'unknown',
235
+ };
236
+ trace.finishSpan(entry.spanId, {
237
+ outcome: outcome === 'ok' ? 'ok' : outcome === 'cancelled' ? 'cancelled' : 'error',
238
+ reasonCode: outcome === 'ok' ? 'completed' : outcome === 'cancelled' ? 'cancelled' : 'provider_failure',
239
+ payload: { kind: 'provider', attempt: finalAttempt },
240
+ });
241
+ if (outcome !== 'ok')
242
+ lastFailedSpanId = entry.spanId;
243
+ };
244
+ const observedOptions = (options = {}) => ({
245
+ ...options,
246
+ onProviderDispatch: (event) => {
247
+ const tracedAttempt = attempt(event, lastFailedSpanId);
248
+ const spanId = trace.startSpan({
249
+ name: 'provider.attempt',
250
+ stage: 'provider',
251
+ reasonCode: 'started',
252
+ payload: { kind: 'provider', attempt: tracedAttempt },
253
+ });
254
+ const key = keyFor(event);
255
+ pending.set(key, [...(pending.get(key) ?? []), { spanId, attempt: tracedAttempt }]);
256
+ return options.onProviderDispatch?.(event) ?? event.envelope;
257
+ },
258
+ onProviderDispatchComplete: (event) => {
259
+ const key = keyFor(event);
260
+ const entries = pending.get(key) ?? [];
261
+ const entry = entries[0];
262
+ if (entry && event.outcome === 'ok' && (event.settlement === 'transport' || event.settlement === 'process')) {
263
+ entry.attempt = {
264
+ ...entry.attempt,
265
+ ...(event.settlement === 'transport' ? { transportOutcome: 'ok' } : { processOutcome: 'ok' }),
266
+ };
267
+ }
268
+ else {
269
+ const closed = entries.shift();
270
+ if (entries.length > 0)
271
+ pending.set(key, entries);
272
+ else
273
+ pending.delete(key);
274
+ if (closed)
275
+ finish(closed, event.outcome, event.error ?? (typeof event.httpStatus === 'number' ? Object.assign(new Error(`HTTP ${event.httpStatus}`), { code: `HTTP_${event.httpStatus}` }) : undefined));
276
+ }
277
+ options.onProviderDispatchComplete?.(event);
278
+ },
279
+ onProviderDispatchRejected: (event) => {
280
+ const denied = trace.startSpan({
281
+ name: 'provider.attempt',
282
+ stage: 'provider',
283
+ reasonCode: 'provider_failure',
284
+ payload: {
285
+ kind: 'provider',
286
+ attempt: {
287
+ ...attempt(event, lastFailedSpanId),
288
+ admission: 'denied',
289
+ },
290
+ },
291
+ });
292
+ trace.finishSpan(denied, { outcome: 'denied', reasonCode: 'provider_failure' });
293
+ lastFailedSpanId = denied;
294
+ options.onProviderDispatchRejected?.(event);
295
+ },
296
+ });
297
+ const invoke = async (options, call) => {
298
+ try {
299
+ const result = await call(observedOptions(options));
300
+ for (const entries of pending.values())
301
+ for (const entry of entries)
302
+ finish(entry, 'ok');
303
+ pending.clear();
304
+ return result;
305
+ }
306
+ catch (error) {
307
+ const outcome = options?.signal?.aborted ? 'cancelled' : 'error';
308
+ for (const entries of pending.values())
309
+ for (const entry of entries)
310
+ finish(entry, outcome, error);
311
+ pending.clear();
312
+ throw error;
313
+ }
314
+ };
315
+ return {
316
+ name: provider.name,
317
+ available: () => provider.available(),
318
+ generate: (messages, options) => invoke(options, (observed) => provider.generate(messages, observed)),
319
+ ...(provider.generateWithTools ? {
320
+ generateWithTools: (messages, tools, options) => invoke(options, (observed) => provider.generateWithTools(messages, tools, observed)),
321
+ } : {}),
322
+ ...(provider.generateStream ? {
323
+ generateStream: (messages, options, onDelta) => invoke(options, (observed) => provider.generateStream(messages, observed, onDelta)),
324
+ } : {}),
325
+ };
326
+ }
145
327
  /** A DQL runtime answers `/api/connections` with a connector/connection payload. */
146
328
  async function isDqlRuntime(base) {
147
329
  try {
@@ -161,6 +343,8 @@ export async function runAgent(sub, rest, flags) {
161
343
  return runAsk(rest, flags);
162
344
  case 'threads':
163
345
  return runThreads(flags);
346
+ case 'trace':
347
+ return runAgentTrace(rest, flags);
164
348
  case 'reindex':
165
349
  return runReindex(rest, flags);
166
350
  case 'feedback':
@@ -168,18 +352,41 @@ export async function runAgent(sub, rest, flags) {
168
352
  case 'eval':
169
353
  return runEval(rest, flags);
170
354
  default:
171
- throw new Error('Usage: dql agent <ask|threads|reindex|feedback|eval> [args]\n' +
355
+ throw new Error('Usage: dql agent <ask|threads|trace|reindex|feedback|eval> [args]\n' +
172
356
  ' dql agent ask "<question>" [--provider claude|openai|gemini|ollama] [--user <id>] [--domain <d>] [--purpose <approved-purpose>] [--thread <id>]\n' +
173
357
  ' dql agent threads [--runtime-url <url>]\n' +
358
+ ' dql agent trace list|show|export|validate|replay|compare\n' +
174
359
  ' dql agent reindex [path]\n' +
175
360
  ' dql agent feedback up|down --block <id> --question "..."\n' +
176
361
  ' dql agent eval agent-evals.yml [--provider claude|openai|gemini|ollama] [--execute] [--save]');
177
362
  }
178
363
  }
364
+ /**
365
+ * Canonical CLI Ask entrypoint. Direct and threaded CLI questions use the
366
+ * exact local AgentRun engine as the browser, so trace evidence follows the
367
+ * router's candidate/cascade/freeze/execution authority rather than a legacy
368
+ * answer-loop approximation.
369
+ */
179
370
  async function runAsk(rest, flags) {
180
371
  const question = rest.join(' ').trim();
181
372
  if (!question)
182
373
  throw new Error('Usage: dql agent ask "<question>"');
374
+ const threadId = flags.thread;
375
+ return threadId
376
+ ? runThreadAsk(question, threadId, flags)
377
+ : runCanonicalCliAsk(question, undefined, flags);
378
+ }
379
+ /**
380
+ * Compatibility entrypoint retained for downstream imports during the CLI
381
+ * transition. It deliberately delegates before allocating any legacy state,
382
+ * so even an old internal caller gets the canonical AgentRun trace rather than
383
+ * an incomplete synthetic receipt.
384
+ */
385
+ async function runLegacyDirectAsk(rest, flags) {
386
+ const question = rest.join(' ').trim();
387
+ if (!question)
388
+ throw new Error('Usage: dql agent ask "<question>"');
389
+ return runCanonicalCliAsk(question, flags.thread, flags);
183
390
  // Thread-scoped ask: hand the question to the runtime's agent-run engine with
184
391
  // the thread id, so the SERVER injects prior turns and persists this run as a
185
392
  // new turn (the same conversation store the notebook UI uses).
@@ -187,8 +394,31 @@ async function runAsk(rest, flags) {
187
394
  if (threadId)
188
395
  return runThreadAsk(question, threadId, flags);
189
396
  const projectRoot = findProjectRoot(process.cwd());
397
+ const traceStore = new AskTraceSqliteStoreV1({ path: defaultAskTraceSqlitePath(projectRoot) });
398
+ const trace = createAskTraceObserverV1({
399
+ store: traceStore,
400
+ runId: `cli-${randomUUID()}`,
401
+ surface: 'cli',
402
+ mode: 'ask',
403
+ questionFingerprint: `sha256:${createHash('sha256').update(question).digest('hex')}`,
404
+ });
405
+ const classifySpan = trace.startSpan({
406
+ name: 'request.classify',
407
+ stage: 'request',
408
+ reasonCode: 'started',
409
+ payload: { kind: 'stage', route: 'direct_cli_legacy' },
410
+ });
190
411
  const kgPath = defaultKgPath(projectRoot);
191
- await reindexProject(projectRoot, { kgPath });
412
+ try {
413
+ await reindexProject(projectRoot, { kgPath });
414
+ trace.finishSpan(classifySpan, { outcome: 'ok', reasonCode: 'completed' });
415
+ }
416
+ catch (error) {
417
+ trace.finishSpan(classifySpan, { outcome: 'error', reasonCode: 'unknown' });
418
+ trace.finalize({ status: 'failed' });
419
+ traceStore.close();
420
+ throw error;
421
+ }
192
422
  const providerName = flags.provider;
193
423
  const userId = flags.user;
194
424
  const domain = flags.domain;
@@ -196,12 +426,68 @@ async function runAsk(rest, flags) {
196
426
  const format = flags.format;
197
427
  const reasoningEffort = cliReasoningEffort(flags);
198
428
  const requestedDepth = cliAnalysisDepth(flags);
199
- const provider = await pickProvider(providerName);
429
+ let provider;
430
+ try {
431
+ provider = await pickProvider(providerName);
432
+ }
433
+ catch (error) {
434
+ trace.finalize({ status: 'failed' });
435
+ traceStore.close();
436
+ throw error;
437
+ }
200
438
  const kg = new KGStore(kgPath);
201
439
  const memory = new MemoryStore(defaultMemoryPath(projectRoot));
202
440
  const { skills } = loadSkills(projectRoot);
203
441
  let closeRuntime;
204
442
  try {
443
+ const preflight = trace.startSpan({
444
+ name: 'provider.preflight',
445
+ stage: 'provider',
446
+ reasonCode: 'started',
447
+ payload: {
448
+ kind: 'provider',
449
+ attempt: {
450
+ version: 1,
451
+ phase: 'preflight',
452
+ physicalAttemptIndex: 0,
453
+ providerFingerprint: `sha256:${createHash('sha256').update(provider.name).digest('hex')}`,
454
+ readiness: 'unknown',
455
+ admission: 'unknown',
456
+ provenance: 'live',
457
+ },
458
+ },
459
+ });
460
+ let providerReady = false;
461
+ try {
462
+ providerReady = await provider.available();
463
+ }
464
+ catch {
465
+ providerReady = false;
466
+ }
467
+ trace.finishSpan(preflight, {
468
+ outcome: providerReady ? 'ok' : 'unavailable',
469
+ reasonCode: providerReady ? 'completed' : 'provider_preflight',
470
+ payload: {
471
+ kind: 'provider',
472
+ attempt: {
473
+ version: 1,
474
+ phase: 'preflight',
475
+ physicalAttemptIndex: 0,
476
+ providerFingerprint: `sha256:${createHash('sha256').update(provider.name).digest('hex')}`,
477
+ readiness: providerReady ? 'ready' : 'unavailable',
478
+ admission: 'unknown',
479
+ cause: providerReady ? undefined : 'authentication',
480
+ safeAction: providerReady ? undefined : 'fix_provider_configuration',
481
+ provenance: 'live',
482
+ },
483
+ },
484
+ });
485
+ if (!providerReady) {
486
+ throw Object.assign(new Error('The selected AI provider is not ready. Configure or sign in to the provider and retry.'), {
487
+ code: 'AUTHENTICATION_FAILED',
488
+ });
489
+ }
490
+ const tracedProvider = createDirectCliAskTraceProvider(provider, trace);
205
491
  const memoryContext = memory.search({
206
492
  query: question,
207
493
  scopes: ['project', 'user', 'artifact'],
@@ -220,6 +506,12 @@ async function runAsk(rest, flags) {
220
506
  closeRuntime = close;
221
507
  const schemaContext = await fetchRuntimeSchemaContext(runtimeBase);
222
508
  recordCliRuntimeSchemaSnapshot(projectRoot, schemaContext, 'direct CLI runtime schema');
509
+ const retrievalSpan = trace.startSpan({
510
+ name: 'retrieval',
511
+ stage: 'retrieval',
512
+ reasonCode: 'started',
513
+ payload: { kind: 'retrieval', candidateCount: 0 },
514
+ });
223
515
  const contextPack = await buildLocalContextPack(projectRoot, {
224
516
  question,
225
517
  surface: 'cli',
@@ -233,10 +525,15 @@ async function runAsk(rest, flags) {
233
525
  }
234
526
  : undefined,
235
527
  }).catch(() => undefined);
528
+ trace.finishSpan(retrievalSpan, {
529
+ outcome: 'ok',
530
+ reasonCode: 'completed',
531
+ payload: { kind: 'retrieval', candidateCount: contextPack?.objects.length ?? 0 },
532
+ });
236
533
  const answerLoopTools = buildAnswerLoopTools(projectRoot);
237
534
  const result = await answer({
238
535
  question,
239
- provider,
536
+ provider: tracedProvider,
240
537
  kg,
241
538
  manifest,
242
539
  skills,
@@ -365,6 +662,7 @@ async function runAsk(rest, flags) {
365
662
  });
366
663
  },
367
664
  });
665
+ trace.finalize({ status: 'completed' });
368
666
  if (format === 'json') {
369
667
  console.log(JSON.stringify(result, null, 2));
370
668
  return;
@@ -379,17 +677,25 @@ async function runAsk(rest, flags) {
379
677
  : '';
380
678
  const footer = result.provenanceFooter ? `\n\n— ${result.provenanceFooter}` : '';
381
679
  console.log(`${badge}\n\n${result.text}${footer}${cite}`);
382
- if (result.result) {
383
- console.log(`\nRows: ${result.result.rowCount}`);
384
- console.log(JSON.stringify(result.result.rows.slice(0, 5), null, 2));
680
+ const resultPayload = result.result;
681
+ if (resultPayload) {
682
+ console.log(`\nRows: ${resultPayload.rowCount}`);
683
+ console.log(JSON.stringify(resultPayload.rows.slice(0, 5), null, 2));
385
684
  }
386
685
  printDqlArtifactPreview(result);
387
686
  }
687
+ catch (error) {
688
+ const cancelled = error instanceof Error && error.name === 'AbortError';
689
+ trace.finalize({ status: cancelled ? 'cancelled' : 'failed' });
690
+ throw error;
691
+ }
388
692
  finally {
389
693
  kg.close();
390
694
  memory.close();
391
- if (closeRuntime)
392
- await closeRuntime();
695
+ const close = closeRuntime;
696
+ if (close)
697
+ await close();
698
+ traceStore.close();
393
699
  }
394
700
  }
395
701
  function printDqlArtifactPreview(result) {
@@ -410,56 +716,103 @@ function printDqlArtifactPreview(result) {
410
716
  console.log(`Promote: ${result.promoteCommand}`);
411
717
  }
412
718
  }
719
+ function printCanonicalCliAskRun(run, threadId) {
720
+ const badge = run.trustState === 'certified'
721
+ ? '✓ Certified'
722
+ : run.trustState === 'grounded'
723
+ ? '✓ Verified (grounded)'
724
+ : run.trustState === 'review_required'
725
+ ? '! AI-generated · review required'
726
+ : run.trustState === 'blocked'
727
+ ? '✕ Blocked'
728
+ : '· Reply';
729
+ console.log(`${badge}\n\n${(run.answer ?? run.summary ?? '').trim()}`);
730
+ if (threadId)
731
+ console.log(`\nThread: ${threadId}`);
732
+ const trace = run.traceReference;
733
+ if (trace?.traceId) {
734
+ console.log(`\nTrace: ${trace.traceId} (${trace.recordingStatus ?? 'recording'})`);
735
+ }
736
+ else {
737
+ console.log('\nTrace: unavailable');
738
+ }
739
+ }
413
740
  /**
414
- * `dql agent ask --thread <id>` POST the question to the runtime's
415
- * `/api/agent-runs` with the threadId in the body. The server injects the
416
- * thread's prior turns into the conversation context and records the completed
417
- * run as the next turn, so follow-ups resolve "those"/"that product" correctly
418
- * across CLI invocations (and across the notebook UI, which shares the store).
741
+ * Submit every user-visible CLI Ask through the runtime AgentRun endpoint.
742
+ * The runtime owns run IDs, conversation hydration, candidate provenance,
743
+ * cascade/freeze authority, physical tool/SQL execution, and final trace
744
+ * reference. The CLI only supplies user intent and an optional host-minted
745
+ * surface capability.
419
746
  */
420
- async function runThreadAsk(question, threadId, flags) {
747
+ export async function runCanonicalCliAsk(question, threadId, flags) {
421
748
  const projectRoot = findProjectRoot(process.cwd());
422
749
  const format = flags.format;
423
- const { runtimeBase, close } = await resolveAgentRuntime(projectRoot, flags);
750
+ const { runtimeBase, close, askTraceCapability: embeddedCapability } = await resolveAgentRuntime(projectRoot, flags);
424
751
  try {
752
+ const askTraceCapability = embeddedCapability ?? await requestLoopbackCliAskTraceCapability(runtimeBase);
425
753
  const reasoningEffort = cliReasoningEffort(flags);
426
754
  const analysisDepth = cliAnalysisDepth(flags);
755
+ const domain = flags.domain;
756
+ const purpose = flags.purpose || undefined;
757
+ const userId = flags.user;
758
+ const provider = runtimeProviderForCliFlag(flags.provider);
759
+ // `surface` is deliberately absent from public JSON. The runtime assigns
760
+ // trace surface only after consuming its own loopback capability; this
761
+ // context carries user intent, never attribution authority.
762
+ const workspaceContext = {
763
+ ...(domain ? { domain } : {}),
764
+ ...(purpose ? { purpose } : {}),
765
+ ...(userId ? { userId } : {}),
766
+ ...(provider ? { provider } : {}),
767
+ };
768
+ const hasWorkspaceContext = Object.keys(workspaceContext).length > 0;
427
769
  const response = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/agent-runs`, {
428
770
  method: 'POST',
429
- headers: { 'Content-Type': 'application/json' },
771
+ headers: {
772
+ 'Content-Type': 'application/json',
773
+ ...(askTraceCapability ? { 'X-DQL-Ask-Trace-Capability': askTraceCapability } : {}),
774
+ },
430
775
  body: JSON.stringify({
431
776
  question,
432
- threadId,
777
+ ...(threadId ? { threadId } : {}),
433
778
  ...(reasoningEffort ? { reasoningEffort } : {}),
434
779
  ...(analysisDepth ? { analysisDepth } : {}),
780
+ ...(hasWorkspaceContext ? { workspaceContext } : {}),
435
781
  }),
436
782
  });
437
783
  if (!response.ok)
438
784
  throw new Error(`Runtime returned ${response.status}: ${await response.text()}`);
439
785
  const payload = (await response.json());
440
- if (!payload.run)
441
- throw new Error('Runtime did not return an agent run.');
786
+ if (!payload.run?.id)
787
+ throw new Error('Runtime did not return a canonical agent run.');
788
+ const run = payload.run;
442
789
  if (format === 'json') {
443
- console.log(JSON.stringify(payload.run, null, 2));
790
+ // Preserve the established top-level run shape while adding compact,
791
+ // content-free trace discoverability for scripts and support bundles.
792
+ console.log(JSON.stringify({
793
+ ...run,
794
+ runId: run.id,
795
+ ...(run.traceReference?.traceId ? { traceId: run.traceReference.traceId } : {}),
796
+ traceRecordingStatus: run.traceReference?.recordingStatus ?? 'unavailable',
797
+ }, null, 2));
444
798
  return;
445
799
  }
446
- const run = payload.run;
447
- const badge = run.trustState === 'certified'
448
- ? '✓ Certified'
449
- : run.trustState === 'grounded'
450
- ? '✓ Verified (grounded)'
451
- : run.trustState === 'review_required'
452
- ? '! AI-generated · review required'
453
- : run.trustState === 'blocked'
454
- ? '✕ Blocked'
455
- : '· Reply';
456
- console.log(`${badge}\n\n${(run.answer ?? run.summary ?? '').trim()}`);
457
- console.log(`\nThread: ${threadId}`);
800
+ printCanonicalCliAskRun(run, threadId);
458
801
  }
459
802
  finally {
460
803
  await close();
461
804
  }
462
805
  }
806
+ /**
807
+ * `dql agent ask --thread <id>` — POST the question to the runtime's
808
+ * `/api/agent-runs` with the threadId in the body. The server injects the
809
+ * thread's prior turns into the conversation context and records the completed
810
+ * run as the next turn, so follow-ups resolve "those"/"that product" correctly
811
+ * across CLI invocations (and across the notebook UI, which shares the store).
812
+ */
813
+ async function runThreadAsk(question, threadId, flags) {
814
+ return runCanonicalCliAsk(question, threadId, flags);
815
+ }
463
816
  /** `dql agent threads` — list server-persisted conversation threads. */
464
817
  async function runThreads(flags) {
465
818
  const projectRoot = findProjectRoot(process.cwd());
@@ -559,7 +912,7 @@ async function runEval(rest, flags) {
559
912
  // started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
560
913
  const cassetteMode = flags.cassette;
561
914
  const provider = cassetteMode === 'record' || cassetteMode === 'replay'
562
- ? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode)
915
+ ? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode, evalCassetteCanonicalizationV2(projectRoot))
563
916
  : rawProvider;
564
917
  const reasoningEffort = cliReasoningEffort(flags);
565
918
  const requestedDepth = cliAnalysisDepth(flags);
@@ -633,6 +986,11 @@ async function runEval(rest, flags) {
633
986
  const runtimeRun = via === 'runtime'
634
987
  ? await driveViaRuntime({ runtimeBase, question: testCase.question })
635
988
  : undefined;
989
+ // Runtime mode is scored from the persisted AgentRun. The transport
990
+ // adapter intentionally has no AgentAnswer.contextPack, so borrowing the
991
+ // local preflight pack here would fabricate retrieval/route evidence for a
992
+ // different execution path.
993
+ const runtimeProjection = runtimeRun ? projectRuntimeRun(runtimeRun) : undefined;
636
994
  const result = runtimeRun
637
995
  ? answerFromRuntimeRun(runtimeRun)
638
996
  : await answer({
@@ -713,7 +1071,7 @@ async function runEval(rest, flags) {
713
1071
  });
714
1072
  },
715
1073
  });
716
- const evaluation = evaluateCase(testCase, result);
1074
+ const evaluation = evaluateCase(testCase, result, runtimeProjection);
717
1075
  const durationMs = Date.now() - startedAt;
718
1076
  const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
719
1077
  const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
@@ -740,7 +1098,7 @@ async function runEval(rest, flags) {
740
1098
  executionMatched: evaluation.executionMatched,
741
1099
  ...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
742
1100
  kind: result.kind,
743
- route: runtimeRun ? evalRouteForRun(runtimeRun.route) : result.contextPack?.routeDecision.route,
1101
+ route: runtimeProjection?.route ?? result.contextPack?.routeDecision.route,
744
1102
  // Only the runtime driver can see the router's clarification options.
745
1103
  // In-process runs leave this undefined, so a clarify there scores as a
746
1104
  // dead end — the conservative reading, and another reason `--via runtime`
@@ -752,12 +1110,13 @@ async function runEval(rest, flags) {
752
1110
  && Boolean(runtimeRun.answer?.trim()),
753
1111
  meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
754
1112
  } : {}),
755
- intent: result.contextPack?.routeDecision.intent,
1113
+ ...(runtimeProjection?.observability ? { observability: runtimeProjection.observability } : {}),
1114
+ intent: runtimeRun?.routeDecision?.category ?? result.contextPack?.routeDecision.intent,
756
1115
  reviewStatus: result.reviewStatus,
757
- contextObjects: result.contextPack?.objects.length ?? 0,
1116
+ contextObjects: runtimeProjection?.retrievalCandidateCount ?? result.contextPack?.objects.length,
758
1117
  followUp: Boolean(testCase.followUp),
759
1118
  draftSaved,
760
- toolCalls: result.evidence?.toolCalls?.length ?? 0,
1119
+ toolCalls: runtimeProjection?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0,
761
1120
  expected: testCase.expected,
762
1121
  validationCode: evaluation.validationCode,
763
1122
  trace: buildEvalTrace({
@@ -766,6 +1125,7 @@ async function runEval(rest, flags) {
766
1125
  evaluation,
767
1126
  durationMs,
768
1127
  draftSaved,
1128
+ runtime: runtimeProjection,
769
1129
  }),
770
1130
  });
771
1131
  }
@@ -776,6 +1136,18 @@ async function runEval(rest, flags) {
776
1136
  }
777
1137
  const passed = results.filter((r) => r.passed).length;
778
1138
  const metrics = computeEvalMetrics(results);
1139
+ // Runtime evals execute in a separate host, so its cassette directory is
1140
+ // supplied explicitly by the eval workflow for reporting. The summary does
1141
+ // not inspect prompts or provider credentials; it only classifies the
1142
+ // checked-in response provenance.
1143
+ const cassetteDirectory = via === 'runtime'
1144
+ ? process.env.DQL_EVAL_CASSETTE_DIR
1145
+ : cassetteMode === 'record' || cassetteMode === 'replay'
1146
+ ? cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')
1147
+ : undefined;
1148
+ const cassetteEvidence = cassetteDirectory
1149
+ ? cassetteEvidenceSummary(new CassetteStore(cassetteDirectory))
1150
+ : undefined;
779
1151
  const thresholds = {
780
1152
  minToolRequirement: flags.minToolRequirement ?? null,
781
1153
  minExecutionMatch: flags.minExecutionMatch ?? null,
@@ -788,7 +1160,15 @@ async function runEval(rest, flags) {
788
1160
  const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
789
1161
  const ok = passed === results.length && thresholdsPassed;
790
1162
  if (flags.format === 'json') {
791
- console.log(JSON.stringify({ ok, passed, total: results.length, thresholds, metrics, results }, null, 2));
1163
+ console.log(JSON.stringify({
1164
+ ok,
1165
+ passed,
1166
+ total: results.length,
1167
+ thresholds,
1168
+ metrics,
1169
+ ...(cassetteEvidence ? { cassetteEvidence } : {}),
1170
+ results,
1171
+ }, null, 2));
792
1172
  if (!ok)
793
1173
  process.exitCode = 1;
794
1174
  return;
@@ -802,6 +1182,14 @@ async function runEval(rest, flags) {
802
1182
  console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
803
1183
  console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
804
1184
  console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
1185
+ if (cassetteEvidence) {
1186
+ console.log(`Cassette replay entries: ${cassetteEvidence.totalEntries} `
1187
+ + `(${cassetteEvidence.migratedLegacyDeterministicFixtureEntries} migrated legacy deterministic fixture, `
1188
+ + `${cassetteEvidence.syntheticDeterministicOrchestrationFixtureEntries} synthetic deterministic orchestration fixture).`);
1189
+ console.log(cassetteEvidence.realProviderQualityEligible
1190
+ ? 'Real-provider quality evidence: eligible.'
1191
+ : `Real-provider quality evidence: excluded (${cassetteEvidence.realProviderQualityExclusionReasons.join(', ')}).`);
1192
+ }
805
1193
  console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
806
1194
  console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
807
1195
  if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
@@ -851,7 +1239,7 @@ function previewGeneratedDraftPath(projectRoot, domain, slug) {
851
1239
  }
852
1240
  return `blocks/_drafts/${slug}.dql`;
853
1241
  }
854
- function evaluateCase(testCase, result) {
1242
+ function evaluateCase(testCase, result, runtime) {
855
1243
  const expected = testCase.expected;
856
1244
  if (!expected)
857
1245
  return { failures: [] };
@@ -892,12 +1280,30 @@ function evaluateCase(testCase, result) {
892
1280
  failures.push(`certification expected ${expected.certification}, got ${result.certification}`);
893
1281
  if (expected.reviewStatus && result.reviewStatus !== expected.reviewStatus)
894
1282
  failures.push(`reviewStatus expected ${expected.reviewStatus}, got ${result.reviewStatus}`);
895
- if (expected.route && result.contextPack?.routeDecision.route !== expected.route)
896
- failures.push(`route expected ${expected.route}, got ${result.contextPack?.routeDecision.route ?? 'none'}`);
1283
+ const observedRoute = runtime?.route ?? result.contextPack?.routeDecision.route;
1284
+ if (expected.route && observedRoute !== expected.route)
1285
+ failures.push(`route expected ${expected.route}, got ${observedRoute ?? 'none'}`);
897
1286
  if (expected.intent && result.contextPack?.routeDecision.intent !== expected.intent)
898
1287
  failures.push(`intent expected ${expected.intent}, got ${result.contextPack?.routeDecision.intent ?? 'none'}`);
899
- if (expected.missingContextKind && !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind))
1288
+ // `modeling_gap` is a broad terminal kind. A relationship expectation is
1289
+ // satisfied only by the router's persisted relationship-specific witness;
1290
+ // otherwise a missing metric/dimension tuple would be misreported as a
1291
+ // relationship repair opportunity.
1292
+ const runtimeReportsMissingContext = expected.missingContextKind === 'relationship'
1293
+ ? runtime?.terminalOutcome?.gap?.code === 'MISSING_RELATIONSHIP'
1294
+ : runtime?.terminalOutcome?.kind === 'modeling_gap'
1295
+ && expected.missingContextKind === 'modeling_gap';
1296
+ if (expected.missingContextKind
1297
+ && !runtimeReportsMissingContext
1298
+ && !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind)) {
900
1299
  failures.push(`missing context kind ${expected.missingContextKind} was not reported`);
1300
+ }
1301
+ if (expected.terminalOutcomeKind && runtime?.terminalOutcome?.kind !== expected.terminalOutcomeKind) {
1302
+ failures.push(`terminal outcome expected ${expected.terminalOutcomeKind}, got ${runtime?.terminalOutcome?.kind ?? 'none'}`);
1303
+ }
1304
+ if (expected.traceRecordingStatus && runtime?.observability?.recordingStatus !== expected.traceRecordingStatus) {
1305
+ failures.push(`trace recording status expected ${expected.traceRecordingStatus}, got ${runtime?.observability?.recordingStatus ?? 'unavailable'}`);
1306
+ }
901
1307
  for (const token of stringList(expected.sqlContains)) {
902
1308
  if (!result.proposedSql?.toLowerCase().includes(token.toLowerCase()))
903
1309
  failures.push(`SQL did not contain "${token}"`);
@@ -918,7 +1324,7 @@ function evaluateCase(testCase, result) {
918
1324
  failures.push(`draftSaved expected ${expected.draftSaved}, got ${saved}`);
919
1325
  }
920
1326
  if (typeof expected.minToolCalls === 'number') {
921
- const actualToolCalls = result.evidence?.toolCalls?.length ?? 0;
1327
+ const actualToolCalls = runtime?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0;
922
1328
  if (actualToolCalls < expected.minToolCalls) {
923
1329
  failures.push(`toolCalls expected at least ${expected.minToolCalls}, got ${actualToolCalls}`);
924
1330
  }
@@ -957,7 +1363,7 @@ export function evalCaseIsAnswerable(expected) {
957
1363
  return false;
958
1364
  if (expected.sourceTier === 'no_answer')
959
1365
  return false;
960
- if (expected.route === 'clarify')
1366
+ if (expected.route === 'clarify' || expected.route === 'blocked')
961
1367
  return false;
962
1368
  if (Object.keys(expected).length === 0)
963
1369
  return undefined;
@@ -1084,7 +1490,12 @@ function computeEvalMetrics(results) {
1084
1490
  draft_saved_count: results.filter((result) => result.draftSaved).length,
1085
1491
  tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
1086
1492
  avg_tool_calls: average(toolCallCounts),
1087
- avg_context_objects: average(results.map((result) => result.contextObjects)),
1493
+ // Runtime runs only report this when the persisted router recorded an
1494
+ // explicit retrieval count. Treat absent evidence as unknown, not as an
1495
+ // invented empty context pack.
1496
+ avg_context_objects: average(results
1497
+ .map((result) => result.contextObjects)
1498
+ .filter((count) => typeof count === 'number')),
1088
1499
  avg_execution_ms: executionTimes.length ? average(executionTimes) : null,
1089
1500
  };
1090
1501
  }
@@ -1106,12 +1517,13 @@ function agentEvalThresholdsPass(metrics, thresholds) {
1106
1517
  || metrics.wrong_certified_count <= thresholds.maxWrongCertified);
1107
1518
  }
1108
1519
  function buildEvalTrace(input) {
1109
- const { testCase, result, evaluation, durationMs, draftSaved } = input;
1520
+ const { testCase, result, evaluation, durationMs, draftSaved, runtime } = input;
1110
1521
  const routeDecision = result.contextPack?.routeDecision;
1111
1522
  const selectedRelations = result.contextPack?.retrievalDiagnostics.selectedRelations ?? [];
1112
1523
  const allowedRelations = result.contextPack?.allowedSqlContext?.relations ?? [];
1113
1524
  const followUp = testCase.followUp;
1114
1525
  const toolCalls = result.evidence?.toolCalls ?? [];
1526
+ const observedToolCallCount = runtime?.toolCallCount ?? toolCalls.length;
1115
1527
  const routeEvidence = result.evidence?.route ?? [];
1116
1528
  const executionStatus = result.executionError
1117
1529
  ? 'failed'
@@ -1127,33 +1539,44 @@ function buildEvalTrace(input) {
1127
1539
  const rowsExpected = testCase.expected?.rows !== undefined;
1128
1540
  const expectedMinToolCalls = testCase.expected?.minToolCalls;
1129
1541
  const toolStatus = typeof expectedMinToolCalls === 'number'
1130
- ? toolCalls.length >= expectedMinToolCalls ? 'passed' : 'failed'
1131
- : toolCalls.length > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
1542
+ ? observedToolCallCount >= expectedMinToolCalls ? 'passed' : 'failed'
1543
+ : observedToolCallCount > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
1132
1544
  const toolMessage = typeof expectedMinToolCalls === 'number'
1133
- ? toolCalls.length >= expectedMinToolCalls
1134
- ? `Observed ${toolCalls.length} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
1135
- : `Observed ${toolCalls.length} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
1136
- : toolCalls.length > 0
1137
- ? `Observed ${toolCalls.length} provider tool call(s).`
1545
+ ? observedToolCallCount >= expectedMinToolCalls
1546
+ ? `Observed ${observedToolCallCount} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
1547
+ : `Observed ${observedToolCallCount} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
1548
+ : observedToolCallCount > 0
1549
+ ? `Observed ${observedToolCallCount} provider tool call(s).`
1138
1550
  : routeEvidence.length > 0
1139
1551
  ? `Captured ${routeEvidence.length} deterministic route evidence step(s).`
1140
1552
  : 'No provider tool calls were observed for this answer.';
1141
1553
  return [
1142
1554
  {
1143
1555
  stage: 'context',
1144
- status: result.contextPack ? 'passed' : 'not_run',
1145
- message: result.contextPack
1146
- ? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
1147
- : 'No context pack was attached to the answer.',
1148
- payload: result.contextPack
1556
+ status: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
1557
+ ? 'passed'
1558
+ : result.contextPack ? 'passed' : 'not_run',
1559
+ message: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
1560
+ ? `Persisted route evidence recorded ${runtime.retrievalCandidateCount ?? 'an unspecified number of'} retrieved candidate(s).`
1561
+ : result.contextPack
1562
+ ? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
1563
+ : 'No context pack was attached to the answer.',
1564
+ payload: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
1149
1565
  ? {
1150
- contextPackId: result.contextPack.id,
1151
- selectedObjectCount: result.contextPack.objects.length,
1152
- allowedRelationCount: allowedRelations.length,
1153
- selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
1154
- missingContext: result.contextPack.missingContext,
1566
+ evidenceSource: 'persisted_agent_run',
1567
+ retrievalCandidateCount: runtime.retrievalCandidateCount,
1568
+ sourceCoverage: runtime.sourceCoverage,
1569
+ terminalOutcome: runtime.terminalOutcome,
1155
1570
  }
1156
- : undefined,
1571
+ : result.contextPack
1572
+ ? {
1573
+ contextPackId: result.contextPack.id,
1574
+ selectedObjectCount: result.contextPack.objects.length,
1575
+ allowedRelationCount: allowedRelations.length,
1576
+ selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
1577
+ missingContext: result.contextPack.missingContext,
1578
+ }
1579
+ : undefined,
1157
1580
  },
1158
1581
  {
1159
1582
  stage: 'rewrite',
@@ -1165,29 +1588,40 @@ function buildEvalTrace(input) {
1165
1588
  },
1166
1589
  {
1167
1590
  stage: 'lane',
1168
- status: routeDecision ? 'passed' : 'not_run',
1169
- message: routeDecision
1170
- ? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
1171
- : 'No lane decision was attached to the answer.',
1172
- payload: routeDecision
1591
+ status: runtime ? 'passed' : routeDecision ? 'passed' : 'not_run',
1592
+ message: runtime
1593
+ ? `Persisted engine route ${runtime.runRoute}${runtime.route ? ` evaluated as ${runtime.route}` : ''}.`
1594
+ : routeDecision
1595
+ ? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
1596
+ : 'No lane decision was attached to the answer.',
1597
+ payload: runtime
1173
1598
  ? {
1174
- route: routeDecision.route,
1175
- intent: routeDecision.intent,
1176
- reason: routeDecision.reason,
1177
- trustLabel: routeDecision.trustLabel,
1178
- reviewStatus: routeDecision.reviewStatus,
1179
- exactObjectKey: routeDecision.exactObjectKey,
1599
+ engineRoute: runtime.runRoute,
1600
+ evalRoute: runtime.route,
1601
+ status: runtime.status,
1602
+ trustState: runtime.trustState,
1603
+ terminalOutcome: runtime.terminalOutcome,
1180
1604
  }
1181
- : undefined,
1605
+ : routeDecision
1606
+ ? {
1607
+ route: routeDecision.route,
1608
+ intent: routeDecision.intent,
1609
+ reason: routeDecision.reason,
1610
+ trustLabel: routeDecision.trustLabel,
1611
+ reviewStatus: routeDecision.reviewStatus,
1612
+ exactObjectKey: routeDecision.exactObjectKey,
1613
+ }
1614
+ : undefined,
1182
1615
  },
1183
1616
  {
1184
1617
  stage: 'tools',
1185
1618
  status: toolStatus,
1186
1619
  message: toolMessage,
1187
1620
  payload: {
1188
- observedToolCalls: toolCalls.length,
1621
+ observedToolCalls: observedToolCallCount,
1189
1622
  expectedMinToolCalls,
1190
- providerToolCalls: toolCalls.slice(0, 12).map((call) => ({
1623
+ ...(runtime ? { evidenceSource: 'persisted_agent_run.telemetry' } : {}),
1624
+ providerToolCalls: runtime ? [] : toolCalls.slice(0, 12).map((call) => ({
1191
1625
  order: call.order,
1192
1626
  name: call.name,
1193
1627
  status: call.status,
@@ -1261,6 +1695,22 @@ function buildEvalTrace(input) {
1261
1695
  promoteCommand: result.promoteCommand,
1262
1696
  },
1263
1697
  },
1698
+ {
1699
+ stage: 'observability',
1700
+ status: runtime?.observability?.recordingStatus === 'complete'
1701
+ ? 'passed'
1702
+ : runtime?.observability ? 'info' : 'not_run',
1703
+ message: runtime?.observability
1704
+ ? `Local Ask trace recording ${runtime.observability.recordingStatus}.`
1705
+ : 'No runtime trace receipt was attached to this evaluation run.',
1706
+ payload: runtime?.observability
1707
+ ? {
1708
+ recordingStatus: runtime.observability.recordingStatus,
1709
+ storeSchemaVersion: runtime.observability.storeSchemaVersion,
1710
+ ...(runtime.observability.traceFingerprint ? { traceFingerprint: runtime.observability.traceFingerprint } : {}),
1711
+ }
1712
+ : undefined,
1713
+ },
1264
1714
  {
1265
1715
  stage: 'scoring',
1266
1716
  status: evaluation.failures.length === 0 ? 'passed' : 'failed',