@duckcodeailabs/dql-cli 1.14.1 → 1.14.3-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/args.d.ts +4 -0
- package/dist/args.d.ts.map +1 -1
- package/dist/args.js +16 -0
- package/dist/args.js.map +1 -1
- package/dist/assets/dql-notebook/assets/{AgentLogPage-BPz-UWFh.js → AgentLogPage-Jbyb4so-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildDialog-BEl53WA_.js → AiBuildDialog-Du07X1qO.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildResult-B4yfGTTZ.js → AiBuildResult-xKtGthWv.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiSidePanel-CSZAAvuD.js → AiSidePanel-CE_L04dK.js} +1 -1
- package/dist/assets/dql-notebook/assets/AnalyticsHome-CiUE10uF.js +6 -0
- package/dist/assets/dql-notebook/assets/{AppsView-DQOwU9Cg.js → AppsView-DlNaTuwD.js} +30 -35
- package/dist/assets/dql-notebook/assets/AskObservabilityPage-DracZPI9.js +1 -0
- package/dist/assets/dql-notebook/assets/AskTracePage-CSMoMw9a.js +8 -0
- package/dist/assets/dql-notebook/assets/{BlockStudio-B4ap0GdY.js → BlockStudio-DQS5Hcq9.js} +9 -14
- package/dist/assets/dql-notebook/assets/BusinessArtifactView-CqV4i9l_.js +1 -0
- package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-D72byb2g.js → DbtFirstModelingPage-JKhB_Y2H.js} +3 -3
- package/dist/assets/dql-notebook/assets/{GitPage-lQb1uXH2.js → GitPage-DjaNp0o3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GlobalAiRail-CVECf6Xj.js → GlobalAiRail-CsV3zlc8.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GovernedContextPage-trOyMCY6.js → GovernedContextPage-uXUHUOIJ.js} +3 -3
- package/dist/assets/dql-notebook/assets/{HelpDocsPage-D8hLS5lE.js → HelpDocsPage-Hbh9IeYB.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HomePage-eIkfBIep.js → HomePage-CRRl-R_v.js} +1 -1
- package/dist/assets/dql-notebook/assets/LineageDAG-OMUP2sbl.js +1 -0
- package/dist/assets/dql-notebook/assets/LineageDetailView-C964BLuV.js +1 -0
- package/dist/assets/dql-notebook/assets/LineageDrawer-CzY0nfpD.js +1 -0
- package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-CviIf8PN.js → LineagePathBreadcrumb-0Gi-Fm1B.js} +1 -1
- package/dist/assets/dql-notebook/assets/MiniLineageGraph-wn0wLc9m.js +1 -0
- package/dist/assets/dql-notebook/assets/{NewBlockModal-DbCQg-pj.js → NewBlockModal-kMAzk91d.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewNotebookModal-5Vp6XiuK.js → NewNotebookModal-D_cWd_6M.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NotebookEditor-DjqJS44s.js → NotebookEditor-Bax0lxaz.js} +25 -30
- package/dist/assets/dql-notebook/assets/{ReadinessPage-BHGrC5ho.js → ReadinessPage-DEUIXQQY.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SetupOnboarding-BfS9Tdx-.js → SetupOnboarding-CyE-Ex41.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SkillsPage-BFH21wSj.js → SkillsPage-QgvKGPxE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{TrustBadge-zm6g_SxZ.js → TrustBadge-D5CGp0cu.js} +1 -1
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-BjsTE6yf.js +89 -0
- package/dist/assets/dql-notebook/assets/{answer-to-notebook-DPhxIEzF.js → answer-to-notebook-tACdCsTV.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-left-DNEb86Xc.js → arrow-left-CF6wxXU5.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-right-DpPWbwaD.js → arrow-right-C7G3M_S8.js} +1 -1
- package/dist/assets/dql-notebook/assets/{book-open-text-D7s5jo4X.js → book-open-text-DDBO8G7L.js} +1 -1
- package/dist/assets/dql-notebook/assets/chevron-left-DxajrX2v.js +6 -0
- package/dist/assets/dql-notebook/assets/{circle-x-Db3dXKg7.js → circle-x-HBvXjnT6.js} +1 -1
- package/dist/assets/dql-notebook/assets/clock-3-CJIdsPLt.js +6 -0
- package/dist/assets/dql-notebook/assets/dagre.esm-B6nvU4OB.js +1 -0
- package/dist/assets/dql-notebook/assets/{external-link-BxwXitO_.js → external-link-WciAU8Bu.js} +1 -1
- package/dist/assets/dql-notebook/assets/{grip-vertical-Dt4lkWRi.js → grip-vertical-DVu7ok6x.js} +1 -1
- package/dist/assets/dql-notebook/assets/index-B_kaoARS.css +1 -0
- package/dist/assets/dql-notebook/assets/{index-DKo-bwNw.js → index-D3wnucQC.js} +133 -128
- package/dist/assets/dql-notebook/assets/{link-2-Dfo2P6wi.js → link-2-C6uQx50X.js} +1 -1
- package/dist/assets/dql-notebook/assets/{list-tree-DiTmIWAL.js → list-tree-DapsIuQB.js} +1 -1
- package/dist/assets/dql-notebook/assets/{minimize-2-CMTAkPzL.js → minimize-2-BlVoipnT.js} +1 -1
- package/dist/assets/dql-notebook/assets/{panel-right-open-DwYr7FW4.js → panel-right-open-DcnKWxfx.js} +1 -1
- package/dist/assets/dql-notebook/assets/{play-BXhHYQ4x.js → play-CNuRuaAs.js} +1 -1
- package/dist/assets/dql-notebook/assets/{rotate-ccw-BNi6F8pl.js → rotate-ccw-B2p953Zh.js} +1 -1
- package/dist/assets/dql-notebook/assets/{semantic-fields-CNOGysAy.js → semantic-fields-2fWgkTJc.js} +1 -1
- package/dist/assets/dql-notebook/assets/{sliders-horizontal-Ec5MUMUW.js → sliders-horizontal-DwRTu1w4.js} +1 -1
- package/dist/assets/dql-notebook/assets/{star-B9leDkp_.js → star-Cbaj4F3i.js} +1 -1
- package/dist/assets/dql-notebook/assets/style-NlN9F6JE.js +23 -0
- package/dist/assets/dql-notebook/assets/{triangle-alert-BefTYCzx.js → triangle-alert-BPHvH1wA.js} +1 -1
- package/dist/assets/dql-notebook/assets/{upload-SPiOM2tQ.js → upload-DIqK0KE1.js} +1 -1
- package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-C4foXeiQ.js → usePersistedAgentThreadId-DpQXOEDw.js} +1 -1
- package/dist/assets/dql-notebook/assets/{user-round-comGmyw-.js → user-round-DukNupB_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wand-sparkles-BffR4dF8.js → wand-sparkles-CAwLX0b9.js} +1 -1
- package/dist/assets/dql-notebook/assets/{workflow-ChmPTEzH.js → workflow-bOT-NqdO.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wrench-DovaG_ze.js → wrench-1m0_kcza.js} +1 -1
- package/dist/assets/dql-notebook/assets/{x-nRx91AgW.js → x-fp82BETM.js} +1 -1
- package/dist/assets/dql-notebook/index.html +2 -2
- package/dist/commands/agent-eval-cassette.d.ts +97 -6
- package/dist/commands/agent-eval-cassette.d.ts.map +1 -1
- package/dist/commands/agent-eval-cassette.js +165 -22
- package/dist/commands/agent-eval-cassette.js.map +1 -1
- package/dist/commands/agent-eval-runtime.d.ts +23 -2
- package/dist/commands/agent-eval-runtime.d.ts.map +1 -1
- package/dist/commands/agent-eval-runtime.js +19 -2
- package/dist/commands/agent-eval-runtime.js.map +1 -1
- package/dist/commands/agent-trace.d.ts +3 -0
- package/dist/commands/agent-trace.d.ts.map +1 -0
- package/dist/commands/agent-trace.js +169 -0
- package/dist/commands/agent-trace.js.map +1 -0
- package/dist/commands/agent.d.ts +37 -5
- package/dist/commands/agent.d.ts.map +1 -1
- package/dist/commands/agent.js +534 -84
- package/dist/commands/agent.js.map +1 -1
- package/dist/commands/compile.d.ts +13 -1
- package/dist/commands/compile.d.ts.map +1 -1
- package/dist/commands/compile.js +36 -4
- package/dist/commands/compile.js.map +1 -1
- package/dist/commands/notebook.d.ts +2 -0
- package/dist/commands/notebook.d.ts.map +1 -1
- package/dist/commands/notebook.js +4 -0
- package/dist/commands/notebook.js.map +1 -1
- package/dist/commands/sync.d.ts.map +1 -1
- package/dist/commands/sync.js +11 -3
- package/dist/commands/sync.js.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/llm/providers/dql-agent-provider.d.ts +29 -2
- package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
- package/dist/llm/providers/dql-agent-provider.js +1008 -144
- package/dist/llm/providers/dql-agent-provider.js.map +1 -1
- package/dist/llm/types.d.ts +48 -1
- package/dist/llm/types.d.ts.map +1 -1
- package/dist/local-runtime.d.ts +271 -33
- package/dist/local-runtime.d.ts.map +1 -1
- package/dist/local-runtime.js +3658 -546
- package/dist/local-runtime.js.map +1 -1
- package/dist/package.json +10 -10
- package/dist/providers/oauth/claude-oauth.d.ts.map +1 -1
- package/dist/providers/oauth/claude-oauth.js +52 -18
- package/dist/providers/oauth/claude-oauth.js.map +1 -1
- package/dist/providers/oauth/codex-oauth.d.ts.map +1 -1
- package/dist/providers/oauth/codex-oauth.js +72 -44
- package/dist/providers/oauth/codex-oauth.js.map +1 -1
- package/dist/providers/subscription-cli.d.ts +20 -0
- package/dist/providers/subscription-cli.d.ts.map +1 -1
- package/dist/providers/subscription-cli.js +69 -11
- package/dist/providers/subscription-cli.js.map +1 -1
- package/package.json +10 -10
- package/dist/assets/dql-notebook/assets/AnalyticsHome-D5P6Ujwi.js +0 -6
- package/dist/assets/dql-notebook/assets/BusinessArtifactView-BIfNI-0S.js +0 -1
- package/dist/assets/dql-notebook/assets/LineageDAG-CSqcbDrE.js +0 -1
- package/dist/assets/dql-notebook/assets/LineageDetailView-DJZZjZu-.js +0 -1
- package/dist/assets/dql-notebook/assets/LineageDrawer-BcIipIc3.js +0 -1
- package/dist/assets/dql-notebook/assets/MiniLineageGraph-vH_MY_Ju.js +0 -1
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel--oxmjlgr.js +0 -88
- package/dist/assets/dql-notebook/assets/dagre.esm-C7pppQ1a.js +0 -23
- package/dist/assets/dql-notebook/assets/index-B3shyZsg.css +0 -1
- /package/dist/assets/dql-notebook/assets/{dagre-BZV40eAE.css → style-BZV40eAE.css} +0 -0
package/dist/commands/agent.js
CHANGED
|
@@ -21,17 +21,19 @@
|
|
|
21
21
|
* dql agent feedback <up|down> --block <id> --question "..."
|
|
22
22
|
* Records feedback into the KG. Used by clients without MCP access.
|
|
23
23
|
*/
|
|
24
|
-
import { answerFromRuntimeRun, driveViaRuntime,
|
|
25
|
-
import { CassetteStore, cassetteDirFor, withCassette } from './agent-eval-cassette.js';
|
|
24
|
+
import { answerFromRuntimeRun, driveViaRuntime, projectRuntimeRun, } from './agent-eval-runtime.js';
|
|
25
|
+
import { CassetteStore, cassetteEvidenceSummary, cassetteDirFor, evalCassetteCanonicalizationV2, withCassette, } from './agent-eval-cassette.js';
|
|
26
26
|
import { existsSync, readFileSync } from 'node:fs';
|
|
27
27
|
import { join, resolve } from 'node:path';
|
|
28
28
|
import { load as loadYaml } from 'js-yaml';
|
|
29
|
-
import { KGStore, MemoryStore, defaultKgPath, defaultMemoryPath, reindexProject, loadSkills, pickProvider, answer, resolveDomainContextEnvelope, buildAnalysisQuestionPlan, buildLocalContextPack, coerceReasoningEffort, contextRetrievalBudgetForQuestion, deriveGeneratedDraftSlug, loadAgentSemanticLayer, recordQueryRun, recordRuntimeSchemaSnapshot, upsertGeneratedDqlArtifactDraft, upsertGeneratedDraft, validateSqlAgainstLocalContext, } from '@duckcodeailabs/dql-agent';
|
|
29
|
+
import { AskTraceSqliteStoreV1, KGStore, MemoryStore, defaultKgPath, defaultMemoryPath, reindexProject, loadSkills, pickProvider, answer, resolveDomainContextEnvelope, buildAnalysisQuestionPlan, buildLocalContextPack, classifyProviderFailure, coerceReasoningEffort, contextRetrievalBudgetForQuestion, deriveGeneratedDraftSlug, loadAgentSemanticLayer, recordQueryRun, recordRuntimeSchemaSnapshot, upsertGeneratedDqlArtifactDraft, upsertGeneratedDraft, validateSqlAgainstLocalContext, createAskTraceObserverV1, defaultAskTraceSqlitePath, } from '@duckcodeailabs/dql-agent';
|
|
30
|
+
import { createHash, randomUUID } from 'node:crypto';
|
|
30
31
|
import { buildManifest, resolveDbtManifestPath } from '@duckcodeailabs/dql-core';
|
|
31
32
|
import { findProjectRoot } from '../local-runtime.js';
|
|
32
33
|
import { buildAnswerLoopTools, createGroundingContextExpander } from '../llm/answer-loop-tools.js';
|
|
33
34
|
import { judgeAnswer } from './eval-judge.js';
|
|
34
35
|
import { startProjectRuntime } from './notebook.js';
|
|
36
|
+
import { runAgentTrace } from './agent-trace.js';
|
|
35
37
|
/**
|
|
36
38
|
* Resolve the runtime the agent posts certified blocks / generated SQL to.
|
|
37
39
|
*
|
|
@@ -55,7 +57,56 @@ async function resolveAgentRuntime(projectRoot, flags) {
|
|
|
55
57
|
return { runtimeBase: base, close: async () => { } };
|
|
56
58
|
}
|
|
57
59
|
const handle = await startProjectRuntime(projectRoot, { preferredPort: 0 });
|
|
58
|
-
return { runtimeBase: handle.url, close: handle.close };
|
|
60
|
+
return { runtimeBase: handle.url, close: handle.close, askTraceCapability: handle.askTraceCapability };
|
|
61
|
+
}
|
|
62
|
+
function isLoopbackRuntimeUrl(runtimeBase) {
|
|
63
|
+
try {
|
|
64
|
+
const hostname = new URL(runtimeBase).hostname;
|
|
65
|
+
return hostname === '127.0.0.1' || hostname === 'localhost' || hostname === '::1' || hostname === '[::1]';
|
|
66
|
+
}
|
|
67
|
+
catch {
|
|
68
|
+
return false;
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* An already-running local Notebook runtime mints a one-shot, loopback-only
|
|
73
|
+
* capability for this CLI request. Remote runtimes deliberately receive no
|
|
74
|
+
* capability and therefore cannot be relabelled as CLI by arbitrary text.
|
|
75
|
+
*/
|
|
76
|
+
export async function requestLoopbackCliAskTraceCapability(runtimeBase) {
|
|
77
|
+
if (!isLoopbackRuntimeUrl(runtimeBase))
|
|
78
|
+
return undefined;
|
|
79
|
+
try {
|
|
80
|
+
const response = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/ask-traces/cli-capability`, {
|
|
81
|
+
headers: { Accept: 'application/json' },
|
|
82
|
+
});
|
|
83
|
+
if (!response.ok)
|
|
84
|
+
return undefined;
|
|
85
|
+
const payload = await response.json();
|
|
86
|
+
return typeof payload.capability === 'string'
|
|
87
|
+
&& typeof payload.expiresAt === 'string'
|
|
88
|
+
&& payload.scope === 'agent-runs'
|
|
89
|
+
? payload.capability
|
|
90
|
+
: undefined;
|
|
91
|
+
}
|
|
92
|
+
catch {
|
|
93
|
+
// A pre-observability runtime remains usable. Its request stays browser
|
|
94
|
+
// attributed rather than fabricating a client-controlled CLI surface.
|
|
95
|
+
return undefined;
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
function runtimeProviderForCliFlag(value) {
|
|
99
|
+
if (typeof value !== 'string' || !value.trim())
|
|
100
|
+
return undefined;
|
|
101
|
+
switch (value.trim().toLowerCase()) {
|
|
102
|
+
case 'claude': return 'anthropic';
|
|
103
|
+
case 'openai': return 'openai';
|
|
104
|
+
case 'gemini': return 'gemini';
|
|
105
|
+
case 'ollama': return 'ollama';
|
|
106
|
+
// Preserve an invalid explicit value through the host request so the
|
|
107
|
+
// canonical preflight can return a typed `model_not_found` diagnostic.
|
|
108
|
+
default: return value.trim().toLowerCase();
|
|
109
|
+
}
|
|
59
110
|
}
|
|
60
111
|
async function fetchRuntimeSchemaContext(runtimeBase) {
|
|
61
112
|
try {
|
|
@@ -142,6 +193,137 @@ function cliAnalysisDepth(flags) {
|
|
|
142
193
|
const value = flags.analysisDepth?.trim().toLowerCase();
|
|
143
194
|
return value === 'quick' || value === 'deep' ? value : undefined;
|
|
144
195
|
}
|
|
196
|
+
/**
|
|
197
|
+
* Compatibility test adapter for a standalone provider boundary. Production
|
|
198
|
+
* `dql agent ask` no longer calls this: it uses the runtime AgentRun engine so
|
|
199
|
+
* the router, cascade, freeze, tools, SQL, and provider spans share one
|
|
200
|
+
* canonical trace. Keep this adapter truthful for lower-level provider tests
|
|
201
|
+
* without making it an alternate orchestration authority.
|
|
202
|
+
*/
|
|
203
|
+
export function createDirectCliAskTraceProvider(provider, trace) {
|
|
204
|
+
let attemptIndex = 0;
|
|
205
|
+
let lastFailedSpanId;
|
|
206
|
+
const pending = new Map();
|
|
207
|
+
const keyFor = (event) => `${event.provider}:${event.operation}:${event.attemptIndex}`;
|
|
208
|
+
const attempt = (event, retryOfSpanId) => ({
|
|
209
|
+
version: 1,
|
|
210
|
+
phase: 'generation',
|
|
211
|
+
physicalAttemptIndex: ++attemptIndex,
|
|
212
|
+
providerFingerprint: `sha256:${createHash('sha256').update(event.provider).digest('hex')}`,
|
|
213
|
+
...(event.model ? { modelFingerprint: `sha256:${createHash('sha256').update(event.model).digest('hex')}` } : {}),
|
|
214
|
+
...(retryOfSpanId ? { retryOfSpanId } : {}),
|
|
215
|
+
admission: 'admitted',
|
|
216
|
+
provenance: 'live',
|
|
217
|
+
});
|
|
218
|
+
const finish = (entry, outcome, error) => {
|
|
219
|
+
if (!entry.spanId)
|
|
220
|
+
return;
|
|
221
|
+
const diagnostic = outcome === 'ok' ? undefined : classifyProviderFailure({
|
|
222
|
+
phase: 'generation',
|
|
223
|
+
code: error && typeof error === 'object' ? String(error.code ?? '') : undefined,
|
|
224
|
+
message: error instanceof Error ? error.message : String(error ?? ''),
|
|
225
|
+
providerFingerprint: entry.attempt.providerFingerprint,
|
|
226
|
+
modelFingerprint: entry.attempt.modelFingerprint,
|
|
227
|
+
});
|
|
228
|
+
const finalAttempt = outcome === 'ok'
|
|
229
|
+
? entry.attempt
|
|
230
|
+
: {
|
|
231
|
+
...entry.attempt,
|
|
232
|
+
...(diagnostic?.httpStatusClass ? { httpStatusClass: diagnostic.httpStatusClass } : {}),
|
|
233
|
+
...(diagnostic ? { retryable: diagnostic.retryable, safeAction: diagnostic.safeAction } : {}),
|
|
234
|
+
cause: outcome === 'cancelled' ? 'cancelled' : diagnostic?.cause ?? 'unknown',
|
|
235
|
+
};
|
|
236
|
+
trace.finishSpan(entry.spanId, {
|
|
237
|
+
outcome: outcome === 'ok' ? 'ok' : outcome === 'cancelled' ? 'cancelled' : 'error',
|
|
238
|
+
reasonCode: outcome === 'ok' ? 'completed' : outcome === 'cancelled' ? 'cancelled' : 'provider_failure',
|
|
239
|
+
payload: { kind: 'provider', attempt: finalAttempt },
|
|
240
|
+
});
|
|
241
|
+
if (outcome !== 'ok')
|
|
242
|
+
lastFailedSpanId = entry.spanId;
|
|
243
|
+
};
|
|
244
|
+
const observedOptions = (options = {}) => ({
|
|
245
|
+
...options,
|
|
246
|
+
onProviderDispatch: (event) => {
|
|
247
|
+
const tracedAttempt = attempt(event, lastFailedSpanId);
|
|
248
|
+
const spanId = trace.startSpan({
|
|
249
|
+
name: 'provider.attempt',
|
|
250
|
+
stage: 'provider',
|
|
251
|
+
reasonCode: 'started',
|
|
252
|
+
payload: { kind: 'provider', attempt: tracedAttempt },
|
|
253
|
+
});
|
|
254
|
+
const key = keyFor(event);
|
|
255
|
+
pending.set(key, [...(pending.get(key) ?? []), { spanId, attempt: tracedAttempt }]);
|
|
256
|
+
return options.onProviderDispatch?.(event) ?? event.envelope;
|
|
257
|
+
},
|
|
258
|
+
onProviderDispatchComplete: (event) => {
|
|
259
|
+
const key = keyFor(event);
|
|
260
|
+
const entries = pending.get(key) ?? [];
|
|
261
|
+
const entry = entries[0];
|
|
262
|
+
if (entry && event.outcome === 'ok' && (event.settlement === 'transport' || event.settlement === 'process')) {
|
|
263
|
+
entry.attempt = {
|
|
264
|
+
...entry.attempt,
|
|
265
|
+
...(event.settlement === 'transport' ? { transportOutcome: 'ok' } : { processOutcome: 'ok' }),
|
|
266
|
+
};
|
|
267
|
+
}
|
|
268
|
+
else {
|
|
269
|
+
const closed = entries.shift();
|
|
270
|
+
if (entries.length > 0)
|
|
271
|
+
pending.set(key, entries);
|
|
272
|
+
else
|
|
273
|
+
pending.delete(key);
|
|
274
|
+
if (closed)
|
|
275
|
+
finish(closed, event.outcome, event.error ?? (typeof event.httpStatus === 'number' ? Object.assign(new Error(`HTTP ${event.httpStatus}`), { code: `HTTP_${event.httpStatus}` }) : undefined));
|
|
276
|
+
}
|
|
277
|
+
options.onProviderDispatchComplete?.(event);
|
|
278
|
+
},
|
|
279
|
+
onProviderDispatchRejected: (event) => {
|
|
280
|
+
const denied = trace.startSpan({
|
|
281
|
+
name: 'provider.attempt',
|
|
282
|
+
stage: 'provider',
|
|
283
|
+
reasonCode: 'provider_failure',
|
|
284
|
+
payload: {
|
|
285
|
+
kind: 'provider',
|
|
286
|
+
attempt: {
|
|
287
|
+
...attempt(event, lastFailedSpanId),
|
|
288
|
+
admission: 'denied',
|
|
289
|
+
},
|
|
290
|
+
},
|
|
291
|
+
});
|
|
292
|
+
trace.finishSpan(denied, { outcome: 'denied', reasonCode: 'provider_failure' });
|
|
293
|
+
lastFailedSpanId = denied;
|
|
294
|
+
options.onProviderDispatchRejected?.(event);
|
|
295
|
+
},
|
|
296
|
+
});
|
|
297
|
+
const invoke = async (options, call) => {
|
|
298
|
+
try {
|
|
299
|
+
const result = await call(observedOptions(options));
|
|
300
|
+
for (const entries of pending.values())
|
|
301
|
+
for (const entry of entries)
|
|
302
|
+
finish(entry, 'ok');
|
|
303
|
+
pending.clear();
|
|
304
|
+
return result;
|
|
305
|
+
}
|
|
306
|
+
catch (error) {
|
|
307
|
+
const outcome = options?.signal?.aborted ? 'cancelled' : 'error';
|
|
308
|
+
for (const entries of pending.values())
|
|
309
|
+
for (const entry of entries)
|
|
310
|
+
finish(entry, outcome, error);
|
|
311
|
+
pending.clear();
|
|
312
|
+
throw error;
|
|
313
|
+
}
|
|
314
|
+
};
|
|
315
|
+
return {
|
|
316
|
+
name: provider.name,
|
|
317
|
+
available: () => provider.available(),
|
|
318
|
+
generate: (messages, options) => invoke(options, (observed) => provider.generate(messages, observed)),
|
|
319
|
+
...(provider.generateWithTools ? {
|
|
320
|
+
generateWithTools: (messages, tools, options) => invoke(options, (observed) => provider.generateWithTools(messages, tools, observed)),
|
|
321
|
+
} : {}),
|
|
322
|
+
...(provider.generateStream ? {
|
|
323
|
+
generateStream: (messages, options, onDelta) => invoke(options, (observed) => provider.generateStream(messages, observed, onDelta)),
|
|
324
|
+
} : {}),
|
|
325
|
+
};
|
|
326
|
+
}
|
|
145
327
|
/** A DQL runtime answers `/api/connections` with a connector/connection payload. */
|
|
146
328
|
async function isDqlRuntime(base) {
|
|
147
329
|
try {
|
|
@@ -161,6 +343,8 @@ export async function runAgent(sub, rest, flags) {
|
|
|
161
343
|
return runAsk(rest, flags);
|
|
162
344
|
case 'threads':
|
|
163
345
|
return runThreads(flags);
|
|
346
|
+
case 'trace':
|
|
347
|
+
return runAgentTrace(rest, flags);
|
|
164
348
|
case 'reindex':
|
|
165
349
|
return runReindex(rest, flags);
|
|
166
350
|
case 'feedback':
|
|
@@ -168,18 +352,41 @@ export async function runAgent(sub, rest, flags) {
|
|
|
168
352
|
case 'eval':
|
|
169
353
|
return runEval(rest, flags);
|
|
170
354
|
default:
|
|
171
|
-
throw new Error('Usage: dql agent <ask|threads|reindex|feedback|eval> [args]\n' +
|
|
355
|
+
throw new Error('Usage: dql agent <ask|threads|trace|reindex|feedback|eval> [args]\n' +
|
|
172
356
|
' dql agent ask "<question>" [--provider claude|openai|gemini|ollama] [--user <id>] [--domain <d>] [--purpose <approved-purpose>] [--thread <id>]\n' +
|
|
173
357
|
' dql agent threads [--runtime-url <url>]\n' +
|
|
358
|
+
' dql agent trace list|show|export|validate|replay|compare\n' +
|
|
174
359
|
' dql agent reindex [path]\n' +
|
|
175
360
|
' dql agent feedback up|down --block <id> --question "..."\n' +
|
|
176
361
|
' dql agent eval agent-evals.yml [--provider claude|openai|gemini|ollama] [--execute] [--save]');
|
|
177
362
|
}
|
|
178
363
|
}
|
|
364
|
+
/**
|
|
365
|
+
* Canonical CLI Ask entrypoint. Direct and threaded CLI questions use the
|
|
366
|
+
* exact local AgentRun engine as the browser, so trace evidence follows the
|
|
367
|
+
* router's candidate/cascade/freeze/execution authority rather than a legacy
|
|
368
|
+
* answer-loop approximation.
|
|
369
|
+
*/
|
|
179
370
|
async function runAsk(rest, flags) {
|
|
180
371
|
const question = rest.join(' ').trim();
|
|
181
372
|
if (!question)
|
|
182
373
|
throw new Error('Usage: dql agent ask "<question>"');
|
|
374
|
+
const threadId = flags.thread;
|
|
375
|
+
return threadId
|
|
376
|
+
? runThreadAsk(question, threadId, flags)
|
|
377
|
+
: runCanonicalCliAsk(question, undefined, flags);
|
|
378
|
+
}
|
|
379
|
+
/**
|
|
380
|
+
* Compatibility entrypoint retained for downstream imports during the CLI
|
|
381
|
+
* transition. It deliberately delegates before allocating any legacy state,
|
|
382
|
+
* so even an old internal caller gets the canonical AgentRun trace rather than
|
|
383
|
+
* an incomplete synthetic receipt.
|
|
384
|
+
*/
|
|
385
|
+
async function runLegacyDirectAsk(rest, flags) {
|
|
386
|
+
const question = rest.join(' ').trim();
|
|
387
|
+
if (!question)
|
|
388
|
+
throw new Error('Usage: dql agent ask "<question>"');
|
|
389
|
+
return runCanonicalCliAsk(question, flags.thread, flags);
|
|
183
390
|
// Thread-scoped ask: hand the question to the runtime's agent-run engine with
|
|
184
391
|
// the thread id, so the SERVER injects prior turns and persists this run as a
|
|
185
392
|
// new turn (the same conversation store the notebook UI uses).
|
|
@@ -187,8 +394,31 @@ async function runAsk(rest, flags) {
|
|
|
187
394
|
if (threadId)
|
|
188
395
|
return runThreadAsk(question, threadId, flags);
|
|
189
396
|
const projectRoot = findProjectRoot(process.cwd());
|
|
397
|
+
const traceStore = new AskTraceSqliteStoreV1({ path: defaultAskTraceSqlitePath(projectRoot) });
|
|
398
|
+
const trace = createAskTraceObserverV1({
|
|
399
|
+
store: traceStore,
|
|
400
|
+
runId: `cli-${randomUUID()}`,
|
|
401
|
+
surface: 'cli',
|
|
402
|
+
mode: 'ask',
|
|
403
|
+
questionFingerprint: `sha256:${createHash('sha256').update(question).digest('hex')}`,
|
|
404
|
+
});
|
|
405
|
+
const classifySpan = trace.startSpan({
|
|
406
|
+
name: 'request.classify',
|
|
407
|
+
stage: 'request',
|
|
408
|
+
reasonCode: 'started',
|
|
409
|
+
payload: { kind: 'stage', route: 'direct_cli_legacy' },
|
|
410
|
+
});
|
|
190
411
|
const kgPath = defaultKgPath(projectRoot);
|
|
191
|
-
|
|
412
|
+
try {
|
|
413
|
+
await reindexProject(projectRoot, { kgPath });
|
|
414
|
+
trace.finishSpan(classifySpan, { outcome: 'ok', reasonCode: 'completed' });
|
|
415
|
+
}
|
|
416
|
+
catch (error) {
|
|
417
|
+
trace.finishSpan(classifySpan, { outcome: 'error', reasonCode: 'unknown' });
|
|
418
|
+
trace.finalize({ status: 'failed' });
|
|
419
|
+
traceStore.close();
|
|
420
|
+
throw error;
|
|
421
|
+
}
|
|
192
422
|
const providerName = flags.provider;
|
|
193
423
|
const userId = flags.user;
|
|
194
424
|
const domain = flags.domain;
|
|
@@ -196,12 +426,68 @@ async function runAsk(rest, flags) {
|
|
|
196
426
|
const format = flags.format;
|
|
197
427
|
const reasoningEffort = cliReasoningEffort(flags);
|
|
198
428
|
const requestedDepth = cliAnalysisDepth(flags);
|
|
199
|
-
|
|
429
|
+
let provider;
|
|
430
|
+
try {
|
|
431
|
+
provider = await pickProvider(providerName);
|
|
432
|
+
}
|
|
433
|
+
catch (error) {
|
|
434
|
+
trace.finalize({ status: 'failed' });
|
|
435
|
+
traceStore.close();
|
|
436
|
+
throw error;
|
|
437
|
+
}
|
|
200
438
|
const kg = new KGStore(kgPath);
|
|
201
439
|
const memory = new MemoryStore(defaultMemoryPath(projectRoot));
|
|
202
440
|
const { skills } = loadSkills(projectRoot);
|
|
203
441
|
let closeRuntime;
|
|
204
442
|
try {
|
|
443
|
+
const preflight = trace.startSpan({
|
|
444
|
+
name: 'provider.preflight',
|
|
445
|
+
stage: 'provider',
|
|
446
|
+
reasonCode: 'started',
|
|
447
|
+
payload: {
|
|
448
|
+
kind: 'provider',
|
|
449
|
+
attempt: {
|
|
450
|
+
version: 1,
|
|
451
|
+
phase: 'preflight',
|
|
452
|
+
physicalAttemptIndex: 0,
|
|
453
|
+
providerFingerprint: `sha256:${createHash('sha256').update(provider.name).digest('hex')}`,
|
|
454
|
+
readiness: 'unknown',
|
|
455
|
+
admission: 'unknown',
|
|
456
|
+
provenance: 'live',
|
|
457
|
+
},
|
|
458
|
+
},
|
|
459
|
+
});
|
|
460
|
+
let providerReady = false;
|
|
461
|
+
try {
|
|
462
|
+
providerReady = await provider.available();
|
|
463
|
+
}
|
|
464
|
+
catch {
|
|
465
|
+
providerReady = false;
|
|
466
|
+
}
|
|
467
|
+
trace.finishSpan(preflight, {
|
|
468
|
+
outcome: providerReady ? 'ok' : 'unavailable',
|
|
469
|
+
reasonCode: providerReady ? 'completed' : 'provider_preflight',
|
|
470
|
+
payload: {
|
|
471
|
+
kind: 'provider',
|
|
472
|
+
attempt: {
|
|
473
|
+
version: 1,
|
|
474
|
+
phase: 'preflight',
|
|
475
|
+
physicalAttemptIndex: 0,
|
|
476
|
+
providerFingerprint: `sha256:${createHash('sha256').update(provider.name).digest('hex')}`,
|
|
477
|
+
readiness: providerReady ? 'ready' : 'unavailable',
|
|
478
|
+
admission: 'unknown',
|
|
479
|
+
cause: providerReady ? undefined : 'authentication',
|
|
480
|
+
safeAction: providerReady ? undefined : 'fix_provider_configuration',
|
|
481
|
+
provenance: 'live',
|
|
482
|
+
},
|
|
483
|
+
},
|
|
484
|
+
});
|
|
485
|
+
if (!providerReady) {
|
|
486
|
+
throw Object.assign(new Error('The selected AI provider is not ready. Configure or sign in to the provider and retry.'), {
|
|
487
|
+
code: 'AUTHENTICATION_FAILED',
|
|
488
|
+
});
|
|
489
|
+
}
|
|
490
|
+
const tracedProvider = createDirectCliAskTraceProvider(provider, trace);
|
|
205
491
|
const memoryContext = memory.search({
|
|
206
492
|
query: question,
|
|
207
493
|
scopes: ['project', 'user', 'artifact'],
|
|
@@ -220,6 +506,12 @@ async function runAsk(rest, flags) {
|
|
|
220
506
|
closeRuntime = close;
|
|
221
507
|
const schemaContext = await fetchRuntimeSchemaContext(runtimeBase);
|
|
222
508
|
recordCliRuntimeSchemaSnapshot(projectRoot, schemaContext, 'direct CLI runtime schema');
|
|
509
|
+
const retrievalSpan = trace.startSpan({
|
|
510
|
+
name: 'retrieval',
|
|
511
|
+
stage: 'retrieval',
|
|
512
|
+
reasonCode: 'started',
|
|
513
|
+
payload: { kind: 'retrieval', candidateCount: 0 },
|
|
514
|
+
});
|
|
223
515
|
const contextPack = await buildLocalContextPack(projectRoot, {
|
|
224
516
|
question,
|
|
225
517
|
surface: 'cli',
|
|
@@ -233,10 +525,15 @@ async function runAsk(rest, flags) {
|
|
|
233
525
|
}
|
|
234
526
|
: undefined,
|
|
235
527
|
}).catch(() => undefined);
|
|
528
|
+
trace.finishSpan(retrievalSpan, {
|
|
529
|
+
outcome: 'ok',
|
|
530
|
+
reasonCode: 'completed',
|
|
531
|
+
payload: { kind: 'retrieval', candidateCount: contextPack?.objects.length ?? 0 },
|
|
532
|
+
});
|
|
236
533
|
const answerLoopTools = buildAnswerLoopTools(projectRoot);
|
|
237
534
|
const result = await answer({
|
|
238
535
|
question,
|
|
239
|
-
provider,
|
|
536
|
+
provider: tracedProvider,
|
|
240
537
|
kg,
|
|
241
538
|
manifest,
|
|
242
539
|
skills,
|
|
@@ -365,6 +662,7 @@ async function runAsk(rest, flags) {
|
|
|
365
662
|
});
|
|
366
663
|
},
|
|
367
664
|
});
|
|
665
|
+
trace.finalize({ status: 'completed' });
|
|
368
666
|
if (format === 'json') {
|
|
369
667
|
console.log(JSON.stringify(result, null, 2));
|
|
370
668
|
return;
|
|
@@ -379,17 +677,25 @@ async function runAsk(rest, flags) {
|
|
|
379
677
|
: '';
|
|
380
678
|
const footer = result.provenanceFooter ? `\n\n— ${result.provenanceFooter}` : '';
|
|
381
679
|
console.log(`${badge}\n\n${result.text}${footer}${cite}`);
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
console.log(
|
|
680
|
+
const resultPayload = result.result;
|
|
681
|
+
if (resultPayload) {
|
|
682
|
+
console.log(`\nRows: ${resultPayload.rowCount}`);
|
|
683
|
+
console.log(JSON.stringify(resultPayload.rows.slice(0, 5), null, 2));
|
|
385
684
|
}
|
|
386
685
|
printDqlArtifactPreview(result);
|
|
387
686
|
}
|
|
687
|
+
catch (error) {
|
|
688
|
+
const cancelled = error instanceof Error && error.name === 'AbortError';
|
|
689
|
+
trace.finalize({ status: cancelled ? 'cancelled' : 'failed' });
|
|
690
|
+
throw error;
|
|
691
|
+
}
|
|
388
692
|
finally {
|
|
389
693
|
kg.close();
|
|
390
694
|
memory.close();
|
|
391
|
-
|
|
392
|
-
|
|
695
|
+
const close = closeRuntime;
|
|
696
|
+
if (close)
|
|
697
|
+
await close();
|
|
698
|
+
traceStore.close();
|
|
393
699
|
}
|
|
394
700
|
}
|
|
395
701
|
function printDqlArtifactPreview(result) {
|
|
@@ -410,56 +716,103 @@ function printDqlArtifactPreview(result) {
|
|
|
410
716
|
console.log(`Promote: ${result.promoteCommand}`);
|
|
411
717
|
}
|
|
412
718
|
}
|
|
719
|
+
function printCanonicalCliAskRun(run, threadId) {
|
|
720
|
+
const badge = run.trustState === 'certified'
|
|
721
|
+
? '✓ Certified'
|
|
722
|
+
: run.trustState === 'grounded'
|
|
723
|
+
? '✓ Verified (grounded)'
|
|
724
|
+
: run.trustState === 'review_required'
|
|
725
|
+
? '! AI-generated · review required'
|
|
726
|
+
: run.trustState === 'blocked'
|
|
727
|
+
? '✕ Blocked'
|
|
728
|
+
: '· Reply';
|
|
729
|
+
console.log(`${badge}\n\n${(run.answer ?? run.summary ?? '').trim()}`);
|
|
730
|
+
if (threadId)
|
|
731
|
+
console.log(`\nThread: ${threadId}`);
|
|
732
|
+
const trace = run.traceReference;
|
|
733
|
+
if (trace?.traceId) {
|
|
734
|
+
console.log(`\nTrace: ${trace.traceId} (${trace.recordingStatus ?? 'recording'})`);
|
|
735
|
+
}
|
|
736
|
+
else {
|
|
737
|
+
console.log('\nTrace: unavailable');
|
|
738
|
+
}
|
|
739
|
+
}
|
|
413
740
|
/**
|
|
414
|
-
*
|
|
415
|
-
*
|
|
416
|
-
*
|
|
417
|
-
*
|
|
418
|
-
*
|
|
741
|
+
* Submit every user-visible CLI Ask through the runtime AgentRun endpoint.
|
|
742
|
+
* The runtime owns run IDs, conversation hydration, candidate provenance,
|
|
743
|
+
* cascade/freeze authority, physical tool/SQL execution, and final trace
|
|
744
|
+
* reference. The CLI only supplies user intent and an optional host-minted
|
|
745
|
+
* surface capability.
|
|
419
746
|
*/
|
|
420
|
-
async function
|
|
747
|
+
export async function runCanonicalCliAsk(question, threadId, flags) {
|
|
421
748
|
const projectRoot = findProjectRoot(process.cwd());
|
|
422
749
|
const format = flags.format;
|
|
423
|
-
const { runtimeBase, close } = await resolveAgentRuntime(projectRoot, flags);
|
|
750
|
+
const { runtimeBase, close, askTraceCapability: embeddedCapability } = await resolveAgentRuntime(projectRoot, flags);
|
|
424
751
|
try {
|
|
752
|
+
const askTraceCapability = embeddedCapability ?? await requestLoopbackCliAskTraceCapability(runtimeBase);
|
|
425
753
|
const reasoningEffort = cliReasoningEffort(flags);
|
|
426
754
|
const analysisDepth = cliAnalysisDepth(flags);
|
|
755
|
+
const domain = flags.domain;
|
|
756
|
+
const purpose = flags.purpose || undefined;
|
|
757
|
+
const userId = flags.user;
|
|
758
|
+
const provider = runtimeProviderForCliFlag(flags.provider);
|
|
759
|
+
// `surface` is deliberately absent from public JSON. The runtime assigns
|
|
760
|
+
// trace surface only after consuming its own loopback capability; this
|
|
761
|
+
// context carries user intent, never attribution authority.
|
|
762
|
+
const workspaceContext = {
|
|
763
|
+
...(domain ? { domain } : {}),
|
|
764
|
+
...(purpose ? { purpose } : {}),
|
|
765
|
+
...(userId ? { userId } : {}),
|
|
766
|
+
...(provider ? { provider } : {}),
|
|
767
|
+
};
|
|
768
|
+
const hasWorkspaceContext = Object.keys(workspaceContext).length > 0;
|
|
427
769
|
const response = await fetch(`${runtimeBase.replace(/\/$/, '')}/api/agent-runs`, {
|
|
428
770
|
method: 'POST',
|
|
429
|
-
headers: {
|
|
771
|
+
headers: {
|
|
772
|
+
'Content-Type': 'application/json',
|
|
773
|
+
...(askTraceCapability ? { 'X-DQL-Ask-Trace-Capability': askTraceCapability } : {}),
|
|
774
|
+
},
|
|
430
775
|
body: JSON.stringify({
|
|
431
776
|
question,
|
|
432
|
-
threadId,
|
|
777
|
+
...(threadId ? { threadId } : {}),
|
|
433
778
|
...(reasoningEffort ? { reasoningEffort } : {}),
|
|
434
779
|
...(analysisDepth ? { analysisDepth } : {}),
|
|
780
|
+
...(hasWorkspaceContext ? { workspaceContext } : {}),
|
|
435
781
|
}),
|
|
436
782
|
});
|
|
437
783
|
if (!response.ok)
|
|
438
784
|
throw new Error(`Runtime returned ${response.status}: ${await response.text()}`);
|
|
439
785
|
const payload = (await response.json());
|
|
440
|
-
if (!payload.run)
|
|
441
|
-
throw new Error('Runtime did not return
|
|
786
|
+
if (!payload.run?.id)
|
|
787
|
+
throw new Error('Runtime did not return a canonical agent run.');
|
|
788
|
+
const run = payload.run;
|
|
442
789
|
if (format === 'json') {
|
|
443
|
-
|
|
790
|
+
// Preserve the established top-level run shape while adding compact,
|
|
791
|
+
// content-free trace discoverability for scripts and support bundles.
|
|
792
|
+
console.log(JSON.stringify({
|
|
793
|
+
...run,
|
|
794
|
+
runId: run.id,
|
|
795
|
+
...(run.traceReference?.traceId ? { traceId: run.traceReference.traceId } : {}),
|
|
796
|
+
traceRecordingStatus: run.traceReference?.recordingStatus ?? 'unavailable',
|
|
797
|
+
}, null, 2));
|
|
444
798
|
return;
|
|
445
799
|
}
|
|
446
|
-
|
|
447
|
-
const badge = run.trustState === 'certified'
|
|
448
|
-
? '✓ Certified'
|
|
449
|
-
: run.trustState === 'grounded'
|
|
450
|
-
? '✓ Verified (grounded)'
|
|
451
|
-
: run.trustState === 'review_required'
|
|
452
|
-
? '! AI-generated · review required'
|
|
453
|
-
: run.trustState === 'blocked'
|
|
454
|
-
? '✕ Blocked'
|
|
455
|
-
: '· Reply';
|
|
456
|
-
console.log(`${badge}\n\n${(run.answer ?? run.summary ?? '').trim()}`);
|
|
457
|
-
console.log(`\nThread: ${threadId}`);
|
|
800
|
+
printCanonicalCliAskRun(run, threadId);
|
|
458
801
|
}
|
|
459
802
|
finally {
|
|
460
803
|
await close();
|
|
461
804
|
}
|
|
462
805
|
}
|
|
806
|
+
/**
|
|
807
|
+
* `dql agent ask --thread <id>` — POST the question to the runtime's
|
|
808
|
+
* `/api/agent-runs` with the threadId in the body. The server injects the
|
|
809
|
+
* thread's prior turns into the conversation context and records the completed
|
|
810
|
+
* run as the next turn, so follow-ups resolve "those"/"that product" correctly
|
|
811
|
+
* across CLI invocations (and across the notebook UI, which shares the store).
|
|
812
|
+
*/
|
|
813
|
+
async function runThreadAsk(question, threadId, flags) {
|
|
814
|
+
return runCanonicalCliAsk(question, threadId, flags);
|
|
815
|
+
}
|
|
463
816
|
/** `dql agent threads` — list server-persisted conversation threads. */
|
|
464
817
|
async function runThreads(flags) {
|
|
465
818
|
const projectRoot = findProjectRoot(process.cwd());
|
|
@@ -559,7 +912,7 @@ async function runEval(rest, flags) {
|
|
|
559
912
|
// started with DQL_EVAL_CASSETTE_DIR, since it owns its own provider.
|
|
560
913
|
const cassetteMode = flags.cassette;
|
|
561
914
|
const provider = cassetteMode === 'record' || cassetteMode === 'replay'
|
|
562
|
-
? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode)
|
|
915
|
+
? withCassette(rawProvider, new CassetteStore(cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')), cassetteMode, evalCassetteCanonicalizationV2(projectRoot))
|
|
563
916
|
: rawProvider;
|
|
564
917
|
const reasoningEffort = cliReasoningEffort(flags);
|
|
565
918
|
const requestedDepth = cliAnalysisDepth(flags);
|
|
@@ -633,6 +986,11 @@ async function runEval(rest, flags) {
|
|
|
633
986
|
const runtimeRun = via === 'runtime'
|
|
634
987
|
? await driveViaRuntime({ runtimeBase, question: testCase.question })
|
|
635
988
|
: undefined;
|
|
989
|
+
// Runtime mode is scored from the persisted AgentRun. The transport
|
|
990
|
+
// adapter intentionally has no AgentAnswer.contextPack, so borrowing the
|
|
991
|
+
// local preflight pack here would fabricate retrieval/route evidence for a
|
|
992
|
+
// different execution path.
|
|
993
|
+
const runtimeProjection = runtimeRun ? projectRuntimeRun(runtimeRun) : undefined;
|
|
636
994
|
const result = runtimeRun
|
|
637
995
|
? answerFromRuntimeRun(runtimeRun)
|
|
638
996
|
: await answer({
|
|
@@ -713,7 +1071,7 @@ async function runEval(rest, flags) {
|
|
|
713
1071
|
});
|
|
714
1072
|
},
|
|
715
1073
|
});
|
|
716
|
-
const evaluation = evaluateCase(testCase, result);
|
|
1074
|
+
const evaluation = evaluateCase(testCase, result, runtimeProjection);
|
|
717
1075
|
const durationMs = Date.now() - startedAt;
|
|
718
1076
|
const draftSaved = Boolean(result.draftBlock?.path ?? result.draftBlockId);
|
|
719
1077
|
const narration = narrationOutcomeForEval(runtimeRun?.narrationIntegrityReceipt);
|
|
@@ -740,7 +1098,7 @@ async function runEval(rest, flags) {
|
|
|
740
1098
|
executionMatched: evaluation.executionMatched,
|
|
741
1099
|
...(judgeVerdict ? { judgeScore: judgeVerdict.score, judgePass: judgeVerdict.pass } : {}),
|
|
742
1100
|
kind: result.kind,
|
|
743
|
-
route:
|
|
1101
|
+
route: runtimeProjection?.route ?? result.contextPack?.routeDecision.route,
|
|
744
1102
|
// Only the runtime driver can see the router's clarification options.
|
|
745
1103
|
// In-process runs leave this undefined, so a clarify there scores as a
|
|
746
1104
|
// dead end — the conservative reading, and another reason `--via runtime`
|
|
@@ -752,12 +1110,13 @@ async function runEval(rest, flags) {
|
|
|
752
1110
|
&& Boolean(runtimeRun.answer?.trim()),
|
|
753
1111
|
meaningResolved: Boolean(runtimeRun.routeDecision?.meaningResolution),
|
|
754
1112
|
} : {}),
|
|
755
|
-
|
|
1113
|
+
...(runtimeProjection?.observability ? { observability: runtimeProjection.observability } : {}),
|
|
1114
|
+
intent: runtimeRun?.routeDecision?.category ?? result.contextPack?.routeDecision.intent,
|
|
756
1115
|
reviewStatus: result.reviewStatus,
|
|
757
|
-
contextObjects: result.contextPack?.objects.length
|
|
1116
|
+
contextObjects: runtimeProjection?.retrievalCandidateCount ?? result.contextPack?.objects.length,
|
|
758
1117
|
followUp: Boolean(testCase.followUp),
|
|
759
1118
|
draftSaved,
|
|
760
|
-
toolCalls: result.evidence?.toolCalls?.length ?? 0,
|
|
1119
|
+
toolCalls: runtimeProjection?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0,
|
|
761
1120
|
expected: testCase.expected,
|
|
762
1121
|
validationCode: evaluation.validationCode,
|
|
763
1122
|
trace: buildEvalTrace({
|
|
@@ -766,6 +1125,7 @@ async function runEval(rest, flags) {
|
|
|
766
1125
|
evaluation,
|
|
767
1126
|
durationMs,
|
|
768
1127
|
draftSaved,
|
|
1128
|
+
runtime: runtimeProjection,
|
|
769
1129
|
}),
|
|
770
1130
|
});
|
|
771
1131
|
}
|
|
@@ -776,6 +1136,18 @@ async function runEval(rest, flags) {
|
|
|
776
1136
|
}
|
|
777
1137
|
const passed = results.filter((r) => r.passed).length;
|
|
778
1138
|
const metrics = computeEvalMetrics(results);
|
|
1139
|
+
// Runtime evals execute in a separate host, so its cassette directory is
|
|
1140
|
+
// supplied explicitly by the eval workflow for reporting. The summary does
|
|
1141
|
+
// not inspect prompts or provider credentials; it only classifies the
|
|
1142
|
+
// checked-in response provenance.
|
|
1143
|
+
const cassetteDirectory = via === 'runtime'
|
|
1144
|
+
? process.env.DQL_EVAL_CASSETTE_DIR
|
|
1145
|
+
: cassetteMode === 'record' || cassetteMode === 'replay'
|
|
1146
|
+
? cassetteDirFor(projectRoot, rest[0] ?? 'agent-evals')
|
|
1147
|
+
: undefined;
|
|
1148
|
+
const cassetteEvidence = cassetteDirectory
|
|
1149
|
+
? cassetteEvidenceSummary(new CassetteStore(cassetteDirectory))
|
|
1150
|
+
: undefined;
|
|
779
1151
|
const thresholds = {
|
|
780
1152
|
minToolRequirement: flags.minToolRequirement ?? null,
|
|
781
1153
|
minExecutionMatch: flags.minExecutionMatch ?? null,
|
|
@@ -788,7 +1160,15 @@ async function runEval(rest, flags) {
|
|
|
788
1160
|
const thresholdsPassed = agentEvalThresholdsPass(metrics, thresholds);
|
|
789
1161
|
const ok = passed === results.length && thresholdsPassed;
|
|
790
1162
|
if (flags.format === 'json') {
|
|
791
|
-
console.log(JSON.stringify({
|
|
1163
|
+
console.log(JSON.stringify({
|
|
1164
|
+
ok,
|
|
1165
|
+
passed,
|
|
1166
|
+
total: results.length,
|
|
1167
|
+
thresholds,
|
|
1168
|
+
metrics,
|
|
1169
|
+
...(cassetteEvidence ? { cassetteEvidence } : {}),
|
|
1170
|
+
results,
|
|
1171
|
+
}, null, 2));
|
|
792
1172
|
if (!ok)
|
|
793
1173
|
process.exitCode = 1;
|
|
794
1174
|
return;
|
|
@@ -802,6 +1182,14 @@ async function runEval(rest, flags) {
|
|
|
802
1182
|
console.log(`Certified hit rate: ${formatRate(metrics.certified_hit_rate)}`);
|
|
803
1183
|
console.log(`Generated follow-up pass rate: ${formatRate(metrics.generated_followup_pass_rate)}`);
|
|
804
1184
|
console.log(`Safe refusal rate: ${formatRate(metrics.safe_refusal_rate)}`);
|
|
1185
|
+
if (cassetteEvidence) {
|
|
1186
|
+
console.log(`Cassette replay entries: ${cassetteEvidence.totalEntries} `
|
|
1187
|
+
+ `(${cassetteEvidence.migratedLegacyDeterministicFixtureEntries} migrated legacy deterministic fixture, `
|
|
1188
|
+
+ `${cassetteEvidence.syntheticDeterministicOrchestrationFixtureEntries} synthetic deterministic orchestration fixture).`);
|
|
1189
|
+
console.log(cassetteEvidence.realProviderQualityEligible
|
|
1190
|
+
? 'Real-provider quality evidence: eligible.'
|
|
1191
|
+
: `Real-provider quality evidence: excluded (${cassetteEvidence.realProviderQualityExclusionReasons.join(', ')}).`);
|
|
1192
|
+
}
|
|
805
1193
|
console.log(`False refusal rate: ${formatRate(metrics.false_refusal_rate)} (${metrics.false_refusal_count}/${metrics.answerable_case_count} answerable cases refused)`);
|
|
806
1194
|
console.log(`Clarification rate: ${formatRate(metrics.clarification_rate)} (answerable cases asked instead of answered)`);
|
|
807
1195
|
if (metrics.meaning_resolved_rate !== null && metrics.meaning_resolved_rate < 1) {
|
|
@@ -851,7 +1239,7 @@ function previewGeneratedDraftPath(projectRoot, domain, slug) {
|
|
|
851
1239
|
}
|
|
852
1240
|
return `blocks/_drafts/${slug}.dql`;
|
|
853
1241
|
}
|
|
854
|
-
function evaluateCase(testCase, result) {
|
|
1242
|
+
function evaluateCase(testCase, result, runtime) {
|
|
855
1243
|
const expected = testCase.expected;
|
|
856
1244
|
if (!expected)
|
|
857
1245
|
return { failures: [] };
|
|
@@ -892,12 +1280,30 @@ function evaluateCase(testCase, result) {
|
|
|
892
1280
|
failures.push(`certification expected ${expected.certification}, got ${result.certification}`);
|
|
893
1281
|
if (expected.reviewStatus && result.reviewStatus !== expected.reviewStatus)
|
|
894
1282
|
failures.push(`reviewStatus expected ${expected.reviewStatus}, got ${result.reviewStatus}`);
|
|
895
|
-
|
|
896
|
-
|
|
1283
|
+
const observedRoute = runtime?.route ?? result.contextPack?.routeDecision.route;
|
|
1284
|
+
if (expected.route && observedRoute !== expected.route)
|
|
1285
|
+
failures.push(`route expected ${expected.route}, got ${observedRoute ?? 'none'}`);
|
|
897
1286
|
if (expected.intent && result.contextPack?.routeDecision.intent !== expected.intent)
|
|
898
1287
|
failures.push(`intent expected ${expected.intent}, got ${result.contextPack?.routeDecision.intent ?? 'none'}`);
|
|
899
|
-
|
|
1288
|
+
// `modeling_gap` is a broad terminal kind. A relationship expectation is
|
|
1289
|
+
// satisfied only by the router's persisted relationship-specific witness;
|
|
1290
|
+
// otherwise a missing metric/dimension tuple would be misreported as a
|
|
1291
|
+
// relationship repair opportunity.
|
|
1292
|
+
const runtimeReportsMissingContext = expected.missingContextKind === 'relationship'
|
|
1293
|
+
? runtime?.terminalOutcome?.gap?.code === 'MISSING_RELATIONSHIP'
|
|
1294
|
+
: runtime?.terminalOutcome?.kind === 'modeling_gap'
|
|
1295
|
+
&& expected.missingContextKind === 'modeling_gap';
|
|
1296
|
+
if (expected.missingContextKind
|
|
1297
|
+
&& !runtimeReportsMissingContext
|
|
1298
|
+
&& !result.contextPack?.missingContext.some((item) => item.kind === expected.missingContextKind)) {
|
|
900
1299
|
failures.push(`missing context kind ${expected.missingContextKind} was not reported`);
|
|
1300
|
+
}
|
|
1301
|
+
if (expected.terminalOutcomeKind && runtime?.terminalOutcome?.kind !== expected.terminalOutcomeKind) {
|
|
1302
|
+
failures.push(`terminal outcome expected ${expected.terminalOutcomeKind}, got ${runtime?.terminalOutcome?.kind ?? 'none'}`);
|
|
1303
|
+
}
|
|
1304
|
+
if (expected.traceRecordingStatus && runtime?.observability?.recordingStatus !== expected.traceRecordingStatus) {
|
|
1305
|
+
failures.push(`trace recording status expected ${expected.traceRecordingStatus}, got ${runtime?.observability?.recordingStatus ?? 'unavailable'}`);
|
|
1306
|
+
}
|
|
901
1307
|
for (const token of stringList(expected.sqlContains)) {
|
|
902
1308
|
if (!result.proposedSql?.toLowerCase().includes(token.toLowerCase()))
|
|
903
1309
|
failures.push(`SQL did not contain "${token}"`);
|
|
@@ -918,7 +1324,7 @@ function evaluateCase(testCase, result) {
|
|
|
918
1324
|
failures.push(`draftSaved expected ${expected.draftSaved}, got ${saved}`);
|
|
919
1325
|
}
|
|
920
1326
|
if (typeof expected.minToolCalls === 'number') {
|
|
921
|
-
const actualToolCalls = result.evidence?.toolCalls?.length ?? 0;
|
|
1327
|
+
const actualToolCalls = runtime?.toolCallCount ?? result.evidence?.toolCalls?.length ?? 0;
|
|
922
1328
|
if (actualToolCalls < expected.minToolCalls) {
|
|
923
1329
|
failures.push(`toolCalls expected at least ${expected.minToolCalls}, got ${actualToolCalls}`);
|
|
924
1330
|
}
|
|
@@ -957,7 +1363,7 @@ export function evalCaseIsAnswerable(expected) {
|
|
|
957
1363
|
return false;
|
|
958
1364
|
if (expected.sourceTier === 'no_answer')
|
|
959
1365
|
return false;
|
|
960
|
-
if (expected.route === 'clarify')
|
|
1366
|
+
if (expected.route === 'clarify' || expected.route === 'blocked')
|
|
961
1367
|
return false;
|
|
962
1368
|
if (Object.keys(expected).length === 0)
|
|
963
1369
|
return undefined;
|
|
@@ -1084,7 +1490,12 @@ function computeEvalMetrics(results) {
|
|
|
1084
1490
|
draft_saved_count: results.filter((result) => result.draftSaved).length,
|
|
1085
1491
|
tool_observed_case_count: results.filter((result) => result.toolCalls > 0).length,
|
|
1086
1492
|
avg_tool_calls: average(toolCallCounts),
|
|
1087
|
-
|
|
1493
|
+
// Runtime runs only report this when the persisted router recorded an
|
|
1494
|
+
// explicit retrieval count. Treat absent evidence as unknown, not as an
|
|
1495
|
+
// invented empty context pack.
|
|
1496
|
+
avg_context_objects: average(results
|
|
1497
|
+
.map((result) => result.contextObjects)
|
|
1498
|
+
.filter((count) => typeof count === 'number')),
|
|
1088
1499
|
avg_execution_ms: executionTimes.length ? average(executionTimes) : null,
|
|
1089
1500
|
};
|
|
1090
1501
|
}
|
|
@@ -1106,12 +1517,13 @@ function agentEvalThresholdsPass(metrics, thresholds) {
|
|
|
1106
1517
|
|| metrics.wrong_certified_count <= thresholds.maxWrongCertified);
|
|
1107
1518
|
}
|
|
1108
1519
|
function buildEvalTrace(input) {
|
|
1109
|
-
const { testCase, result, evaluation, durationMs, draftSaved } = input;
|
|
1520
|
+
const { testCase, result, evaluation, durationMs, draftSaved, runtime } = input;
|
|
1110
1521
|
const routeDecision = result.contextPack?.routeDecision;
|
|
1111
1522
|
const selectedRelations = result.contextPack?.retrievalDiagnostics.selectedRelations ?? [];
|
|
1112
1523
|
const allowedRelations = result.contextPack?.allowedSqlContext?.relations ?? [];
|
|
1113
1524
|
const followUp = testCase.followUp;
|
|
1114
1525
|
const toolCalls = result.evidence?.toolCalls ?? [];
|
|
1526
|
+
const observedToolCallCount = runtime?.toolCallCount ?? toolCalls.length;
|
|
1115
1527
|
const routeEvidence = result.evidence?.route ?? [];
|
|
1116
1528
|
const executionStatus = result.executionError
|
|
1117
1529
|
? 'failed'
|
|
@@ -1127,33 +1539,44 @@ function buildEvalTrace(input) {
|
|
|
1127
1539
|
const rowsExpected = testCase.expected?.rows !== undefined;
|
|
1128
1540
|
const expectedMinToolCalls = testCase.expected?.minToolCalls;
|
|
1129
1541
|
const toolStatus = typeof expectedMinToolCalls === 'number'
|
|
1130
|
-
?
|
|
1131
|
-
:
|
|
1542
|
+
? observedToolCallCount >= expectedMinToolCalls ? 'passed' : 'failed'
|
|
1543
|
+
: observedToolCallCount > 0 ? 'passed' : routeEvidence.length > 0 ? 'info' : 'not_run';
|
|
1132
1544
|
const toolMessage = typeof expectedMinToolCalls === 'number'
|
|
1133
|
-
?
|
|
1134
|
-
? `Observed ${
|
|
1135
|
-
: `Observed ${
|
|
1136
|
-
:
|
|
1137
|
-
? `Observed ${
|
|
1545
|
+
? observedToolCallCount >= expectedMinToolCalls
|
|
1546
|
+
? `Observed ${observedToolCallCount} provider tool call(s), meeting the minimum of ${expectedMinToolCalls}.`
|
|
1547
|
+
: `Observed ${observedToolCallCount} provider tool call(s), below the minimum of ${expectedMinToolCalls}.`
|
|
1548
|
+
: observedToolCallCount > 0
|
|
1549
|
+
? `Observed ${observedToolCallCount} provider tool call(s).`
|
|
1138
1550
|
: routeEvidence.length > 0
|
|
1139
1551
|
? `Captured ${routeEvidence.length} deterministic route evidence step(s).`
|
|
1140
1552
|
: 'No provider tool calls were observed for this answer.';
|
|
1141
1553
|
return [
|
|
1142
1554
|
{
|
|
1143
1555
|
stage: 'context',
|
|
1144
|
-
status:
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1556
|
+
status: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
1557
|
+
? 'passed'
|
|
1558
|
+
: result.contextPack ? 'passed' : 'not_run',
|
|
1559
|
+
message: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
1560
|
+
? `Persisted route evidence recorded ${runtime.retrievalCandidateCount ?? 'an unspecified number of'} retrieved candidate(s).`
|
|
1561
|
+
: result.contextPack
|
|
1562
|
+
? `Context pack ${result.contextPack.id} selected ${result.contextPack.objects.length} object(s).`
|
|
1563
|
+
: 'No context pack was attached to the answer.',
|
|
1564
|
+
payload: runtime && (runtime.retrievalCandidateCount !== undefined || runtime.sourceCoverage?.length || runtime.terminalOutcome)
|
|
1149
1565
|
? {
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
missingContext: result.contextPack.missingContext,
|
|
1566
|
+
evidenceSource: 'persisted_agent_run',
|
|
1567
|
+
retrievalCandidateCount: runtime.retrievalCandidateCount,
|
|
1568
|
+
sourceCoverage: runtime.sourceCoverage,
|
|
1569
|
+
terminalOutcome: runtime.terminalOutcome,
|
|
1155
1570
|
}
|
|
1156
|
-
:
|
|
1571
|
+
: result.contextPack
|
|
1572
|
+
? {
|
|
1573
|
+
contextPackId: result.contextPack.id,
|
|
1574
|
+
selectedObjectCount: result.contextPack.objects.length,
|
|
1575
|
+
allowedRelationCount: allowedRelations.length,
|
|
1576
|
+
selectedRelations: selectedRelations.slice(0, 12).map((relation) => relation.relation),
|
|
1577
|
+
missingContext: result.contextPack.missingContext,
|
|
1578
|
+
}
|
|
1579
|
+
: undefined,
|
|
1157
1580
|
},
|
|
1158
1581
|
{
|
|
1159
1582
|
stage: 'rewrite',
|
|
@@ -1165,29 +1588,40 @@ function buildEvalTrace(input) {
|
|
|
1165
1588
|
},
|
|
1166
1589
|
{
|
|
1167
1590
|
stage: 'lane',
|
|
1168
|
-
status: routeDecision ? 'passed' : 'not_run',
|
|
1169
|
-
message:
|
|
1170
|
-
? `
|
|
1171
|
-
:
|
|
1172
|
-
|
|
1591
|
+
status: runtime ? 'passed' : routeDecision ? 'passed' : 'not_run',
|
|
1592
|
+
message: runtime
|
|
1593
|
+
? `Persisted engine route ${runtime.runRoute}${runtime.route ? ` evaluated as ${runtime.route}` : ''}.`
|
|
1594
|
+
: routeDecision
|
|
1595
|
+
? `Lane ${routeDecision.route} / ${routeDecision.intent}.`
|
|
1596
|
+
: 'No lane decision was attached to the answer.',
|
|
1597
|
+
payload: runtime
|
|
1173
1598
|
? {
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
exactObjectKey: routeDecision.exactObjectKey,
|
|
1599
|
+
engineRoute: runtime.runRoute,
|
|
1600
|
+
evalRoute: runtime.route,
|
|
1601
|
+
status: runtime.status,
|
|
1602
|
+
trustState: runtime.trustState,
|
|
1603
|
+
terminalOutcome: runtime.terminalOutcome,
|
|
1180
1604
|
}
|
|
1181
|
-
:
|
|
1605
|
+
: routeDecision
|
|
1606
|
+
? {
|
|
1607
|
+
route: routeDecision.route,
|
|
1608
|
+
intent: routeDecision.intent,
|
|
1609
|
+
reason: routeDecision.reason,
|
|
1610
|
+
trustLabel: routeDecision.trustLabel,
|
|
1611
|
+
reviewStatus: routeDecision.reviewStatus,
|
|
1612
|
+
exactObjectKey: routeDecision.exactObjectKey,
|
|
1613
|
+
}
|
|
1614
|
+
: undefined,
|
|
1182
1615
|
},
|
|
1183
1616
|
{
|
|
1184
1617
|
stage: 'tools',
|
|
1185
1618
|
status: toolStatus,
|
|
1186
1619
|
message: toolMessage,
|
|
1187
1620
|
payload: {
|
|
1188
|
-
observedToolCalls:
|
|
1621
|
+
observedToolCalls: observedToolCallCount,
|
|
1189
1622
|
expectedMinToolCalls,
|
|
1190
|
-
|
|
1623
|
+
...(runtime ? { evidenceSource: 'persisted_agent_run.telemetry' } : {}),
|
|
1624
|
+
providerToolCalls: runtime ? [] : toolCalls.slice(0, 12).map((call) => ({
|
|
1191
1625
|
order: call.order,
|
|
1192
1626
|
name: call.name,
|
|
1193
1627
|
status: call.status,
|
|
@@ -1261,6 +1695,22 @@ function buildEvalTrace(input) {
|
|
|
1261
1695
|
promoteCommand: result.promoteCommand,
|
|
1262
1696
|
},
|
|
1263
1697
|
},
|
|
1698
|
+
{
|
|
1699
|
+
stage: 'observability',
|
|
1700
|
+
status: runtime?.observability?.recordingStatus === 'complete'
|
|
1701
|
+
? 'passed'
|
|
1702
|
+
: runtime?.observability ? 'info' : 'not_run',
|
|
1703
|
+
message: runtime?.observability
|
|
1704
|
+
? `Local Ask trace recording ${runtime.observability.recordingStatus}.`
|
|
1705
|
+
: 'No runtime trace receipt was attached to this evaluation run.',
|
|
1706
|
+
payload: runtime?.observability
|
|
1707
|
+
? {
|
|
1708
|
+
recordingStatus: runtime.observability.recordingStatus,
|
|
1709
|
+
storeSchemaVersion: runtime.observability.storeSchemaVersion,
|
|
1710
|
+
...(runtime.observability.traceFingerprint ? { traceFingerprint: runtime.observability.traceFingerprint } : {}),
|
|
1711
|
+
}
|
|
1712
|
+
: undefined,
|
|
1713
|
+
},
|
|
1264
1714
|
{
|
|
1265
1715
|
stage: 'scoring',
|
|
1266
1716
|
status: evaluation.failures.length === 0 ? 'passed' : 'failed',
|