@duckcodeailabs/dql-cli 1.13.5 → 1.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/args.d.ts +17 -0
  2. package/dist/args.d.ts.map +1 -1
  3. package/dist/args.js +30 -0
  4. package/dist/args.js.map +1 -1
  5. package/dist/assets/dql-notebook/assets/{AgentLogPage-BzQOKjyV.js → AgentLogPage-BPz-UWFh.js} +1 -1
  6. package/dist/assets/dql-notebook/assets/{AiBuildDialog-CTxha499.js → AiBuildDialog-BEl53WA_.js} +1 -1
  7. package/dist/assets/dql-notebook/assets/{AiBuildResult--MuD_I6g.js → AiBuildResult-B4yfGTTZ.js} +1 -1
  8. package/dist/assets/dql-notebook/assets/{AiSidePanel-DcI4PMJ1.js → AiSidePanel-CSZAAvuD.js} +1 -1
  9. package/dist/assets/dql-notebook/assets/{AnalyticsHome-Dq54vjNz.js → AnalyticsHome-D5P6Ujwi.js} +1 -1
  10. package/dist/assets/dql-notebook/assets/{AppsView-0SlwWYex.js → AppsView-DQOwU9Cg.js} +4 -4
  11. package/dist/assets/dql-notebook/assets/{BlockStudio-CxxXZD3M.js → BlockStudio-B4ap0GdY.js} +1 -1
  12. package/dist/assets/dql-notebook/assets/{BusinessArtifactView-DxdAmeUN.js → BusinessArtifactView-BIfNI-0S.js} +1 -1
  13. package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-y81gFg_b.js → DbtFirstModelingPage-D72byb2g.js} +1 -1
  14. package/dist/assets/dql-notebook/assets/{GitPage-GQtcwncb.js → GitPage-lQb1uXH2.js} +1 -1
  15. package/dist/assets/dql-notebook/assets/{GlobalAiRail-DM4wkxR_.js → GlobalAiRail-CVECf6Xj.js} +1 -1
  16. package/dist/assets/dql-notebook/assets/{GovernedContextPage-B8Ft6JES.js → GovernedContextPage-trOyMCY6.js} +1 -1
  17. package/dist/assets/dql-notebook/assets/{HelpDocsPage-DnD4nMuF.js → HelpDocsPage-D8hLS5lE.js} +1 -1
  18. package/dist/assets/dql-notebook/assets/{HomePage-Ehb0ITj-.js → HomePage-eIkfBIep.js} +1 -1
  19. package/dist/assets/dql-notebook/assets/{LineageDAG-Lvjc2AQX.js → LineageDAG-CSqcbDrE.js} +1 -1
  20. package/dist/assets/dql-notebook/assets/{LineageDetailView-DX32pfbp.js → LineageDetailView-DJZZjZu-.js} +1 -1
  21. package/dist/assets/dql-notebook/assets/{LineageDrawer-CFq3jPfk.js → LineageDrawer-BcIipIc3.js} +1 -1
  22. package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-CVHXuHls.js → LineagePathBreadcrumb-CviIf8PN.js} +1 -1
  23. package/dist/assets/dql-notebook/assets/{MiniLineageGraph-BKaeT5Kt.js → MiniLineageGraph-vH_MY_Ju.js} +1 -1
  24. package/dist/assets/dql-notebook/assets/{NewBlockModal-DI-JDzFH.js → NewBlockModal-DbCQg-pj.js} +1 -1
  25. package/dist/assets/dql-notebook/assets/{NewNotebookModal-Csw6eF74.js → NewNotebookModal-5Vp6XiuK.js} +1 -1
  26. package/dist/assets/dql-notebook/assets/{NotebookEditor-FEqs3789.js → NotebookEditor-DjqJS44s.js} +1 -1
  27. package/dist/assets/dql-notebook/assets/{ReadinessPage-Bk0S_MEA.js → ReadinessPage-BHGrC5ho.js} +1 -1
  28. package/dist/assets/dql-notebook/assets/{SetupOnboarding-DyLaPFGn.js → SetupOnboarding-BfS9Tdx-.js} +1 -1
  29. package/dist/assets/dql-notebook/assets/{SkillsPage-DLHyvhci.js → SkillsPage-BFH21wSj.js} +1 -1
  30. package/dist/assets/dql-notebook/assets/{TrustBadge-CIDLj2A6.js → TrustBadge-zm6g_SxZ.js} +1 -1
  31. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel--oxmjlgr.js +88 -0
  32. package/dist/assets/dql-notebook/assets/{answer-to-notebook-CFEJHLvs.js → answer-to-notebook-DPhxIEzF.js} +1 -1
  33. package/dist/assets/dql-notebook/assets/{arrow-left-B-Zdcyvm.js → arrow-left-DNEb86Xc.js} +1 -1
  34. package/dist/assets/dql-notebook/assets/{arrow-right-Cn8TM7cp.js → arrow-right-DpPWbwaD.js} +1 -1
  35. package/dist/assets/dql-notebook/assets/{book-open-text-Bko2NNUs.js → book-open-text-D7s5jo4X.js} +1 -1
  36. package/dist/assets/dql-notebook/assets/{circle-x-AWJAmUBB.js → circle-x-Db3dXKg7.js} +1 -1
  37. package/dist/assets/dql-notebook/assets/{dagre.esm-Cl_ucrRh.js → dagre.esm-C7pppQ1a.js} +1 -1
  38. package/dist/assets/dql-notebook/assets/{external-link-DN57tb5f.js → external-link-BxwXitO_.js} +1 -1
  39. package/dist/assets/dql-notebook/assets/{grip-vertical-BOWwFFva.js → grip-vertical-Dt4lkWRi.js} +1 -1
  40. package/dist/assets/dql-notebook/assets/{index-Ck-wqvV2.js → index-DKo-bwNw.js} +4 -4
  41. package/dist/assets/dql-notebook/assets/{link-2-xUhbfBs8.js → link-2-Dfo2P6wi.js} +1 -1
  42. package/dist/assets/dql-notebook/assets/{list-tree-BDWcBr67.js → list-tree-DiTmIWAL.js} +1 -1
  43. package/dist/assets/dql-notebook/assets/{minimize-2-CEeNoMTS.js → minimize-2-CMTAkPzL.js} +1 -1
  44. package/dist/assets/dql-notebook/assets/{panel-right-open-DnSeUvOY.js → panel-right-open-DwYr7FW4.js} +1 -1
  45. package/dist/assets/dql-notebook/assets/{play-EoDj8-fa.js → play-BXhHYQ4x.js} +1 -1
  46. package/dist/assets/dql-notebook/assets/{rotate-ccw-BAMxWT9A.js → rotate-ccw-BNi6F8pl.js} +1 -1
  47. package/dist/assets/dql-notebook/assets/{semantic-fields-BwLOs1kl.js → semantic-fields-CNOGysAy.js} +1 -1
  48. package/dist/assets/dql-notebook/assets/{sliders-horizontal-BYkm8SWW.js → sliders-horizontal-Ec5MUMUW.js} +1 -1
  49. package/dist/assets/dql-notebook/assets/{star-sVexqNCs.js → star-B9leDkp_.js} +1 -1
  50. package/dist/assets/dql-notebook/assets/{triangle-alert-D2njv4o4.js → triangle-alert-BefTYCzx.js} +1 -1
  51. package/dist/assets/dql-notebook/assets/{upload-CXTMxIA8.js → upload-SPiOM2tQ.js} +1 -1
  52. package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-CPDeaAgL.js → usePersistedAgentThreadId-C4foXeiQ.js} +4 -4
  53. package/dist/assets/dql-notebook/assets/{user-round-zVsI_uud.js → user-round-comGmyw-.js} +1 -1
  54. package/dist/assets/dql-notebook/assets/{wand-sparkles-BMIwIzXT.js → wand-sparkles-BffR4dF8.js} +1 -1
  55. package/dist/assets/dql-notebook/assets/{workflow-Dj9y1sNk.js → workflow-ChmPTEzH.js} +1 -1
  56. package/dist/assets/dql-notebook/assets/{wrench-DPQi6zrs.js → wrench-DovaG_ze.js} +1 -1
  57. package/dist/assets/dql-notebook/assets/{x-XhhbtinL.js → x-nRx91AgW.js} +1 -1
  58. package/dist/assets/dql-notebook/index.html +1 -1
  59. package/dist/commands/agent-eval-cassette.d.ts +73 -0
  60. package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
  61. package/dist/commands/agent-eval-cassette.js +170 -0
  62. package/dist/commands/agent-eval-cassette.js.map +1 -0
  63. package/dist/commands/agent-eval-runtime.d.ts +97 -0
  64. package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
  65. package/dist/commands/agent-eval-runtime.js +155 -0
  66. package/dist/commands/agent-eval-runtime.js.map +1 -0
  67. package/dist/commands/agent.d.ts +115 -1
  68. package/dist/commands/agent.d.ts.map +1 -1
  69. package/dist/commands/agent.js +289 -62
  70. package/dist/commands/agent.js.map +1 -1
  71. package/dist/commands/eval.d.ts +15 -1
  72. package/dist/commands/eval.d.ts.map +1 -1
  73. package/dist/commands/eval.js +32 -3
  74. package/dist/commands/eval.js.map +1 -1
  75. package/dist/index.js +5 -1
  76. package/dist/index.js.map +1 -1
  77. package/dist/llm/analyst-loop-tools.d.ts +19 -0
  78. package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
  79. package/dist/llm/analyst-loop-tools.js +56 -0
  80. package/dist/llm/analyst-loop-tools.js.map +1 -0
  81. package/dist/llm/providers/dql-agent-provider.d.ts +24 -1
  82. package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
  83. package/dist/llm/providers/dql-agent-provider.js +343 -10
  84. package/dist/llm/providers/dql-agent-provider.js.map +1 -1
  85. package/dist/llm/types.d.ts +19 -1
  86. package/dist/llm/types.d.ts.map +1 -1
  87. package/dist/local-runtime.d.ts +132 -2
  88. package/dist/local-runtime.d.ts.map +1 -1
  89. package/dist/local-runtime.js +1120 -184
  90. package/dist/local-runtime.js.map +1 -1
  91. package/dist/package.json +10 -10
  92. package/package.json +10 -10
  93. package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-C0oKTU6G.js +0 -88
@@ -23,9 +23,10 @@ import { getRunner as getLLMRunner } from './llm/index.js';
23
23
  import { rethrowIfCancelled } from './llm/cancellation.js';
24
24
  import { fetchLatestPublishedDqlVersion, resolveDqlRuntimeVersionStatus } from './version-status.js';
25
25
  import { resolveRetrievalHealthStatus } from './retrieval-health.js';
26
- import { createDqlAgentProviderRunner, resolveAgentFollowUpContext } from './llm/providers/dql-agent-provider.js';
26
+ import { applyFinding, createResearchState, narrationMaxTokensForFacts, nextHypothesis, rerankCandidates, synthesizeResearchNarrative, AgenticExecutionCapabilityGate, mintFinalSqlAuthorization, verifyAgenticSqlExecutionCapability, qualifyAuthorizationReferences, validateSqlAgainstLocalContext as validateAuthorizedSqlReferences, verifyFinalSql, } from '@duckcodeailabs/dql-agent';
27
+ import { applyEvalCassette, createDqlAgentProviderRunner, createGovernedTextProvider, resolveAgentFollowUpContext } from './llm/providers/dql-agent-provider.js';
27
28
  import { listRemoteMcpSettings, saveRemoteMcpSettings } from './llm/mcp-config.js';
28
- import { ClaudeProvider, ConversationStore, advanceThreadState, buildConversationSnapshot, conversationHistoryFromContext, recallRelevantTurns, renderConversationEnvelopeForPrompt, GeminiProvider, MemoryStore, OllamaProvider, OpenAIProvider, buildBlockBusinessFingerprint, buildBlockSqlFingerprints, buildAnalysisQuestionPlan, composeSemanticQueryForQuestion, aggregationIntegrityIssuesForSql, buildAggregationSafetyProof, buildLocalContextPack, applyContextPackCompatibility, toAgentRetrievalEvidence, prepareConversationPath, defaultMemoryPath, ensureDefaultMemoryFiles, ensureAgentProjectReady, isAgentProjectIndexReady, currentMetadataFingerprint, ensureMetadataCatalogFresh, readIndexedDomainKnowledge, readIndexedKnowledge360, compactSemanticRuntimeFailure, classifyAnalyticalFailure, normalizeWarehouseSqlFailure, parseProposal, propose, proposePlan, recordGovernedCorrection, HintStore, defaultHintIndexPath, ensureHintIndexFresh, listHintsFromGit, getHintEvaluationFromGit, getCorrectionTraceFromGit, inspectGovernedHint, editGovernedHintCandidate, reopenGovernedHint, retireHint, supersedeHint, hintsConflict, mineJoinPatterns, reviewGovernedHint, AgentRunEngine, SqliteAgentRunStore, defaultAgentRunGates, createLlmAgentRunPlanner, createHybridRouter, computeResultStats, buildDeterministicDashboardStory, synthesizeAnswer, streamOrGenerate, narrateResult, buildProposePreview, buildFromPrompt, internalRelationIdsInSql, defaultAgentRunStorePath, defaultAgentRunSqlitePath, resolveLocalOwner, resolveProposeConfig, recordQueryRun, recordRuntimeSchemaSnapshot, latestRuntimeSchemaSnapshotForProject, loadSkills, migrateLegacySkills, configuredSkillsPath, skillsDir, draftDomainSkillBootstrap, buildDomainSkillBootstrapPrompt, mergeDomainSkillBootstrapEnrichment, writeSkill, previewSkillChange, buildContextAuthoringProposal, contextAuthoringDependencyClosure, FileContextAuthoringProposalStore, deleteSkill, deriveGeneratedDraftSlug, deriveAnalyticalRepair, reindexProject, invalidateAgentProjectState, recordAgentRuntimeVersion, resolveDomainContextEnvelope, projectEmbeddingProvider, isHashedEmbeddingProvider, clearProjectEmbeddingCache, upgradeVectorIndexForProject, openMetadataCatalog, defaultKgPath, planAppFromPrompt, KGStore, planResearch, loadSemanticMetrics, cascadeTraceToEvidenceRouteSteps, createCascadeAnswerResult, createCascadeTrace, routeReasoningEffort, createAgentRunBudget, routeForCascadeAnswerTier, clampReasoningEffort, bumpReasoningEffort, resolveThinkingMode, coerceThinkingMode, upsertGeneratedDqlArtifactDraft, loadAgentSemanticLayer, isTrustedConversationTurn, resolveInternalRelationIds, analyticalError, tagAnalyticalError, withAnalyticalErrorOrigin, withAnalyticalErrorOriginSync, assertProviderPayloadAllowed, createProviderDispatchEgressReceipt, prepareProviderWireEnvelopeForDispatch, markProviderMetadataArray, createProviderEgressReceipt, redactProviderResultRows, composeVerifiedAnalyticalNarrative, DEFAULT_ASK_ROW_EGRESS_POLICY, ZERO_ROW_EGRESS_POLICY, resolveProviderResultRowEgressPolicy, } from '@duckcodeailabs/dql-agent';
29
+ import { composeBusinessExplanation, ClaudeProvider, ConversationStore, advanceThreadState, buildConversationSnapshot, conversationHistoryFromContext, recallRelevantTurns, renderConversationEnvelopeForPrompt, GeminiProvider, MemoryStore, OllamaProvider, OpenAIProvider, buildBlockBusinessFingerprint, buildBlockSqlFingerprints, buildAnalysisQuestionPlan, composeSemanticQueryForQuestion, aggregationIntegrityIssuesForSql, buildAggregationSafetyProof, buildLocalContextPack, applyContextPackCompatibility, toAgentRetrievalEvidence, prepareConversationPath, defaultMemoryPath, ensureDefaultMemoryFiles, ensureAgentProjectReady, isAgentProjectIndexReady, currentMetadataFingerprint, ensureMetadataCatalogFresh, readIndexedDomainKnowledge, readIndexedKnowledge360, compactSemanticRuntimeFailure, classifyAnalyticalFailure, normalizeWarehouseSqlFailure, parseProposal, propose, proposePlan, recordGovernedCorrection, HintStore, defaultHintIndexPath, ensureHintIndexFresh, listHintsFromGit, getHintEvaluationFromGit, getCorrectionTraceFromGit, inspectGovernedHint, editGovernedHintCandidate, reopenGovernedHint, retireHint, supersedeHint, hintsConflict, mineJoinPatterns, reviewGovernedHint, AgentRunEngine, SqliteAgentRunStore, defaultAgentRunGates, createLlmAgentRunPlanner, createHybridRouter, computeResultStats, buildDeterministicDashboardStory, synthesizeAnswer, streamOrGenerate, narrateResult, buildProposePreview, buildFromPrompt, internalRelationIdsInSql, defaultAgentRunStorePath, defaultAgentRunSqlitePath, resolveLocalOwner, resolveProposeConfig, recordQueryRun, recordRuntimeSchemaSnapshot, latestRuntimeSchemaSnapshotForProject, loadSkills, migrateLegacySkills, configuredSkillsPath, skillsDir, draftDomainSkillBootstrap, buildDomainSkillBootstrapPrompt, mergeDomainSkillBootstrapEnrichment, writeSkill, previewSkillChange, buildContextAuthoringProposal, contextAuthoringDependencyClosure, FileContextAuthoringProposalStore, deleteSkill, deriveGeneratedDraftSlug, deriveAnalyticalRepair, reindexProject, invalidateAgentProjectState, recordAgentRuntimeVersion, resolveDomainContextEnvelope, projectEmbeddingProvider, isHashedEmbeddingProvider, clearProjectEmbeddingCache, upgradeVectorIndexForProject, openMetadataCatalog, defaultKgPath, planAppFromPrompt, KGStore, planResearch, loadSemanticMetrics, cascadeTraceToEvidenceRouteSteps, createCascadeAnswerResult, createCascadeTrace, routeReasoningEffort, createAgentRunBudget, isProbeSafeColumn, deadlineScale, routeForCascadeAnswerTier, clampReasoningEffort, bumpReasoningEffort, resolveThinkingMode, coerceThinkingMode, upsertGeneratedDqlArtifactDraft, loadAgentSemanticLayer, isTrustedConversationTurn, resolveInternalRelationIds, analyticalError, tagAnalyticalError, withAnalyticalErrorOrigin, withAnalyticalErrorOriginSync, assertProviderPayloadAllowed, createProviderDispatchEgressReceipt, prepareProviderWireEnvelopeForDispatch, markProviderMetadataArray, createProviderEgressReceipt, redactProviderResultRows, composeVerifiedAnalyticalNarrative, buildCoverageGap, capResearchBranches, buildResearchEvidenceLedger, buildAnalyticalTurnPlan, resolveTopRankedRegionDependency, DEFAULT_ASK_ROW_EGRESS_POLICY, ZERO_ROW_EGRESS_POLICY, resolveProviderResultRowEgressPolicy, normalizeCanonicalQueryResult, normalizeAnalyticalExecutionFingerprint, normalizeAnalyticalExecutionReceipt, createAgentRunCancellationError, } from '@duckcodeailabs/dql-agent';
29
30
  import { addSqlResultFilter, dashboardFilterableResultColumns, filterableResultColumns, replaceBlockStudioSql } from './sql-result-filter.js';
30
31
  import { gatherProposeEnrichment } from './propose-enrich.js';
31
32
  import { handleAppsApi, proposeAppAiBuild, recommendVisualization, } from './apps-api.js';
@@ -328,6 +329,10 @@ const CLIENT_PLAN_AUTHORITY_KEYS = new Set([
328
329
  'priorResolvedAnalyticalPlan',
329
330
  'resolvedAnalyticalPlan',
330
331
  'analyticalFrame',
332
+ // Only the local compound executor may inject this after it has derived a
333
+ // canonical parent result binding. A browser-provided lookalike cannot become
334
+ // a child filter or skip ordinary member validation.
335
+ 'analyticalTaskDependencyBinding',
331
336
  ]);
332
337
  /**
333
338
  * Browser/embedding context is useful retrieval and history input, but it is
@@ -378,6 +383,52 @@ export function agentRunDeadlineMs(request, env = process.env, activeProviderId)
378
383
  ? AGENT_RESEARCH_DEADLINE_MS
379
384
  : AGENT_LOOKUP_DEADLINE_MS;
380
385
  }
386
+ /**
387
+ * Run ready independent compound clauses concurrently, but wait for a typed
388
+ * parent result before executing a declared dependent clause. The scheduler
389
+ * itself has no authority to query or filter; callers supply both execution and
390
+ * a dependency resolver so immutable-plan and SQL guards remain unchanged.
391
+ */
392
+ export async function scheduleCompoundAnalyticalTasks(input) {
393
+ const pending = [...input.tasks];
394
+ const settled = new Map();
395
+ while (pending.length > 0) {
396
+ const ready = pending.filter((task) => task.dependencies.every((dependencyId) => settled.has(dependencyId)));
397
+ if (ready.length === 0) {
398
+ for (const task of pending.splice(0)) {
399
+ settled.set(task.id, {
400
+ task,
401
+ error: 'The compound task dependency graph could not be resolved.',
402
+ dependencyError: {
403
+ ok: false,
404
+ code: 'RESULT_CONTRACT_MISMATCH',
405
+ message: 'The dependent task could not run because its parent dependency was unresolved.',
406
+ },
407
+ });
408
+ }
409
+ break;
410
+ }
411
+ for (const task of ready)
412
+ pending.splice(pending.indexOf(task), 1);
413
+ const batch = await Promise.all(ready.map(async (task) => {
414
+ if (!task.dependency || task.dependency.kind !== 'top_ranked_region')
415
+ return input.runTask(task);
416
+ const parent = settled.get(task.dependency.sourceTaskId);
417
+ const resolution = input.resolveDependency(task, parent);
418
+ if (!resolution.ok) {
419
+ const dependencyError = resolution;
420
+ return { task, error: dependencyError.message, dependencyError };
421
+ }
422
+ return input.runTask(task, resolution.binding);
423
+ }));
424
+ for (const result of batch)
425
+ settled.set(result.task.id, result);
426
+ }
427
+ return input.tasks.map((task) => settled.get(task.id) ?? {
428
+ task,
429
+ error: 'The compound task did not produce an outcome.',
430
+ });
431
+ }
381
432
  /**
382
433
  * Decide how a settled answer gets its business-facing prose.
383
434
  *
@@ -420,11 +471,64 @@ export function shouldSynthesizeAgentRunAnswer(governedAnswer, requestedMode = '
420
471
  rowEgress: DEFAULT_ASK_ROW_EGRESS_POLICY,
421
472
  }).mode !== 'skip';
422
473
  }
474
+ /**
475
+ * A receipt may be rendered in the local inspector and exported into evaluation
476
+ * output. Keep only stable validation codes there; provider error messages can
477
+ * contain a prompt excerpt, result value, or connector detail and must not
478
+ * become user-visible durable data.
479
+ */
480
+ function narrationIntegrityFailureCodes(failures) {
481
+ return [...new Set(failures.map((failure) => {
482
+ const match = failure.trim().match(/^([A-Z][A-Z0-9_]{1,80})/);
483
+ return match?.[1] ?? 'NARRATION_VALIDATION_FAILED';
484
+ }).filter(Boolean))].slice(0, 8);
485
+ }
423
486
  /**
424
487
  * AGT-010 — the semantic route label is descriptive, while the exact
425
488
  * route-specific aggregation proof is authoritative for governed trust.
426
489
  * Missing proof remains blocked for legacy or malformed results.
427
490
  */
491
+ /**
492
+ * Trust for ONE answer, by the same rule the single-answer path uses: a route
493
+ * label is not authority, and a semantic route earns `governed` only when its
494
+ * aggregation proof actually passed.
495
+ */
496
+ export function trustStateForAgentAnswer(answer) {
497
+ if (answer.certification === 'certified' || answer.kind === 'certified')
498
+ return 'certified';
499
+ return semanticAnswerHasPassedAggregationProof(answer) ? 'governed' : 'review_required';
500
+ }
501
+ const TRUST_RANK = {
502
+ certified: 3,
503
+ governed: 2,
504
+ grounded: 1,
505
+ review_required: 0,
506
+ };
507
+ /**
508
+ * A compound answer is exactly as trustworthy as its WEAKEST successful child.
509
+ *
510
+ * The previous rule was `every child completed ? 'governed' : 'review_required'`,
511
+ * which stamped `governed` on a parent whose children were review-required
512
+ * generated SQL — completion is not proof. That is a governance violation and
513
+ * the worst possible failure for this product: the reader is told a number
514
+ * carries governed authority when nothing proved it.
515
+ *
516
+ * `certified` is deliberately NOT reachable here. Certified trust is granted
517
+ * only by executing the exact certified artifact; a parent that merely
518
+ * assembled certified children did not execute one, so it caps at `governed`.
519
+ */
520
+ export function compoundTrustState(childTrust) {
521
+ if (childTrust.length === 0)
522
+ return 'review_required';
523
+ const weakest = childTrust.reduce((low, current) => (TRUST_RANK[current] ?? 0) < (TRUST_RANK[low] ?? 0) ? current : low);
524
+ return weakest === 'certified' ? 'governed' : weakest;
525
+ }
526
+ /** Neutral parent outcome: governed only when every child completed governed. */
527
+ export function compoundStopReason(completedCount, childCount, trustState) {
528
+ return completedCount === childCount && childCount > 0 && trustState === 'governed'
529
+ ? 'governed_compound_answer'
530
+ : 'human_review_required';
531
+ }
428
532
  export function semanticAnswerHasPassedAggregationProof(governedAnswer) {
429
533
  return governedAnswer.route?.tier === 'semantic_metric'
430
534
  && governedAnswer.aggregationSafetyProof?.status === 'safe';
@@ -564,7 +668,7 @@ export function validAskRepairDqlWrapper(source) {
564
668
  }
565
669
  /** Build retained automatic-repair authority from analytical failure state only. */
566
670
  export function analyticalRepairCapabilityForAgentRun(run, resolvedTargetFingerprint) {
567
- if (run.status !== 'blocked')
671
+ if (run.status !== 'blocked' || run.stopReason !== 'blocked')
568
672
  return undefined;
569
673
  const failedRun = analyticalFailedRunFromAgentRun(run);
570
674
  const failure = failedRun?.failure;
@@ -963,6 +1067,33 @@ export function slimAgentRunForTransport(run) {
963
1067
  : {}),
964
1068
  };
965
1069
  }
1070
+ /**
1071
+ * INDEX projection for `GET /api/agent-runs` — strictly lighter than the
1072
+ * presentation projection above.
1073
+ *
1074
+ * A run history list renders one ROW per run: question, route, status, trust,
1075
+ * timing, summary. It never renders an answer body, an event stream, or a step
1076
+ * trace. Shipping those anyway dominated the response: on 300 real stored runs
1077
+ * a 20-run page was 47.61 MB whole and still 6.80 MB under the presentation
1078
+ * projection, of which 6.35 MB was `artifacts[].payload` alone.
1079
+ *
1080
+ * Artifact identity (`id`/`kind`/`title`/`trustState`) is kept so a row can say
1081
+ * what it produced; only the payload body is dropped. The complete immutable
1082
+ * record stays available from `GET /api/agent-runs/:id`.
1083
+ *
1084
+ * Acceptance: PERF-003, E2E-022.
1085
+ */
1086
+ export function agentRunListEntryForTransport(run) {
1087
+ const slim = slimAgentRunForTransport(run);
1088
+ const artifacts = (slim.artifacts ?? []).map((artifact) => {
1089
+ const record = agentRunRecord(artifact);
1090
+ if (!record)
1091
+ return artifact;
1092
+ const { payload: _payload, ...rest } = record;
1093
+ return rest;
1094
+ });
1095
+ return { ...slim, artifacts, steps: [], events: [] };
1096
+ }
966
1097
  export function conversationTurnInputFromRun(run) {
967
1098
  const artifact = run.artifacts.find((candidate) => candidate.kind === 'answer')
968
1099
  ?? run.artifacts.find((candidate) => candidate.kind === 'research_run')
@@ -970,7 +1101,9 @@ export function conversationTurnInputFromRun(run) {
970
1101
  const payload = agentRunRecord(artifact?.payload);
971
1102
  // A blocked run may retain diagnostic artifacts, but their result-shaped
972
1103
  // payload is never accepted conversation evidence or prose authority.
973
- const result = run.status === 'blocked' ? undefined : agentRunRecord(payload?.result);
1104
+ const result = run.status === 'blocked' || run.status === 'cancelled'
1105
+ ? undefined
1106
+ : agentRunRecord(payload?.result);
974
1107
  const columns = conversationResultColumns(result?.columns);
975
1108
  // The visual preview stays tiny, but member resolution needs a wider bounded
976
1109
  // value window. Deriving dimensions from only the eight preview rows caused a
@@ -1010,6 +1143,7 @@ export function conversationTurnInputFromRun(run) {
1010
1143
  sql: agentRunString(payload?.proposedSql) ?? agentRunString(payload?.sql),
1011
1144
  dqlArtifact: agentRunRecord(payload?.dqlArtifact),
1012
1145
  cascade: agentRunRecord(payload?.cascade),
1146
+ narrationIntegrityReceipt: run.narrationIntegrityReceipt,
1013
1147
  result: columns.length > 0 || rows.length > 0
1014
1148
  ? {
1015
1149
  columns,
@@ -1137,12 +1271,7 @@ export class RunScopedProviderDispatchEvidence {
1137
1271
  * from what it already has.
1138
1272
  */
1139
1273
  expectedDispatchMs() {
1140
- if (this.observedDispatchDurations.length === 0)
1141
- return ASSUMED_PROVIDER_DISPATCH_MS;
1142
- const sorted = [...this.observedDispatchDurations].sort((left, right) => left - right);
1143
- // The slowest observed call is the honest predictor: an optimistic median
1144
- // still admits a dispatch that the deadline then kills.
1145
- return sorted[sorted.length - 1];
1274
+ return predictDispatchMs(this.observedDispatchDurations);
1146
1275
  }
1147
1276
  /** True when the remaining wall clock cannot fit another provider call. */
1148
1277
  cannotFitAnotherDispatch() {
@@ -1335,6 +1464,50 @@ function mergeRunScopedProviderDispatchEvidence(run, evidence) {
1335
1464
  diagnosticReceiptV2,
1336
1465
  };
1337
1466
  }
1467
+ /**
1468
+ * Final physical generated-SQL boundary.
1469
+ *
1470
+ * This receives the exact prepared statement immediately before the connector
1471
+ * callback. It intentionally validates before invoking `execute`: a bad
1472
+ * capability or unproven prepared reference must result in zero warehouse
1473
+ * calls, not a post-execution warning. It is module-exported only for the
1474
+ * local-runtime boundary harness; it is never an HTTP API or durable artifact.
1475
+ *
1476
+ * @internal
1477
+ */
1478
+ export async function executePreparedAgenticSqlBoundary(input) {
1479
+ const capability = input.capability;
1480
+ if (capability) {
1481
+ const authorization = mintFinalSqlAuthorization({
1482
+ sql: input.preparedSql,
1483
+ proven: capability.provenIdentifiers.map((identifier) => ({
1484
+ identifier,
1485
+ evidence: capability.evidence[identifier] ?? 'catalog',
1486
+ })),
1487
+ runId: capability.runId,
1488
+ executionId: capability.executionId,
1489
+ snapshotId: capability.snapshotId,
1490
+ planId: capability.planId,
1491
+ targetFingerprint: capability.targetFingerprint,
1492
+ bindings: input.bindings,
1493
+ });
1494
+ const validation = validateAuthorizedSqlReferences(input.preparedSql, undefined);
1495
+ const verdict = verifyFinalSql(authorization, input.preparedSql, qualifyAuthorizationReferences(input.preparedSql, {
1496
+ relations: validation.referencedRelations ?? [],
1497
+ columns: validation.referencedColumns ?? [],
1498
+ }), {
1499
+ ...input.scope,
1500
+ bindings: input.bindings,
1501
+ });
1502
+ if (process.env.DQL_ORCHESTRATOR_TRACE) {
1503
+ console.warn(`[dql] execution authorization: ${verdict.ok ? 'admitted' : 'REFUSED'} proven=${authorization.provenIdentifiers.length}${verdict.ok ? '' : ` reason=${verdict.reason}`}`);
1504
+ }
1505
+ if (!verdict.ok) {
1506
+ throw analyticalError(verdict.reason ?? 'The statement was not authorized for execution.', { origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql' });
1507
+ }
1508
+ }
1509
+ return input.execute();
1510
+ }
1338
1511
  export async function startLocalServer(opts) {
1339
1512
  const { rootDir, executor, connection: rawConnection, preferredPort, projectRoot = process.cwd() } = opts;
1340
1513
  const bindHost = opts.host ?? process.env.DQL_HOST ?? '127.0.0.1';
@@ -2032,6 +2205,9 @@ export async function startLocalServer(opts) {
2032
2205
  });
2033
2206
  };
2034
2207
  async function runGovernedAgentAnswerForRun(request, repair, route = 'generated_answer', onProgress, routeDecision) {
2208
+ return runGovernedAgentAnswerForRunInner(request, repair, route, onProgress, routeDecision);
2209
+ }
2210
+ async function runGovernedAgentAnswerForRunInner(request, repair, route = 'generated_answer', onProgress, routeDecision) {
2035
2211
  const governed = resolveGovernedAnswerRunner(projectRoot);
2036
2212
  let resolvedProvider = governed?.provider ?? null;
2037
2213
  let runner = governed?.runner ?? null;
@@ -2134,6 +2310,9 @@ export async function startLocalServer(opts) {
2134
2310
  snapshotId: runProjectSnapshot.snapshotId,
2135
2311
  })
2136
2312
  : undefined;
2313
+ // Local to this exact answer invocation. Compound children each enter this
2314
+ // function separately, so no child can consume another child's capability.
2315
+ const agenticExecutionCapabilityGate = new AgenticExecutionCapabilityGate();
2137
2316
  await runner.run({
2138
2317
  provider: resolvedProvider,
2139
2318
  ...(agentRunProviderEvidenceContext.getStore()
@@ -2158,9 +2337,13 @@ export async function startLocalServer(opts) {
2158
2337
  },
2159
2338
  reasoningEffort,
2160
2339
  ...(analysisDepth ? { analysisDepth } : {}),
2340
+ orchestrationMode: route === 'research' ? 'research' : 'ask',
2161
2341
  allowProviderSemanticMemberSelection: route === 'research',
2162
2342
  researchResultRowsOptIn: route === 'research' && request.researchResultRowsOptIn === true,
2163
2343
  projectRoot,
2344
+ // Keys the execution authorization, so the proofs the analyst loop
2345
+ // gathers can be checked against the statement this run executes.
2346
+ ...(request.runId ? { agentRunId: request.runId } : {}),
2164
2347
  preparedContextPack: preparedAgentContextPacks.get(request),
2165
2348
  domainContext,
2166
2349
  projectSnapshot: { snapshotId: runProjectSnapshot.snapshotId, manifest: runProjectSnapshot.manifest },
@@ -2328,6 +2511,20 @@ export async function startLocalServer(opts) {
2328
2511
  analyticalReferenceInstant: new Date().toISOString(),
2329
2512
  executeCertifiedBlock: (node, invocation) => executeCertifiedBlockForAgent(node, invocation, semanticConnection, semanticConnectionName),
2330
2513
  executeGeneratedSql: (sql, artifact) => executeGeneratedArtifactForAgent(request.question, sql, artifact, semanticConnection, semanticConnectionName),
2514
+ executeAgenticGeneratedSql: async (capability, sql, artifact) => {
2515
+ if (!agenticExecutionCapabilityGate.consume(capability)) {
2516
+ throw analyticalError('This analyst execution capability was already consumed; DQL did not retry it with stale proof.', {
2517
+ origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
2518
+ });
2519
+ }
2520
+ return executeGeneratedArtifactForAgent(request.question, sql, artifact, semanticConnection, semanticConnectionName, capability, {
2521
+ runId: request.runId,
2522
+ executionId: capability.executionId,
2523
+ snapshotId: runProjectSnapshot.snapshotId,
2524
+ planId: routeDecision?.resolvedAnalyticalPlan?.planId,
2525
+ targetFingerprint: generatedProposalTargetIdentity?.identityFingerprint,
2526
+ });
2527
+ },
2331
2528
  executeDqlArtifact: (artifact) => executeArtifactReferenceForAgent(artifact, request.question, semanticConnection, semanticConnectionName),
2332
2529
  getSchemaContext: (question, preparedContextPack) => getSchemaContextForAgent(question, preparedContextPack, semanticConnection, request.executionTarget?.target === 'connection'
2333
2530
  ? request.executionTarget.connectionName
@@ -2598,9 +2795,198 @@ export async function startLocalServer(opts) {
2598
2795
  }
2599
2796
  const answerRunExecutor = async ({ request, route, routeDecision, attempt, repairHint, emit }) => {
2600
2797
  const runStartedAtMs = Date.now();
2798
+ const turnPlan = buildAnalyticalTurnPlan({
2799
+ question: request.question,
2800
+ mode: route === 'research' ? 'research' : 'ask',
2801
+ turnId: request.runId,
2802
+ candidateIds: routeDecision?.retrievalEvidence?.candidateIds ?? [],
2803
+ frozen: routeDecision?.resolvedAnalyticalPlan?.mode === 'authoritative',
2804
+ snapshotId: routeDecision?.resolvedAnalyticalPlan?.snapshotId ?? routeDecision?.retrievalEvidence?.snapshotId,
2805
+ sourceFingerprint: routeDecision?.resolvedAnalyticalPlan?.sourceFingerprint ?? routeDecision?.retrievalEvidence?.sourceFingerprint,
2806
+ });
2807
+ const childTurn = Boolean(request.workspaceContext && typeof request.workspaceContext === 'object'
2808
+ && request.workspaceContext.analyticalTaskChild === true);
2809
+ // Compound questions are a bounded task graph, not one query whose answer
2810
+ // is copied into several labels. Independent children share the parent's
2811
+ // signal/deadline and return truthful partial success.
2812
+ if (turnPlan.tasks.length > 1 && !childTurn && (attempt ?? 0) === 0) {
2813
+ const runChildTask = async (task, dependencyBinding) => {
2814
+ if (request.signal?.aborted)
2815
+ rethrowIfCancelled(request.signal.reason, request.signal);
2816
+ try {
2817
+ const childRequest = {
2818
+ ...request,
2819
+ question: task.question,
2820
+ ...(dependencyBinding ? {
2821
+ conversationContext: {
2822
+ ...(request.conversationContext ?? {}),
2823
+ // This is the complete parent-to-child data boundary: no
2824
+ // parent prose, SQL, or rows cross into a dependent clause.
2825
+ analyticalTaskDependencyBinding: dependencyBinding,
2826
+ },
2827
+ } : {}),
2828
+ workspaceContext: {
2829
+ ...(request.workspaceContext && typeof request.workspaceContext === 'object' ? request.workspaceContext : {}),
2830
+ analyticalTaskChild: true,
2831
+ analyticalParentRunId: request.runId,
2832
+ analyticalTaskId: task.id,
2833
+ ...(dependencyBinding ? { analyticalTaskDependencyBinding: dependencyBinding } : {}),
2834
+ },
2835
+ };
2836
+ const answer = await runGovernedAgentAnswerForRun(childRequest, { attempt: 0, repairHint }, route, (message) => emit({ type: 'executor.started', message: `Task ${task.id}: ${message}`, route }), undefined);
2837
+ if (answer.result) {
2838
+ const canonical = normalizeCanonicalQueryResult({
2839
+ ...answer.result,
2840
+ resultFingerprint: answer.result.resultFingerprint ?? answer.result.executionReceipt?.resultFingerprint,
2841
+ executionReceipt: answer.result.executionReceipt,
2842
+ answerTier: answer.route?.tier ?? answer.sourceTier,
2843
+ });
2844
+ answer.result = {
2845
+ ...answer.result,
2846
+ columns: canonical.columns,
2847
+ rows: canonical.rows,
2848
+ rowCount: canonical.rowCount,
2849
+ resultFingerprint: canonical.resultFingerprint,
2850
+ ...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
2851
+ ...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
2852
+ };
2853
+ }
2854
+ return { task, answer };
2855
+ }
2856
+ catch (error) {
2857
+ rethrowIfCancelled(error, request.signal);
2858
+ return { task, error: error instanceof Error ? error.message : String(error) };
2859
+ }
2860
+ };
2861
+ const scheduledChildren = await scheduleCompoundAnalyticalTasks({
2862
+ tasks: turnPlan.tasks.slice(0, 6),
2863
+ runTask: async (task, binding) => {
2864
+ const child = await runChildTask(task, binding);
2865
+ return { task, value: child.answer, error: child.error };
2866
+ },
2867
+ resolveDependency: (task, parent) => {
2868
+ const sourceTaskId = task.dependency?.sourceTaskId ?? '';
2869
+ return resolveTopRankedRegionDependency(sourceTaskId, parent?.value?.result
2870
+ ? normalizeCanonicalQueryResult({
2871
+ ...parent.value.result,
2872
+ resultFingerprint: parent.value.result.resultFingerprint ?? parent.value.result.executionReceipt?.resultFingerprint,
2873
+ executionReceipt: parent.value.result.executionReceipt,
2874
+ answerTier: parent.value.route?.tier ?? parent.value.sourceTier,
2875
+ })
2876
+ : undefined, parent?.task);
2877
+ },
2878
+ });
2879
+ const childResults = scheduledChildren.map(({ task, value, error, dependencyError }) => ({
2880
+ task,
2881
+ answer: value,
2882
+ error,
2883
+ ...(dependencyError ? {
2884
+ dependencyGap: buildCoverageGap({
2885
+ code: dependencyError.code,
2886
+ phase: 'planning',
2887
+ message: dependencyError.message,
2888
+ searchedSources: routeDecision?.retrievalEvidence?.candidateIds ?? [],
2889
+ attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
2890
+ missing: ['unambiguous_top_region'],
2891
+ recoverable: true,
2892
+ planFrozen: turnPlan.frozen,
2893
+ nextActions: ['Ask for a single top region or review the parent result before retrying the customer task.'],
2894
+ }),
2895
+ } : {}),
2896
+ }));
2897
+ const outcomes = childResults.map(({ task, answer, error }) => ({
2898
+ version: 1,
2899
+ taskId: task.id,
2900
+ status: error || answer?.kind === 'no_answer' ? 'gap' : 'completed',
2901
+ ...(answer?.answer || answer?.text ? { summary: answer.answer ?? answer.text } : {}),
2902
+ ...(answer?.result?.resultFingerprint ? { resultFingerprint: answer.result.resultFingerprint } : {}),
2903
+ ...(error || answer?.kind === 'no_answer' ? {
2904
+ gap: childResults.find((candidate) => candidate.task.id === task.id)?.dependencyGap ?? buildCoverageGap({
2905
+ code: answer?.refusalCode === 'ambiguous' ? 'AMBIGUOUS_MEANING' : 'EXECUTION_FAILED',
2906
+ phase: answer?.executionError ? 'execution' : 'meaning',
2907
+ message: error ?? answer?.answer ?? answer?.text ?? 'The task did not produce an accepted analytical result.',
2908
+ searchedSources: routeDecision?.retrievalEvidence?.candidateIds ?? [],
2909
+ attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
2910
+ missing: [],
2911
+ recoverable: false,
2912
+ planFrozen: turnPlan.frozen,
2913
+ nextActions: ['Review the task context and retry the clause.'],
2914
+ }),
2915
+ } : {}),
2916
+ }));
2917
+ const completedCount = outcomes.filter((outcome) => outcome.status === 'completed').length;
2918
+ const tasks = turnPlan.tasks.map((task) => ({
2919
+ ...task,
2920
+ status: outcomes.find((outcome) => outcome.taskId === task.id)?.status === 'completed' ? 'completed' : 'gap',
2921
+ }));
2922
+ const answerText = childResults.map(({ task, answer, error }) => `${task.question}: ${error ?? answer?.answer ?? answer?.text ?? 'No accepted result was produced.'}`).join('\n\n');
2923
+ const compoundTrust = compoundTrustState(childResults
2924
+ .filter(({ answer, error }) => !error && answer && answer.kind !== 'no_answer')
2925
+ .map(({ answer }) => trustStateForAgentAnswer(answer)));
2926
+ return {
2927
+ summary: completedCount === outcomes.length
2928
+ ? `Answered ${completedCount} analytical clauses.`
2929
+ : `Answered ${completedCount} of ${outcomes.length} analytical clauses; the remaining clauses need review.`,
2930
+ answer: answerText,
2931
+ status: completedCount === outcomes.length ? 'completed' : completedCount > 0 ? 'needs_review' : 'needs_clarification',
2932
+ // The parent is only as trustworthy as its weakest SUCCESSFUL child.
2933
+ // Completion is not proof: the previous rule stamped `governed` on a
2934
+ // parent assembled from review-required generated SQL.
2935
+ trustState: compoundTrust,
2936
+ // The stop reason has to agree with the trust it reports. Claiming a
2937
+ // governed semantic answer over generated/certified children
2938
+ // misrepresents provenance as much as the trust label does.
2939
+ stopReason: compoundStopReason(completedCount, outcomes.length, compoundTrust),
2940
+ artifacts: childResults.map(({ task, answer, error }) => agentRunArtifact('answer', `Task: ${task.question}`, {
2941
+ taskId: task.id,
2942
+ question: task.question,
2943
+ answer: answer?.answer ?? answer?.text,
2944
+ resultFingerprint: answer?.result?.resultFingerprint,
2945
+ error,
2946
+ })),
2947
+ evaluations: outcomes.map((outcome) => agentRunEvaluation(`analytical-task-${outcome.taskId}`, `Analytical task ${outcome.taskId}`, outcome.status === 'completed', outcome.status === 'completed' ? 'info' : 'warning', outcome.summary ?? outcome.gap?.message ?? 'Task outcome recorded.', outcome)),
2948
+ analyticalTurnPlan: { ...turnPlan, tasks, frozen: true },
2949
+ analyticalTaskOutcomes: outcomes,
2950
+ };
2951
+ }
2601
2952
  let governedAnswer;
2602
2953
  try {
2603
2954
  governedAnswer = await runGovernedAgentAnswerForRun(request, { attempt, repairHint }, route, (message) => emit({ type: 'executor.started', message, route }), routeDecision);
2955
+ // Keep the canonical result contract on the answer itself, not only on
2956
+ // the narration preview. Conversation persistence, follow-up member
2957
+ // resolution, Apply, and the notebook table all read this payload; if a
2958
+ // connector returned positional rows, dropping normalization here makes
2959
+ // the next question lose its customer/product/region binding (AGT-032).
2960
+ if (governedAnswer.result) {
2961
+ const canonical = normalizeCanonicalQueryResult({
2962
+ ...governedAnswer.result,
2963
+ resultFingerprint: governedAnswer.result.resultFingerprint
2964
+ ?? governedAnswer.result.executionReceipt?.resultFingerprint,
2965
+ executionReceipt: governedAnswer.result.executionReceipt,
2966
+ answerTier: governedAnswer.route?.tier ?? governedAnswer.sourceTier,
2967
+ trustState: governedAnswer.certification === 'certified'
2968
+ ? 'certified'
2969
+ : governedAnswer.kind === 'no_answer'
2970
+ ? 'blocked'
2971
+ : governedAnswer.reviewStatus === 'analyst_review_required'
2972
+ ? 'review_required'
2973
+ : 'governed',
2974
+ });
2975
+ governedAnswer.result = {
2976
+ ...governedAnswer.result,
2977
+ columns: canonical.columns,
2978
+ rows: canonical.rows,
2979
+ rowCount: canonical.rowCount,
2980
+ // Preserve the execution-service/receipt fingerprint. A local UI
2981
+ // digest is only a legacy fallback in normalizeCanonicalQueryResult;
2982
+ // it must never replace the cryptographic identity of this run.
2983
+ resultFingerprint: canonical.resultFingerprint,
2984
+ ...(canonical.executionTime !== undefined ? { executionTime: canonical.executionTime } : {}),
2985
+ ...(canonical.truncated ? { truncated: true } : {}),
2986
+ ...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
2987
+ ...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
2988
+ };
2989
+ }
2604
2990
  // Surface the approved Hint-Graph corrections that shaped this answer so the
2605
2991
  // UI can show an "applied learnings" chip (memoryContext is already on the answer).
2606
2992
  if (!governedAnswer.appliedHints) {
@@ -2818,6 +3204,38 @@ export async function startLocalServer(opts) {
2818
3204
  // has already spent its one evidence-aware repair. Keep this terminal and
2819
3205
  // inspectable; an ordinary Ask must never silently become a second Research run.
2820
3206
  const isModelDeclined = governedAnswer.kind === 'no_answer' && governedAnswer.refusalCode === 'model_declined';
3207
+ // Before an analytical plan is frozen, a generated Ask gap is a recoverable
3208
+ // coverage result rather than a terminal governed error. Let the run engine
3209
+ // consume the typed evaluation and make one bounded cascade decision. Frozen
3210
+ // certified/semantic plans and provider/policy/execution failures remain
3211
+ // fail-closed.
3212
+ const canRecoverPreFreezeGap = (isGroundingGap || isModelDeclined)
3213
+ && route === 'generated_answer'
3214
+ && (request.requestedMode === undefined || request.requestedMode === 'auto' || request.requestedMode === 'ask')
3215
+ && routeDecision?.resolvedAnalyticalPlan?.mode !== 'authoritative';
3216
+ const typedCoverageGap = (isGroundingGap || isModelDeclined)
3217
+ ? buildCoverageGap({
3218
+ code: governedAnswer.refusalCode === 'modeling_gap' ? 'MISSING_RELATIONSHIP' : 'MISSING_RUNTIME_CAPABILITY',
3219
+ phase: 'planning',
3220
+ message: governedAnswer.refusalDetails?.message
3221
+ ?? (isModelDeclined
3222
+ ? 'The available context did not produce a safe analytical tuple.'
3223
+ : 'The retrieved context did not prove the metadata required for this analytical question.'),
3224
+ searchedSources: ['certified_blocks', 'semantic_metrics', 'dbt_manifest', 'relationship_graph', 'warehouse_metadata'],
3225
+ attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
3226
+ missing: governedAnswer.refusalDetails?.offending
3227
+ ? [
3228
+ governedAnswer.refusalDetails.offending.relation,
3229
+ governedAnswer.refusalDetails.offending.column,
3230
+ ].filter((value) => Boolean(value))
3231
+ : ['an executable metric/dimension/relationship tuple'],
3232
+ recoverable: canRecoverPreFreezeGap,
3233
+ planFrozen: routeDecision?.resolvedAnalyticalPlan?.mode === 'authoritative',
3234
+ nextActions: canRecoverPreFreezeGap
3235
+ ? ['continue through DBT-grounded relational context', 'run review-required generated SQL', 'start bounded Research if coverage remains incomplete']
3236
+ : ['review the retained metadata gap', 'select a governed metric or dimension', 'repair modeling if the relationship is missing'],
3237
+ })
3238
+ : undefined;
2821
3239
  // Only a genuinely AMBIGUOUS question is surfaced as "needs clarification".
2822
3240
  // Grounding/compiler gaps are terminal review states with their evidence trace;
2823
3241
  // provider outages are blocked so the UI can offer an explicit retry.
@@ -2867,7 +3285,34 @@ export async function startLocalServer(opts) {
2867
3285
  ? { ...preview, rows: redactProviderResultRows(preview.rows, narrationMaxRows) }
2868
3286
  : undefined;
2869
3287
  let narrationSource;
2870
- let narrationValidationFailures = [];
3288
+ // The receipt is the durable source of truth for evaluation and inspector
3289
+ // display. Do not reconstruct this later from rows or the reader-facing
3290
+ // fallback sentence: both are presentation artifacts, not evidence that a
3291
+ // fact-grounded narrator actually ran.
3292
+ const verifiedFactNarration = narrationPlan.mode === 'verified_facts'
3293
+ && Boolean(governedAnswer.analyticalFacts && governedAnswer.resolvedAnalyticalPlan?.analyticalFrame);
3294
+ let narrationIntegrityReceipt = narrationPlan.mode === 'skip'
3295
+ ? {
3296
+ version: 1,
3297
+ mode: 'skip',
3298
+ outcome: 'skipped',
3299
+ attempted: false,
3300
+ factCount: 0,
3301
+ maxRows: 0,
3302
+ validationFailures: [],
3303
+ skipReason: narrationPlan.reason,
3304
+ }
3305
+ : {
3306
+ version: 1,
3307
+ mode: verifiedFactNarration ? 'verified_facts' : 'preview_grounded',
3308
+ // This is deliberately pessimistic until a narration outcome is
3309
+ // observed, so an exception cannot be persisted as a silent skip.
3310
+ outcome: 'error',
3311
+ attempted: true,
3312
+ factCount: verifiedFactNarration ? governedAnswer.analyticalFacts?.facts.length ?? 0 : 0,
3313
+ maxRows: narrationPlan.maxRows,
3314
+ validationFailures: [],
3315
+ };
2871
3316
  if (narrationPlan.mode !== 'skip' && narrationProvider) {
2872
3317
  const narrationStartedAtMs = Date.now();
2873
3318
  const draft = governedAnswer.answer ?? governedAnswer.text;
@@ -2880,8 +3325,16 @@ export async function startLocalServer(opts) {
2880
3325
  columnCount: providerPreview?.columns.length ?? 0,
2881
3326
  },
2882
3327
  });
2883
- const narrationDispatchOptions = () => ({
2884
- maxTokens: 350,
3328
+ const narrationDispatchOptions = (factCount = 0) => ({
3329
+ // The ceiling has to grow with the result. Every claim must echo the
3330
+ // fact ids it rests on, and a fact id is a long hex string that
3331
+ // tokenizes badly — ten of them consume most of the budget before a
3332
+ // word of prose is written. At a flat 350 a ten-row answer was
3333
+ // truncated mid-sentence ("...Elizabeth Shea (875), and Dyl"), so the
3334
+ // JSON never closed, BOTH attempts failed as UNPARSEABLE_CLAIMS, and
3335
+ // the reader got "Verified narration was unavailable" above a robot
3336
+ // dump of the very rows the model had just described correctly.
3337
+ maxTokens: narrationMaxTokensForFacts(factCount),
2885
3338
  temperature: 0.3,
2886
3339
  maxProviderDispatches: 2,
2887
3340
  ...(agentRunProviderEvidenceContext.getStore()
@@ -2898,10 +3351,19 @@ export async function startLocalServer(opts) {
2898
3351
  factSet: governedAnswer.analyticalFacts,
2899
3352
  question: request.question,
2900
3353
  maxRows: narrationPlan.maxRows,
2901
- complete: async ({ system, user }) => streamOrGenerate(narrationProvider, [{ role: 'system', content: system }, { role: 'user', content: user }], narrationDispatchOptions(), () => { }),
3354
+ complete: async ({ system, user }) => streamOrGenerate(narrationProvider, [{ role: 'system', content: system }, { role: 'user', content: user }], narrationDispatchOptions(governedAnswer.analyticalFacts?.facts.length ?? 0), () => { }),
2902
3355
  });
2903
3356
  narrationSource = composed.source;
2904
- narrationValidationFailures = composed.validationFailures;
3357
+ narrationIntegrityReceipt = {
3358
+ ...narrationIntegrityReceipt,
3359
+ outcome: composed.source === 'llm' ? 'success' : 'deterministic_fallback',
3360
+ validationFailures: narrationIntegrityFailureCodes(composed.validationFailures),
3361
+ };
3362
+ if (process.env.DQL_ORCHESTRATOR_TRACE) {
3363
+ console.warn(`[dql] narration: source=${composed.source}${narrationIntegrityReceipt.validationFailures.length > 0
3364
+ ? ` rejected=${narrationIntegrityReceipt.validationFailures.join(',')}`
3365
+ : ''}`);
3366
+ }
2905
3367
  synthesizedAnswer = composed.source === 'llm'
2906
3368
  ? composed.narrative.text
2907
3369
  // A verification failure is not silent: the deterministic join is a
@@ -2931,11 +3393,22 @@ export async function startLocalServer(opts) {
2931
3393
  narrationSource = result.source;
2932
3394
  if (result.text)
2933
3395
  synthesizedAnswer = result.text;
3396
+ narrationIntegrityReceipt = {
3397
+ ...narrationIntegrityReceipt,
3398
+ outcome: result.source === 'llm' ? 'success' : 'deterministic_fallback',
3399
+ validationFailures: [],
3400
+ };
2934
3401
  }
2935
3402
  }
2936
3403
  catch {
2937
3404
  // Keep the governed draft on any narration failure.
2938
3405
  synthesizedAnswer = undefined;
3406
+ narrationIntegrityReceipt = {
3407
+ ...narrationIntegrityReceipt,
3408
+ outcome: 'error',
3409
+ validationFailures: [],
3410
+ errorCode: 'narration_error',
3411
+ };
2939
3412
  }
2940
3413
  finally {
2941
3414
  narrationDurationMs = Date.now() - narrationStartedAtMs;
@@ -2971,7 +3444,7 @@ export async function startLocalServer(opts) {
2971
3444
  ? 'blocked'
2972
3445
  : needsClarification
2973
3446
  ? 'needs_clarification'
2974
- : isGroundingGap || isModelDeclined
3447
+ : (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
2975
3448
  ? 'blocked'
2976
3449
  : isCertified || isSemantic
2977
3450
  ? 'completed'
@@ -2980,7 +3453,7 @@ export async function startLocalServer(opts) {
2980
3453
  ? 'blocked'
2981
3454
  : needsClarification
2982
3455
  ? 'not_applicable'
2983
- : isGroundingGap || isModelDeclined
3456
+ : (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
2984
3457
  ? 'blocked'
2985
3458
  : isCertified
2986
3459
  ? 'certified'
@@ -2994,7 +3467,7 @@ export async function startLocalServer(opts) {
2994
3467
  // A gap is terminal, but it is NOT a provider outage: it still owes the
2995
3468
  // user its evidence trace, and its next-actions below stay the
2996
3469
  // grounding-gap set rather than "retry the provider".
2997
- : isGroundingGap || isModelDeclined
3470
+ : (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
2998
3471
  ? 'human_review_required'
2999
3472
  : isCertified
3000
3473
  ? 'certified_answer_found'
@@ -3075,7 +3548,11 @@ export async function startLocalServer(opts) {
3075
3548
  trustState,
3076
3549
  stopReason,
3077
3550
  artifacts: isTerminalFailure
3078
- ? [agentRunArtifact('answer', 'Failed governed analytical run', governedAnswer, governedAnswer.sourceCertifiedBlock ?? governedAnswer.block?.name, 'blocked')]
3551
+ ? [agentRunArtifact('answer',
3552
+ // Was 'Failed governed analytical run' — internal orchestration state
3553
+ // used as the card heading a user reads. It names our pipeline, not
3554
+ // what happened to their question.
3555
+ terminalFailureTitle(governedAnswer), governedAnswer, governedAnswer.sourceCertifiedBlock ?? governedAnswer.block?.name, 'blocked')]
3079
3556
  : governedAnswer.kind === 'no_answer'
3080
3557
  // A refusal still keeps the DQL draft the answer loop produced (when any),
3081
3558
  // so the "Review DQL draft" next-action isn't a dead link and the user can
@@ -3083,7 +3560,7 @@ export async function startLocalServer(opts) {
3083
3560
  ? (governedAnswer.semanticExecutionTrace
3084
3561
  ? [agentRunArtifact('answer', governedAnswer.refusalCode === 'ambiguous'
3085
3562
  ? 'Semantic path selection required'
3086
- : 'Semantic compilation details', governedAnswer, undefined, needsClarification ? 'not_applicable' : 'review_required')]
3563
+ : 'Semantic compilation details', typedCoverageGap ? { ...governedAnswer, coverageGap: typedCoverageGap } : governedAnswer, undefined, needsClarification ? 'not_applicable' : 'review_required')]
3087
3564
  : governedAnswer.dqlArtifact && !isProviderError && !isGroundingGap && !isModelDeclined && !isPolicyBlocked
3088
3565
  ? [agentRunArtifact('dql_block_draft', 'DQL draft (review required)', governedAnswer.dqlArtifact, undefined, 'review_required')]
3089
3566
  // Every refusal still owes the user an account of itself. Emitting no
@@ -3092,7 +3569,7 @@ export async function startLocalServer(opts) {
3092
3569
  // answered", the DQL draft and the compiled SQL all become
3093
3570
  // unreachable exactly when the user most needs to see why it stopped
3094
3571
  // and carry the query into a notebook.
3095
- : [agentRunArtifact('answer', 'No answer was accepted', governedAnswer, undefined, needsClarification ? 'not_applicable' : 'blocked')])
3572
+ : [agentRunArtifact('answer', 'No answer was accepted', typedCoverageGap ? { ...governedAnswer, coverageGap: typedCoverageGap } : governedAnswer, undefined, needsClarification ? 'not_applicable' : 'blocked')])
3096
3573
  : [agentRunArtifact('answer', isCertified ? 'Certified answer' : isSemantic ? 'Governed semantic answer' : isExploratory ? 'Exploratory DBT-grounded answer' : 'Review-required answer', governedAnswer, governedAnswer.sourceCertifiedBlock ?? governedAnswer.block?.name, isCertified ? 'certified' : isSemantic ? 'governed' : 'review_required')],
3097
3574
  evaluations: [
3098
3575
  agentRunEvaluation('route-decision', 'Route decision', true, 'info', routeDecision?.reason ?? 'Routed request to governed answer.', {
@@ -3127,22 +3604,34 @@ export async function startLocalServer(opts) {
3127
3604
  ...(isExecutionFailure ? [agentRunEvaluation('query-execution', 'Query execution', false, 'blocking', `The governed query failed before it produced a result: ${governedAnswer.executionError}`)] : []),
3128
3605
  ...(isGroundingGap ? [
3129
3606
  {
3130
- ...agentRunEvaluation('grounding-gap', 'Metadata grounding', false, 'warning', 'The bounded lookup could not prove the required metadata grounding. No automatic retry or Research escalation was started.', {
3607
+ ...agentRunEvaluation('grounding-gap', 'Metadata grounding', false, 'warning', canRecoverPreFreezeGap
3608
+ ? 'The first governed lookup did not prove the required metadata grounding. Continue through the bounded relational/generated cascade before asking the user to repair modeling.'
3609
+ : 'The bounded lookup could not prove the required metadata grounding. This plan is already frozen, so no automatic Research escalation was started.', {
3131
3610
  refusalCode: governedAnswer.refusalCode,
3132
3611
  refusalDetails: governedAnswer.refusalDetails,
3133
3612
  validationWarnings: governedAnswer.validationWarnings,
3134
3613
  route: governedAnswer.route,
3614
+ coverageGap: typedCoverageGap,
3135
3615
  }),
3616
+ ...(canRecoverPreFreezeGap ? {
3617
+ suggestedRepair: 'Continue the unfrozen Ask cascade through DBT-grounded relational context and review-required generated SQL.',
3618
+ } : {}),
3136
3619
  },
3137
3620
  ] : []),
3138
3621
  ...(isModelDeclined ? [
3139
3622
  {
3140
- ...agentRunEvaluation('declined-despite-context', 'Answer grounding', false, 'blocking', 'The bounded lookup could not compose a governed query after its in-lane repair. Start Research explicitly to investigate beyond this lookup budget.', {
3623
+ ...agentRunEvaluation('declined-despite-context', 'Answer grounding', false, 'blocking', canRecoverPreFreezeGap
3624
+ ? 'The governed lookup could not compose a query after its bounded in-lane repair. Continue with the Research context ledger before returning a typed gap.'
3625
+ : 'The bounded lookup could not compose a governed query after its in-lane repair. Start Research explicitly to investigate beyond this lookup budget.', {
3141
3626
  refusalCode: governedAnswer.refusalCode,
3142
3627
  refusalDetails: governedAnswer.refusalDetails,
3143
3628
  validationWarnings: governedAnswer.validationWarnings,
3144
3629
  route: governedAnswer.route,
3630
+ coverageGap: typedCoverageGap,
3145
3631
  }),
3632
+ ...(canRecoverPreFreezeGap ? {
3633
+ suggestedRepair: 'Search the surrounding metadata and relationship context before giving up on the analytical turn.',
3634
+ } : {}),
3146
3635
  },
3147
3636
  ] : []),
3148
3637
  ...(isPolicyBlocked ? [
@@ -3160,12 +3649,38 @@ export async function startLocalServer(opts) {
3160
3649
  ...(governedAnswer.executionError ? [
3161
3650
  agentRunEvaluation('execution-error', 'Execution error', false, 'warning', governedAnswer.executionError),
3162
3651
  ] : []),
3652
+ // Keep only the content-free receipt codes in the durable inspection
3653
+ // record. Raw verifier prose can contain a result value, prompt excerpt,
3654
+ // or provider error and is not safe evidence to surface or persist.
3655
+ ...(narrationIntegrityReceipt.outcome === 'deterministic_fallback'
3656
+ && narrationIntegrityReceipt.validationFailures.length > 0 ? [
3657
+ agentRunEvaluation('narration-verification', 'Narration verification', false, 'warning', `The drafted narration was rejected against the result fact set, so the deterministic record was shown instead: ${narrationIntegrityReceipt.validationFailures.join(', ')}.`, { narrationSource, validationFailures: narrationIntegrityReceipt.validationFailures }),
3658
+ ] : []),
3163
3659
  ],
3164
3660
  nextActions,
3165
3661
  providerEgressReceipts: finalProviderEgressReceipts,
3166
3662
  telemetry: finalTelemetry,
3663
+ narrationIntegrityReceipt,
3167
3664
  };
3168
3665
  };
3666
+ /**
3667
+ * A heading for a run that ended without an answer, in the user's terms.
3668
+ *
3669
+ * Says WHICH stage stopped, because "it failed" and "it was stopped before
3670
+ * running" call for different next moves: one is worth retrying, the other
3671
+ * needs the question or the model changed.
3672
+ */
3673
+ const terminalFailureTitle = (answer) => {
3674
+ switch (answer.refusalCode) {
3675
+ case 'policy_blocked': return 'Blocked by a governance policy';
3676
+ case 'modeling_gap': return 'Not modeled yet';
3677
+ case 'grounding_gap': return 'Not enough context to answer safely';
3678
+ case 'model_declined': return 'The assistant declined to answer';
3679
+ case 'provider_error': return 'The AI provider did not respond';
3680
+ case 'ambiguous': return 'Needs one detail before running';
3681
+ default: return 'No answer was produced';
3682
+ }
3683
+ };
3169
3684
  const conversationRunExecutor = async ({ request, routeDecision, emitAnswerDelta }) => {
3170
3685
  const kind = routeDecision?.conversationalKind ?? 'smalltalk';
3171
3686
  const isGeneralKnowledge = routeDecision?.category === 'general_knowledge';
@@ -3182,6 +3697,15 @@ export async function startLocalServer(opts) {
3182
3697
  let text = kind === 'answer_explanation'
3183
3698
  ? buildPriorAnswerExplanation(request.question, request.conversationContext)
3184
3699
  : undefined;
3700
+ // A definitional question that NAMES a governed artifact is answerable from
3701
+ // the catalog: the description, domain, and dimensions are already recorded.
3702
+ // Reaching for a provider to paraphrase facts we hold can only add drift, and
3703
+ // the generic conversational reply this replaces used none of them.
3704
+ //
3705
+ // Returns undefined unless the question names something real, so a turn that
3706
+ // does not match keeps today's behaviour exactly.
3707
+ if (!text)
3708
+ text = buildGovernedObjectExplanation(request.question);
3185
3709
  if (text) {
3186
3710
  emitAnswerDelta?.(text);
3187
3711
  }
@@ -3648,10 +4172,18 @@ export async function startLocalServer(opts) {
3648
4172
  const conversationHistory = request.history?.length
3649
4173
  ? request.history
3650
4174
  : conversationHistoryFromContext(request.conversationContext);
4175
+ // The provider that will plan the investigation as hypotheses. Absent or
4176
+ // unreachable, `planResearch` keeps its deterministic template, so
4177
+ // research never depends on a model being available.
4178
+ const researchPlanner = resolveGovernedAnswerRunner(projectRoot);
4179
+ const researchPlannerProvider = researchPlanner
4180
+ ? createGovernedTextProvider(researchPlanner.provider, projectRoot)
4181
+ : undefined;
3651
4182
  const plan = await planResearch({
3652
4183
  question: request.question,
3653
4184
  metrics,
3654
4185
  blocks,
4186
+ ...(researchPlannerProvider ? { provider: researchPlannerProvider } : {}),
3655
4187
  intent: request.intent,
3656
4188
  isFollowUp: conversationHistory.length > 0,
3657
4189
  history: conversationHistory,
@@ -3687,6 +4219,7 @@ export async function startLocalServer(opts) {
3687
4219
  plan,
3688
4220
  };
3689
4221
  let researchRun;
4222
+ const researchRuns = [];
3690
4223
  let researchWorkspaceError;
3691
4224
  if (!needsClarification) {
3692
4225
  try {
@@ -3711,26 +4244,155 @@ export async function startLocalServer(opts) {
3711
4244
  });
3712
4245
  emit({
3713
4246
  type: 'artifact.created',
3714
- message: 'Saved notebook research workspace record.',
4247
+ message: 'Saved the immutable root research plan; executing bounded child branches.',
3715
4248
  route: 'research',
3716
4249
  trustState: 'review_required',
3717
- payload: { researchRunId: created.id, notebookPath },
3718
- });
3719
- const executed = await runNotebookResearch(storage, created, {
3720
- domain: agentRunWorkspaceValue(request, 'domain'),
3721
- owner: agentRunWorkspaceValue(request, 'owner'),
3722
- sourceCellFingerprint,
3723
- question: request.question,
3724
- intent: researchIntent,
3725
- context: researchContextEnvelope,
3726
- executionConnection: researchExecutionConnection,
3727
- executionConnectionName: researchExecutionConnectionName,
3728
- signal: request.signal,
3729
- baselineSql: agentRunString(researchSource?.sql),
3730
- baselineDqlArtifact: researchSource?.dqlArtifact,
3731
- baselineRunId: agentRunString(researchSource?.runId),
4250
+ payload: { researchRunId: created.id, notebookPath, branchCap: 6 },
3732
4251
  });
3733
- researchRun = withNotebookResearchChecklist(executed);
4252
+ // The root record is a plan/dossier parent. Only child runs are
4253
+ // observed research executions. This prevents one root result from
4254
+ // being copied into six fabricated ledger entries (AGT-016/033).
4255
+ // An explicit Research request still gets one real child when the
4256
+ // catalog planner has no grounded step (for example, an empty
4257
+ // starter project). The child is an observed metadata/baseline
4258
+ // attempt, not a fabricated successful finding; its durable status
4259
+ // and receipt determine the ledger entry.
4260
+ const fallbackBranch = {
4261
+ thought: 'Inspect the requested analytical question against the frozen root context.',
4262
+ action: {
4263
+ kind: 'lookup_metric',
4264
+ target: routeDecision?.resolvedAnalyticalPlan?.executionId ?? request.question,
4265
+ },
4266
+ expectation: 'Whether the frozen context contains enough evidence for a bounded answer.',
4267
+ };
4268
+ const branches = capResearchBranches(plan.steps.length > 0 ? plan.steps : [fallbackBranch], 6);
4269
+ // The replan edge. Each branch tests one hypothesis; folding its
4270
+ // outcome back into the state is what lets the investigation stop
4271
+ // when the question is settled instead of grinding through a plan
4272
+ // frozen before any observation. `nextHypothesis` returning
4273
+ // undefined is how the loop learns to stop — it enforces the hop
4274
+ // budget and reports when nothing is open.
4275
+ let researchState = createResearchState(request.question, branches.map((branch, position) => ({
4276
+ id: `h${position + 1}`,
4277
+ statement: branch.thought,
4278
+ priorConfidence: 1 - position / (branches.length + 1),
4279
+ })));
4280
+ for (let index = 0; index < branches.length; index += 1) {
4281
+ const step = branches[index];
4282
+ if (request.signal?.aborted)
4283
+ rethrowIfCancelled(request.signal.reason, request.signal);
4284
+ // A hypothesis an earlier finding already closed is not
4285
+ // re-investigated, and an exhausted hop budget stops the run.
4286
+ const stillOpen = nextHypothesis(researchState);
4287
+ if (!stillOpen) {
4288
+ emit({
4289
+ type: 'executor.started',
4290
+ message: `Stopping early: ${researchState.hopsUsed} of ${branches.length} branches settled what could be settled.`,
4291
+ route: 'research',
4292
+ });
4293
+ break;
4294
+ }
4295
+ const branchId = `${step.action.kind}:${step.action.target}`;
4296
+ const branchQuestion = `${request.question}\nResearch branch ${index + 1} (${branchId}): ${step.expectation}`;
4297
+ const childId = `${created.id}:research:${index + 1}`;
4298
+ const child = storage.createRun({
4299
+ id: childId,
4300
+ notebookPath,
4301
+ title: `${agentRunTitle(request.question, 'Agent research')} · branch ${index + 1}`,
4302
+ question: branchQuestion,
4303
+ sourceCell,
4304
+ sourceCellId,
4305
+ sourceCellName,
4306
+ sourceCellFingerprint,
4307
+ intent: researchIntent,
4308
+ domain: agentRunWorkspaceValue(request, 'domain'),
4309
+ owner: agentRunWorkspaceValue(request, 'owner'),
4310
+ context: {
4311
+ ...researchContextEnvelope,
4312
+ rootRunId: created.id,
4313
+ rootPlanId: plan.rootPlanId,
4314
+ branch: {
4315
+ id: branchId,
4316
+ index: index + 1,
4317
+ expectation: step.expectation,
4318
+ action: step.action,
4319
+ },
4320
+ },
4321
+ });
4322
+ emit({
4323
+ type: 'artifact.created',
4324
+ message: `Started research branch ${index + 1} of ${branches.length}.`,
4325
+ route: 'research',
4326
+ trustState: 'review_required',
4327
+ payload: { researchRunId: child.id, parentResearchRunId: created.id, branchId },
4328
+ });
4329
+ try {
4330
+ const executed = await runNotebookResearch(storage, child, {
4331
+ domain: agentRunWorkspaceValue(request, 'domain'),
4332
+ owner: agentRunWorkspaceValue(request, 'owner'),
4333
+ sourceCellFingerprint,
4334
+ question: branchQuestion,
4335
+ intent: researchIntent,
4336
+ context: {
4337
+ ...researchContextEnvelope,
4338
+ rootRunId: created.id,
4339
+ rootPlanId: plan.rootPlanId,
4340
+ branch: { id: branchId, index: index + 1, expectation: step.expectation, action: step.action },
4341
+ },
4342
+ executionConnection: researchExecutionConnection,
4343
+ executionConnectionName: researchExecutionConnectionName,
4344
+ signal: request.signal,
4345
+ baselineSql: agentRunString(researchSource?.sql),
4346
+ baselineDqlArtifact: researchSource?.dqlArtifact,
4347
+ baselineRunId: agentRunString(researchSource?.runId),
4348
+ });
4349
+ const branchRun = withNotebookResearchChecklist(executed);
4350
+ researchRuns.push(branchRun);
4351
+ // Observe, then decide. A branch that produced rows is evidence
4352
+ // for its hypothesis; one that did not is inconclusive, which is
4353
+ // a real outcome and not a failure.
4354
+ // Rows are not support. A branch that returned data has been
4355
+ // OBSERVED, not confirmed — deciding whether the observation
4356
+ // matches what the hypothesis predicted needs the expectation,
4357
+ // and nothing available at this layer can judge it. Recording
4358
+ // rows as `supports` would let the dossier report a driver the
4359
+ // evidence never established, which is the failure mode the
4360
+ // whole verified-fact chain exists to prevent.
4361
+ researchState = applyFinding(researchState, {
4362
+ id: `f${index + 1}`,
4363
+ hypothesisId: `h${index + 1}`,
4364
+ verdict: 'inconclusive',
4365
+ summary: branchRun.summary ?? '',
4366
+ strength: (branchRun.resultPreview?.rows?.length ?? 0) > 0
4367
+ ? 0.5
4368
+ : 0.1,
4369
+ });
4370
+ }
4371
+ catch (error) {
4372
+ // A child is a real durable run even when cancellation stops the
4373
+ // shared branch budget. Persist the truthful stop before
4374
+ // propagating cancellation to the parent run.
4375
+ const message = error instanceof Error ? error.message : String(error);
4376
+ storage.updateRun(child.id, {
4377
+ status: 'error',
4378
+ error: message,
4379
+ summary: 'Research branch stopped before producing a result.',
4380
+ reviewStatus: 'needs_review',
4381
+ });
4382
+ const stopped = storage.getRun(child.id);
4383
+ if (stopped)
4384
+ researchRuns.push(withNotebookResearchChecklist(stopped));
4385
+ researchState = applyFinding(researchState, {
4386
+ id: `f${index + 1}`,
4387
+ hypothesisId: `h${index + 1}`,
4388
+ verdict: 'inconclusive',
4389
+ summary: message,
4390
+ strength: 0,
4391
+ });
4392
+ rethrowIfCancelled(error, request.signal);
4393
+ }
4394
+ }
4395
+ researchRun = researchRuns[0];
3734
4396
  }
3735
4397
  finally {
3736
4398
  storage.close();
@@ -3741,15 +4403,81 @@ export async function startLocalServer(opts) {
3741
4403
  researchWorkspaceError = formatNotebookResearchStorageError(error);
3742
4404
  }
3743
4405
  }
3744
- const researchResultPreview = researchRun?.resultPreview;
4406
+ const researchResultPreview = researchRuns
4407
+ .map((run) => run.resultPreview)
4408
+ // Keep zero-row executions in the proof path. `coerceNarrateResultData`
4409
+ // intentionally omits empty row sets for prose, but an empty result is
4410
+ // still an executed result when it carries its execution identity.
4411
+ .find((preview) => {
4412
+ const record = agentRunRecord(preview);
4413
+ return Boolean(record && Array.isArray(record.rows));
4414
+ })
4415
+ ?? researchRun?.resultPreview;
3745
4416
  const researchResultData = coerceNarrateResultData(researchResultPreview);
3746
4417
  const researchResultRecord = agentRunRecord(researchResultPreview);
4418
+ // Keep each bounded research branch inspectable as an evidence ledger.
4419
+ // The notebook workspace remains the durable execution record; this
4420
+ // additive projection gives Ask/Research narration a stable list of
4421
+ // observations, failures, facts, and receipts without exposing provider
4422
+ // chain-of-thought. Six is the hard branch budget for one root question.
4423
+ const researchLedger = buildResearchEvidenceLedger({
4424
+ rootQuestion: request.question,
4425
+ planId: plan.rootPlanId,
4426
+ snapshotId: routeDecision?.resolvedAnalyticalPlan?.snapshotId,
4427
+ entries: researchRuns.slice(0, 6).map((branch, index) => {
4428
+ const branchContext = agentRunRecord(branch.context)?.branch;
4429
+ const branchPreviewRecord = agentRunRecord(branch.resultPreview);
4430
+ const branchPreview = coerceNarrateResultData(branchPreviewRecord);
4431
+ const previewRecord = branchPreviewRecord;
4432
+ const executionReceipt = normalizeAnalyticalExecutionReceipt(previewRecord?.executionReceipt);
4433
+ const resultFingerprint = normalizeAnalyticalExecutionFingerprint(previewRecord?.resultFingerprint)
4434
+ ?? executionReceipt?.resultFingerprint;
4435
+ // A child run ID or context-pack ID proves that planning happened,
4436
+ // not that a query executed. Only the canonical result fingerprint
4437
+ // (on the result or its execution receipt) can make a branch
4438
+ // observed (AGT-016/033).
4439
+ const executionProof = resultFingerprint;
4440
+ const observed = branch.status === 'ready' && Boolean(executionProof);
4441
+ return {
4442
+ id: branch.id,
4443
+ branchId: agentRunString(branchContext?.id) ?? `branch:${index + 1}`,
4444
+ question: branch.question,
4445
+ status: observed ? 'observed' : branch.status === 'error' ? 'failed' : 'skipped',
4446
+ ...(branchPreviewRecord && Array.isArray(branchPreviewRecord.rows)
4447
+ ? { rowCount: branchPreviewRecord.rows.length }
4448
+ : {}),
4449
+ ...(resultFingerprint ? { resultFingerprint } : {}),
4450
+ ...(executionReceipt ? { executionReceipt } : {}),
4451
+ facts: [agentRunString(branchContext?.expectation) ?? branch.question, ...(branch.evidence?.citations ?? []).flatMap((item) => typeof item === 'string' ? [item] : [])].slice(0, 8),
4452
+ // A context-pack ID or child run ID is not execution evidence. Keep
4453
+ // the receipt list empty until the execution service supplies a
4454
+ // receipt/fingerprint (AGT-016/033).
4455
+ receipts: executionProof ? [executionProof] : [],
4456
+ ...(!observed ? {
4457
+ error: branch.error
4458
+ ?? (branch.status === 'error'
4459
+ ? branch.summary
4460
+ : 'Research branch did not produce an execution receipt or result fingerprint.'),
4461
+ } : {}),
4462
+ };
4463
+ }),
4464
+ stoppingReason: needsClarification
4465
+ ? 'not_started'
4466
+ : researchRuns.some((run) => run.status === 'error')
4467
+ ? 'insufficient_evidence'
4468
+ : plan.steps.length > 6
4469
+ ? 'budget'
4470
+ : 'completed',
4471
+ });
3747
4472
  // A query that ran and matched 0 rows STILL executed — treat it as a clean,
3748
4473
  // grounded execution (not "no result"), so an empty answer is surfaced as
3749
4474
  // "0 rows matched" rather than silently downgraded to review-required.
3750
4475
  const researchDidExecute = Boolean(researchResultData) || Boolean(researchResultRecord && Array.isArray(researchResultRecord.rows));
3751
4476
  const researchZeroRows = !researchResultData && researchDidExecute;
3752
- const researchExecutedCleanly = researchDidExecute && !researchWorkspaceError && researchRun?.status !== 'error';
4477
+ const researchExecutedCleanly = researchDidExecute
4478
+ && !researchWorkspaceError
4479
+ && researchRuns.length > 0
4480
+ && researchRuns.every((run) => run.status === 'ready');
3753
4481
  const narration = !needsClarification && researchResultData
3754
4482
  ? await narrateForAgentRun({
3755
4483
  question: request.question,
@@ -3759,20 +4487,40 @@ export async function startLocalServer(opts) {
3759
4487
  reviewRequired: true,
3760
4488
  }, request.researchResultRowsOptIn === true)
3761
4489
  : undefined;
4490
+ // The cross-branch story. Every branch tested a hypothesis and produced a
4491
+ // finding; narrating only the one result the executor happened to carry
4492
+ // reported a single fact and discarded the rest, which is the visible
4493
+ // half of "research answers one question instead of telling a story".
4494
+ const researchStory = !needsClarification && plan.steps.length > 0
4495
+ ? synthesizeResearchNarrative({
4496
+ question: request.question,
4497
+ branches: researchRuns.map((branch, index) => ({
4498
+ statement: plan.steps[index]?.thought ?? branch.question ?? '',
4499
+ produced: branch.status === 'ready'
4500
+ && (branch.resultPreview?.rows?.length ?? 0) > 0,
4501
+ ...(branch.summary ? { summary: branch.summary } : {}),
4502
+ ...(branch.status ? { status: branch.status } : {}),
4503
+ })),
4504
+ })
4505
+ : undefined;
3762
4506
  const summary = needsClarification
3763
4507
  ? 'Needs clarification before running deeper research.'
3764
- : narration?.summary
3765
- ?? (researchZeroRows
3766
- ? 'The query executed cleanly against real data and matched 0 rows.'
3767
- : researchRun?.status === 'ready'
3768
- ? 'Saved a grounded research dossier with context evidence and next review actions.'
3769
- : researchRun?.status === 'error'
3770
- ? 'Saved a research dossier, but the preview needs review before promotion.'
3771
- : researchWorkspaceError
3772
- ? 'Prepared a grounded research plan; durable research storage is unavailable in this runtime.'
3773
- : plan.done
3774
- ? 'Prepared a direct grounded-answer plan.'
3775
- : 'Prepared a grounded research plan over real DQL assets.');
4508
+ // The story leads; the verified-fact narration follows it, so the
4509
+ // numbers still come from the narrator that checks them.
4510
+ : researchStory
4511
+ ? `${researchStory}${narration?.summary ? `\n\n${narration.summary}` : ''}`
4512
+ : narration?.summary
4513
+ ?? (researchZeroRows
4514
+ ? 'The query executed cleanly against real data and matched 0 rows.'
4515
+ : researchRun?.status === 'ready'
4516
+ ? 'Saved a grounded research dossier with context evidence and next review actions.'
4517
+ : researchRun?.status === 'error'
4518
+ ? 'Saved a research dossier, but the preview needs review before promotion.'
4519
+ : researchWorkspaceError
4520
+ ? 'Prepared a grounded research plan; durable research storage is unavailable in this runtime.'
4521
+ : plan.done
4522
+ ? 'Prepared a direct grounded-answer plan.'
4523
+ : 'Prepared a grounded research plan over real DQL assets.');
3776
4524
  return {
3777
4525
  summary,
3778
4526
  answer: plan.followUp?.question ?? narration?.summary
@@ -3785,8 +4533,11 @@ export async function startLocalServer(opts) {
3785
4533
  ? []
3786
4534
  : [agentRunArtifact('research_run', 'Research plan', {
3787
4535
  plan,
4536
+ researchLedger,
3788
4537
  researchRun,
4538
+ researchRuns,
3789
4539
  researchRunId: researchRun?.id,
4540
+ researchRunIds: researchRuns.map((run) => run.id),
3790
4541
  notebookPath,
3791
4542
  workspaceError: researchWorkspaceError,
3792
4543
  routeDecision,
@@ -3817,7 +4568,7 @@ export async function startLocalServer(opts) {
3817
4568
  : [
3818
4569
  ...(researchRun?.id ? [{ id: 'open-research', label: 'Open research dossier', artifactKind: 'research_run' }] : []),
3819
4570
  { id: 'create-block', label: 'Review DQL draft', route: 'dql_block_draft', artifactKind: 'dql_block_draft' },
3820
- ...(researchRun?.generatedSql || researchRun?.reviewedSql ? [{ id: 'insert-sql', label: 'Insert SQL preview', route: 'sql_cell', artifactKind: 'sql_cell' }] : []),
4571
+ ...(researchRuns.some((run) => run.generatedSql || run.reviewedSql) ? [{ id: 'insert-sql', label: 'Insert SQL preview', route: 'sql_cell', artifactKind: 'sql_cell' }] : []),
3821
4572
  ],
3822
4573
  };
3823
4574
  },
@@ -4117,6 +4868,22 @@ export async function startLocalServer(opts) {
4117
4868
  // lookup, and governed execution for the lifetime of a request. This removes
4118
4869
  // both positional catalog truncation and the previous duplicate retrieval pass.
4119
4870
  const preparedAgentContextPacks = new WeakMap();
4871
+ /**
4872
+ * Cross-encoder pass over the fused candidates, when a provider is available.
4873
+ * Advisory throughout: it may only reorder ids retrieval returned, and any
4874
+ * failure leaves retrieval's own ordering in place.
4875
+ */
4876
+ const agentRerankCandidates = (() => {
4877
+ const governed = resolveGovernedAnswerRunner(projectRoot);
4878
+ const provider = governed
4879
+ ? createGovernedTextProvider(governed.provider, projectRoot)
4880
+ : undefined;
4881
+ if (!provider)
4882
+ return undefined;
4883
+ return (question, candidates) => rerankCandidates(provider, question, candidates, {
4884
+ timeoutMs: Math.round(2_500 * deadlineScale()),
4885
+ });
4886
+ })();
4120
4887
  const pendingAgentContextPacks = new WeakMap();
4121
4888
  const buildAgentRunContextPack = async (request) => {
4122
4889
  const prepared = preparedAgentContextPacks.get(request);
@@ -4184,6 +4951,10 @@ export async function startLocalServer(opts) {
4184
4951
  },
4185
4952
  strictness: request.analysisDepth === 'deep' ? 'exploratory' : 'balanced',
4186
4953
  limit: request.analysisDepth === 'deep' ? 120 : 80,
4954
+ // The runtime PRE-BUILDS this pack, so wiring the reranker only at the
4955
+ // provider's own `buildLocalContextPack` left it unreachable on the
4956
+ // common path — the prepared pack is used and that call never happens.
4957
+ ...(agentRerankCandidates ? { rerankCandidates: agentRerankCandidates } : {}),
4187
4958
  domainContext: requestedDomain
4188
4959
  ? resolveUiDomainContext({
4189
4960
  manifest: snapshot.manifest,
@@ -4239,6 +5010,56 @@ export async function startLocalServer(opts) {
4239
5010
  };
4240
5011
  // Compact fallback used only for plain conversational replies. Analytical
4241
5012
  // turns use the structured, question-ranked evidence path above.
5013
+ /**
5014
+ * Explain a governed artifact the question names, from catalog metadata alone.
5015
+ *
5016
+ * Certified blocks are offered first: when a concept exists both as a
5017
+ * certified block and a raw model, the certified one is the authored
5018
+ * definition and the other is an implementation detail.
5019
+ */
5020
+ const buildGovernedObjectExplanation = (question) => {
5021
+ try {
5022
+ const blocks = collectPlanBlocks(projectRoot, { certifiedOnly: true });
5023
+ const certifiedNames = new Set(blocks.map((block) => block.name));
5024
+ const all = [
5025
+ ...blocks.map((block) => ({ block, status: 'certified' })),
5026
+ ...collectPlanBlocks(projectRoot, { certifiedOnly: false })
5027
+ .filter((block) => !certifiedNames.has(block.name))
5028
+ .map((block) => ({ block, status: 'draft' })),
5029
+ ];
5030
+ // Metrics as well as blocks. "How is revenue defined here?" names a
5031
+ // semantic metric, not a block, and answering it from the metric's own
5032
+ // description is the whole point of holding one.
5033
+ const metricObjects = loadSemanticMetrics(projectRoot).map((metric) => ({
5034
+ objectKey: `semantic:metric:${metric.name}`,
5035
+ objectType: 'semantic_metric',
5036
+ name: metric.name,
5037
+ ...(metric.description ? { description: metric.description } : {}),
5038
+ ...(metric.domain ? { domain: metric.domain } : {}),
5039
+ status: 'governed',
5040
+ payload: {},
5041
+ }));
5042
+ const explanation = composeBusinessExplanation(question, [
5043
+ ...all.map(({ block, status }) => ({
5044
+ objectKey: `dql:block:${block.name}`,
5045
+ objectType: 'dql_block',
5046
+ name: block.name,
5047
+ ...(block.description ? { description: block.description } : {}),
5048
+ ...(block.domain ? { domain: block.domain } : {}),
5049
+ status,
5050
+ payload: {
5051
+ ...(block.dimensions?.length ? { dimensions: block.dimensions } : {}),
5052
+ },
5053
+ })),
5054
+ ...metricObjects,
5055
+ ]);
5056
+ return explanation?.text;
5057
+ }
5058
+ catch {
5059
+ // Never let an explanation attempt break a conversational turn.
5060
+ return undefined;
5061
+ }
5062
+ };
4242
5063
  const buildAgentRunCatalogContext = () => {
4243
5064
  try {
4244
5065
  const blocks = collectPlanBlocks(projectRoot, { certifiedOnly: true });
@@ -4288,11 +5109,12 @@ export async function startLocalServer(opts) {
4288
5109
  },
4289
5110
  getCatalogContext: buildRankedAgentRunCatalogContext,
4290
5111
  });
4291
- // Hybrid router: keep the deterministic decision when it is confident (certified
4292
- // fast paths + greetings stay 0-LLM); spend one cheap classification call only for
4293
- // the ambiguous middle so Auto reliably picks quick-answer vs deep research without
4294
- // the user clicking "Dig deeper". Same provider completion as the planner.
5112
+ // Explicit selections and conversation-only turns remain deterministic;
5113
+ // each fresh natural-language analytical turn gets one bounded candidate-ID
5114
+ // interpretation call by default. The rollback is host-owned and cannot be
5115
+ // supplied by an HTTP/MCP client.
4295
5116
  const agentRunRouter = createHybridRouter({
5117
+ requireMeaningCallForNaturalLanguage: opts.requireMeaningCallForNaturalLanguage ?? true,
4296
5118
  complete: async ({ system, user, signal }) => {
4297
5119
  const provider = await createBlockStudioAssistProvider(projectRoot);
4298
5120
  if (!provider)
@@ -4332,6 +5154,25 @@ export async function startLocalServer(opts) {
4332
5154
  // A run may outlive its streaming browser connection, so cancellation is
4333
5155
  // server-owned and keyed by run id rather than relying on fetch abort alone.
4334
5156
  const activeAgentRunControllers = new Map();
5157
+ const cancelActiveAgentRun = (id) => {
5158
+ const controller = activeAgentRunControllers.get(id);
5159
+ if (!controller)
5160
+ return false;
5161
+ const progress = agentRunStore.getProgress(id);
5162
+ if (progress) {
5163
+ agentRunStore.saveProgress({
5164
+ ...progress,
5165
+ lifecycle: {
5166
+ ...progress.lifecycle,
5167
+ state: 'cancelling',
5168
+ revision: progress.lifecycle.revision + 1,
5169
+ updatedAt: new Date().toISOString(),
5170
+ },
5171
+ });
5172
+ }
5173
+ controller.abort(createAgentRunCancellationError());
5174
+ return true;
5175
+ };
4335
5176
  const activeAnalyticalRepairReservations = new Set();
4336
5177
  const consumedAnalyticalRepairCapabilities = new Set();
4337
5178
  // Server-side conversation threads: persisted multi-turn state (survives refresh).
@@ -4770,12 +5611,43 @@ export async function startLocalServer(opts) {
4770
5611
  * string, so the only remaining reasons to differ are the connection and the
4771
5612
  * governance gates, both of which report themselves honestly.
4772
5613
  */
4773
- const executeGeneratedSqlDirect = async (question, sql, seed, executionConnection, bindings, executionConnectionName) => {
5614
+ const executeGeneratedSqlDirect = async (question, sql, seed, executionConnection, bindings, executionConnectionName, agenticCapability, agenticScope) => {
4774
5615
  const activeConnection = requireActiveConnection(executionConnection);
4775
5616
  const rowBound = clampAnalyticalRowBound(seed?.limit ?? 200);
4776
5617
  const trimmed = sql.trim().replace(/;\s*$/, '').trim();
4777
5618
  if (!trimmed)
4778
5619
  throw analyticalError('The generated SQL was empty.', { origin: 'host', stage: 'execute' });
5620
+ const bindingValue = {
5621
+ sqlParams: bindings?.sqlParams ?? [],
5622
+ variables: bindings?.variables ?? {},
5623
+ };
5624
+ if (agenticCapability) {
5625
+ // The proposal itself is immutable. A changed literal, comment, or
5626
+ // whitespace is drift here rather than a benign formatting change.
5627
+ const currentSnapshotId = projectSnapshot().snapshotId;
5628
+ let targetFingerprint;
5629
+ if (agenticCapability.targetFingerprint) {
5630
+ try {
5631
+ targetFingerprint = (await observeWarehouseTargetIdentity(executor, activeConnection)).identityFingerprint;
5632
+ }
5633
+ catch {
5634
+ throw analyticalError('DQL could not re-confirm the selected execution target, so the analyst-approved query was not run.', {
5635
+ origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
5636
+ });
5637
+ }
5638
+ }
5639
+ const capabilityVerdict = verifyAgenticSqlExecutionCapability(agenticCapability, sql, {
5640
+ ...agenticScope,
5641
+ bindings: bindingValue,
5642
+ snapshotId: currentSnapshotId,
5643
+ ...(targetFingerprint ? { targetFingerprint } : {}),
5644
+ });
5645
+ if (!capabilityVerdict.ok) {
5646
+ throw analyticalError(capabilityVerdict.reason ?? 'The analyst-approved SQL no longer matches this execution.', {
5647
+ origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
5648
+ });
5649
+ }
5650
+ }
4779
5651
  // RESOLVE FIRST, VALIDATE SECOND. Executing verbatim means executing the
4780
5652
  // statement the notebook would execute — and the notebook resolves
4781
5653
  // `@metric()` / `@dim()` refs and dbt macros before it runs anything.
@@ -4800,8 +5672,6 @@ export async function startLocalServer(opts) {
4800
5672
  // release denied; a governance boundary must not move as a side effect.
4801
5673
  const sourceDomain = seed?.source?.match(/\bdomain\s*=\s*"([^"]+)"/i)?.[1] ?? 'uncategorized';
4802
5674
  assertAppAccess({ app, domain: sourceDomain, level: 'execute' });
4803
- // A semantic query the loop already compiled (MetricFlow / dbt Cloud) must
4804
- // execute through its pinned target binding, not as loose SQL.
4805
5675
  const semanticExecutionHolder = { value: null };
4806
5676
  const execution = await analyticalExecutionService.execute({
4807
5677
  sql: semantic.sql,
@@ -4813,27 +5683,46 @@ export async function startLocalServer(opts) {
4813
5683
  variables: bindings?.variables,
4814
5684
  semanticRefs: semantic.semanticRefs,
4815
5685
  executePrepared: async (preparation) => {
4816
- const pinnedSemanticCompile = compiledSemanticQueries.get(executionFingerprint(preparation.preparedSql))
4817
- ?? compiledSemanticQueries.get(executionFingerprint(trimmed));
4818
- if (pinnedSemanticCompile) {
4819
- const semanticExecution = await executeTargetBoundSemanticQuery({
4820
- executor,
4821
- connection: activeConnection,
4822
- projectRoot,
4823
- plannedAdapter: pinnedSemanticCompile.engine,
4824
- metricFlow: pinnedSemanticCompile.engine === 'metricflow-cli'
4825
- ? resolveMetricFlowTargetMetadata(projectRoot, projectConfig)
4826
- : undefined,
4827
- compile: async () => pinnedSemanticCompile,
4828
- prepareSql: () => ({ sql: preparation.executedSql, connection: preparation.connection }),
4829
- rowBound,
4830
- });
4831
- if (semanticExecution) {
4832
- semanticExecutionHolder.value = semanticExecution;
4833
- return semanticExecution.result;
5686
+ // Both the native executor and the target-bound semantic adapter can
5687
+ // reach the warehouse from this callback. Admit the exact prepared
5688
+ // bytes before either path so a matching cached semantic compile cannot
5689
+ // bypass the generated proposal's capability.
5690
+ return executePreparedAgenticSqlBoundary({
5691
+ capability: agenticCapability,
5692
+ preparedSql: preparation.executedSql,
5693
+ bindings: bindingValue,
5694
+ scope: {
5695
+ ...agenticScope,
5696
+ snapshotId: projectSnapshot().snapshotId,
5697
+ ...(agenticCapability ? { targetFingerprint: agenticCapability.targetFingerprint } : {}),
5698
+ },
5699
+ execute: async () => {
5700
+ // A semantic query the loop already compiled (MetricFlow / dbt
5701
+ // Cloud) must execute through its pinned target binding, not as
5702
+ // loose SQL.
5703
+ const pinnedSemanticCompile = compiledSemanticQueries.get(executionFingerprint(preparation.preparedSql))
5704
+ ?? compiledSemanticQueries.get(executionFingerprint(trimmed));
5705
+ if (pinnedSemanticCompile) {
5706
+ const semanticExecution = await executeTargetBoundSemanticQuery({
5707
+ executor,
5708
+ connection: activeConnection,
5709
+ projectRoot,
5710
+ plannedAdapter: pinnedSemanticCompile.engine,
5711
+ metricFlow: pinnedSemanticCompile.engine === 'metricflow-cli'
5712
+ ? resolveMetricFlowTargetMetadata(projectRoot, projectConfig)
5713
+ : undefined,
5714
+ compile: async () => pinnedSemanticCompile,
5715
+ prepareSql: () => ({ sql: preparation.executedSql, connection: preparation.connection }),
5716
+ rowBound,
5717
+ });
5718
+ if (semanticExecution) {
5719
+ semanticExecutionHolder.value = semanticExecution;
5720
+ return semanticExecution.result;
5721
+ }
5722
+ }
5723
+ return executor.executeQuery(preparation.executedSql, bindings?.sqlParams ?? [], runtimeVariables(bindings?.variables ?? {}), preparation.connection);
4834
5724
  }
4835
- }
4836
- return executor.executeQuery(preparation.executedSql, bindings?.sqlParams ?? [], runtimeVariables(bindings?.variables ?? {}), preparation.connection);
5725
+ });
4837
5726
  },
4838
5727
  });
4839
5728
  const semanticExecution = semanticExecutionHolder.value;
@@ -4891,14 +5780,19 @@ export async function startLocalServer(opts) {
4891
5780
  executableArtifact,
4892
5781
  };
4893
5782
  };
4894
- const executeGeneratedArtifactForAgent = async (question, sql, seed, executionConnection, executionConnectionName) => {
5783
+ const executeGeneratedArtifactForAgent = async (question, sql, seed, executionConnection, executionConnectionName, agenticCapability, agenticScope) => {
4895
5784
  // A seed that is already a certified/saved artifact keeps the DQL-first
4896
5785
  // path: there the `.dql` source IS the contract, and its parameters and
4897
5786
  // semantic refs must be compiled, not bypassed.
4898
5787
  if (seed && seed.kind !== 'sql_block') {
5788
+ if (agenticCapability) {
5789
+ throw analyticalError('The analyst-approved SQL cannot be redirected through a saved artifact.', {
5790
+ origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
5791
+ });
5792
+ }
4899
5793
  return executeArtifactReferenceForAgent({ ...seed, limit: seed.limit ?? 200 }, question, executionConnection, executionConnectionName);
4900
5794
  }
4901
- return executeGeneratedSqlDirect(question, sql, seed, executionConnection, undefined, executionConnectionName);
5795
+ return executeGeneratedSqlDirect(question, sql, seed, executionConnection, undefined, executionConnectionName, agenticCapability, agenticScope);
4902
5796
  };
4903
5797
  /**
4904
5798
  * EXP-001: execution host for the deliberately narrow non-governed lane.
@@ -5506,7 +6400,10 @@ export async function startLocalServer(opts) {
5506
6400
  let invocation;
5507
6401
  let plan;
5508
6402
  try {
5509
- const program = new Parser(repairedSource, '<bounded-dql-repair>').parse();
6403
+ // The source label is echoed into parse errors, which reach the user.
6404
+ // `<bounded-dql-repair>` is an internal artifact id and tells them nothing;
6405
+ // it appeared verbatim in a reported failure card.
6406
+ const program = new Parser(repairedSource, 'repaired query').parse();
5510
6407
  const blocks = program.statements.filter((statement) => statement.kind === NodeKind.BlockDecl);
5511
6408
  if (blocks.length !== 1 || program.statements.length !== 1) {
5512
6409
  throw new Error('The repaired source must contain exactly one DQL block.');
@@ -6113,6 +7010,19 @@ export async function startLocalServer(opts) {
6113
7010
  : notebookResearchString(governedAnswer?.answer)
6114
7011
  ?? notebookResearchString(governedAnswer?.text)
6115
7012
  ?? notebookResearchSummary(question, resultPreview, previewError);
7013
+ const previewRecord = agentRunRecord(resultPreview);
7014
+ const executionReceipt = normalizeAnalyticalExecutionReceipt(previewRecord?.executionReceipt);
7015
+ // Do not treat a child run ID as execution evidence. The canonical
7016
+ // fingerprint is the only proof that this Research branch produced a
7017
+ // result; planning, SQL text, and a durable run record remain review
7018
+ // required until that proof exists (AGT-016/033).
7019
+ const executionProof = normalizeAnalyticalExecutionFingerprint(previewRecord?.resultFingerprint)
7020
+ ?? executionReceipt?.resultFingerprint;
7021
+ const executionUnavailable = !previewError && !executionProof;
7022
+ const terminalError = previewError
7023
+ ?? (executionUnavailable
7024
+ ? 'Research did not produce an executed result or execution receipt; the branch remains review-required.'
7025
+ : undefined);
6116
7026
  const recommendation = previewError
6117
7027
  ? 'Review the SQL, selected metadata, and connection context before rerunning.'
6118
7028
  : dqlArtifact && !reviewedSql
@@ -6137,8 +7047,14 @@ export async function startLocalServer(opts) {
6137
7047
  question,
6138
7048
  intent,
6139
7049
  context,
6140
- status: previewError ? 'error' : 'ready',
6141
- summary,
7050
+ // A plan, generated SQL, or DQL artifact is not an observed execution.
7051
+ // Only a result carrying an execution fingerprint/receipt may become
7052
+ // `ready`; otherwise persist a failed review state with no fabricated
7053
+ // evidence (AGT-016/033).
7054
+ status: terminalError ? 'error' : 'ready',
7055
+ summary: terminalError && executionUnavailable
7056
+ ? 'Research branch stopped without an executed result; no observation was recorded.'
7057
+ : summary,
6142
7058
  recommendation,
6143
7059
  resultPreview,
6144
7060
  evidence,
@@ -6158,7 +7074,7 @@ export async function startLocalServer(opts) {
6158
7074
  ...(display && display.ok ? display.warnings : []),
6159
7075
  ],
6160
7076
  reviewStatus: 'needs_review',
6161
- error: previewError,
7077
+ error: terminalError,
6162
7078
  lastRunAt: startedAt,
6163
7079
  }) ?? run;
6164
7080
  }
@@ -6985,7 +7901,12 @@ export async function startLocalServer(opts) {
6985
7901
  return;
6986
7902
  }
6987
7903
  if (operationMatch && req.method === 'DELETE') {
6988
- const operation = operationCoordinator.cancel(decodeURIComponent(operationMatch[1]));
7904
+ const operationId = decodeURIComponent(operationMatch[1]);
7905
+ const existingOperation = operationCoordinator.get(operationId);
7906
+ if (existingOperation?.type === 'agent_run' && existingOperation.scope.startsWith('agent-run:')) {
7907
+ cancelActiveAgentRun(existingOperation.scope.slice('agent-run:'.length));
7908
+ }
7909
+ const operation = operationCoordinator.cancel(operationId);
6989
7910
  res.writeHead(operation ? 200 : 404, { 'Content-Type': 'application/json; charset=utf-8' });
6990
7911
  res.end(serializeJSON(operation ?? { error: 'Operation not found.' }));
6991
7912
  return;
@@ -8098,12 +9019,17 @@ export async function startLocalServer(opts) {
8098
9019
  const limit = Number.isFinite(rawLimit) && rawLimit > 0
8099
9020
  ? Math.min(200, Math.floor(rawLimit))
8100
9021
  : 50;
8101
- const runs = agentRunStore
8102
- .list()
9022
+ const stored = agentRunStore.list();
9023
+ // This is an INDEX payload: one row per run, no answer bodies. Shipping
9024
+ // stored runs whole meant a measured 47.61 MB for 20 real runs.
9025
+ // `GET /api/agent-runs/:id` still serves the complete immutable record.
9026
+ const runs = stored
9027
+ .slice()
8103
9028
  .sort((a, b) => b.startedAt.localeCompare(a.startedAt))
8104
- .slice(0, limit);
9029
+ .slice(0, limit)
9030
+ .map(agentRunListEntryForTransport);
8105
9031
  res.writeHead(200, { 'Content-Type': 'application/json; charset=utf-8' });
8106
- res.end(serializeJSON({ runs, total: agentRunStore.list().length, limit }));
9032
+ res.end(serializeJSON({ runs, total: stored.length, limit }));
8107
9033
  return;
8108
9034
  }
8109
9035
  /**
@@ -8641,6 +9567,14 @@ export async function startLocalServer(opts) {
8641
9567
  res.end(serializeJSON({ error: 'Agent run not found.' }));
8642
9568
  return;
8643
9569
  }
9570
+ if (run.status !== 'blocked' || run.stopReason !== 'blocked') {
9571
+ res.writeHead(409, { 'Content-Type': 'application/json; charset=utf-8' });
9572
+ res.end(serializeJSON({
9573
+ code: 'REPAIR_CAPABILITY_REQUIRED',
9574
+ error: 'Only a terminal blocked run with a blocked stop reason may derive an analytical repair.',
9575
+ }));
9576
+ return;
9577
+ }
8644
9578
  const source = analyticalFailedRunFromAgentRun(run);
8645
9579
  if (!source) {
8646
9580
  res.writeHead(409, { 'Content-Type': 'application/json; charset=utf-8' });
@@ -8679,25 +9613,11 @@ export async function startLocalServer(opts) {
8679
9613
  if (req.method === 'POST' && /^\/api\/agent-runs\/[^/]+\/cancel$/.test(path)) {
8680
9614
  const match = path.match(/^\/api\/agent-runs\/([^/]+)\/cancel$/);
8681
9615
  const id = decodeURIComponent(match?.[1] ?? '');
8682
- const controller = activeAgentRunControllers.get(id);
8683
- if (!controller) {
9616
+ if (!cancelActiveAgentRun(id)) {
8684
9617
  res.writeHead(404, { 'Content-Type': 'application/json; charset=utf-8' });
8685
9618
  res.end(serializeJSON({ ok: false, error: 'This run is no longer active.' }));
8686
9619
  return;
8687
9620
  }
8688
- const progress = agentRunStore.getProgress(id);
8689
- if (progress) {
8690
- agentRunStore.saveProgress({
8691
- ...progress,
8692
- lifecycle: {
8693
- ...progress.lifecycle,
8694
- state: 'cancelling',
8695
- revision: progress.lifecycle.revision + 1,
8696
- updatedAt: new Date().toISOString(),
8697
- },
8698
- });
8699
- }
8700
- controller.abort(new Error('Stopped by user.'));
8701
9621
  res.writeHead(202, { 'Content-Type': 'application/json; charset=utf-8' });
8702
9622
  res.end(serializeJSON({ ok: true, id }));
8703
9623
  return;
@@ -8842,7 +9762,10 @@ export async function startLocalServer(opts) {
8842
9762
  return;
8843
9763
  }
8844
9764
  res.writeHead(201, { 'Content-Type': 'application/json; charset=utf-8' });
8845
- res.end(serializeJSON({ run: completedRun }));
9765
+ // Same run, same reader as the stream above, so it ships the same
9766
+ // projection. Sending the stored record here instead made an ordinary
9767
+ // completed Ask a 4.66 MB reply.
9768
+ res.end(serializeJSON({ run: slimAgentRunForTransport(completedRun) }));
8846
9769
  }
8847
9770
  finally {
8848
9771
  if (runId)
@@ -17527,21 +18450,31 @@ function connectionDriverLabel(connection) {
17527
18450
  * The notebook SPA expects columns as string[] (just names).
17528
18451
  */
17529
18452
  function normalizeQueryResult(result, semanticRefs) {
17530
- const rawCols = Array.isArray(result?.columns) ? result.columns : [];
17531
- const columns = rawCols.map((c) => typeof c === 'string' ? c : typeof c?.name === 'string' ? c.name : String(c));
18453
+ const canonical = normalizeCanonicalQueryResult({
18454
+ columns: result?.columns,
18455
+ rows: result?.rows,
18456
+ rowCount: result?.rowCount,
18457
+ executionTime: result?.executionTime,
18458
+ executionTimeMs: result?.executionTimeMs,
18459
+ truncated: result?.truncated,
18460
+ resultFingerprint: result?.resultFingerprint,
18461
+ executionReceipt: result?.executionReceipt,
18462
+ trustState: result?.trustState,
18463
+ answerTier: result?.answerTier,
18464
+ });
17532
18465
  const rawRows = Array.isArray(result?.rows) ? result.rows : [];
17533
- const rows = rawRows.slice(0, NOTEBOOK_EXECUTE_PREVIEW_ROW_LIMIT);
18466
+ const rows = canonical.rows.slice(0, NOTEBOOK_EXECUTE_PREVIEW_ROW_LIMIT);
17534
18467
  const hasRefs = semanticRefs && (semanticRefs.metrics.length > 0 || semanticRefs.dimensions.length > 0);
17535
18468
  return {
17536
- columns,
18469
+ columns: canonical.columns,
17537
18470
  rows,
17538
- rowCount: typeof result?.rowCount === 'number' ? result.rowCount : rawRows.length,
17539
- executionTime: typeof result?.executionTimeMs === 'number'
17540
- ? result.executionTimeMs
17541
- : typeof result?.executionTime === 'number'
17542
- ? result.executionTime
17543
- : 0,
17544
- ...(rawRows.length > rows.length ? { truncated: true } : {}),
18471
+ rowCount: canonical.rowCount,
18472
+ resultFingerprint: canonical.resultFingerprint,
18473
+ executionTime: canonical.executionTime ?? 0,
18474
+ ...(rawRows.length > rows.length || canonical.truncated ? { truncated: true } : {}),
18475
+ ...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
18476
+ ...(canonical.trustState ? { trustState: canonical.trustState } : {}),
18477
+ ...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
17545
18478
  ...(hasRefs ? { semanticRefs } : {}),
17546
18479
  };
17547
18480
  }
@@ -25400,7 +26333,13 @@ async function createBlockStudioAssistProvider(projectRoot, requestedProvider) {
25400
26333
  default:
25401
26334
  return null;
25402
26335
  }
25403
- return await provider.available() ? provider : null;
26336
+ // Route through the eval cassette too. This constructor serves the MEANING
26337
+ // call and narration — the two dispatches that decide routing and wording —
26338
+ // so leaving it unwrapped meant a recorded suite still hit a live model for
26339
+ // exactly the calls whose non-determinism it was recorded to remove. A local
26340
+ // baseline reproduced that: the same question blocked on one run and answered
26341
+ // on the next, and zero cassettes were written.
26342
+ return await provider.available() ? applyEvalCassette(provider) : null;
25404
26343
  }
25405
26344
  /** Convert a governed answer's result payload into a bounded synthesis preview. */
25406
26345
  function agentResultToSynthesisPreview(result) {
@@ -28237,7 +29176,45 @@ async function buildAgentSchemaContextFromCatalog(projectRoot, question, prepare
28237
29176
  const RUNTIME_SNAPSHOT_MAX_AGE_MS = 60 * 60 * 1000; // 1 hour
28238
29177
  // A resolver compares at most 12 compact cards and never performs tool calls;
28239
29178
  // ten seconds is the full allowance, not the start of another planning loop.
28240
- const AGENT_MEANING_TIMEOUT_MS = 10_000;
29179
+ /**
29180
+ * Ceiling on the one bounded meaning-resolution call.
29181
+ *
29182
+ * 10s assumes a hosted model. A local Ollama model needs ~7s for a ONE-WORD
29183
+ * reply, so a 600-token resolution over a dozen candidates never lands: it
29184
+ * aborts, the router falls back to its evidence-only decision, and
29185
+ * `mayAssumeInterpretation` goes false — which sends every ambiguous question to
29186
+ * the clarification gate (AGT-017). The effect is that a local model cannot
29187
+ * answer anything ambiguous, in a product whose whole positioning is local-first.
29188
+ *
29189
+ * Scaled by the same `DQL_AGENT_DEADLINE_SCALE` as the run budget, so one
29190
+ * setting moves the provider's whole time envelope together rather than leaving
29191
+ * an inner bound to silently cap an outer one.
29192
+ */
29193
+ /**
29194
+ * Predict how long the next provider call will take, for admission control.
29195
+ *
29196
+ * With fewer than three samples the MAX is the only honest predictor: there is
29197
+ * no distribution yet, and admitting a call the deadline then kills wastes the
29198
+ * whole remaining budget.
29199
+ *
29200
+ * With a real sample, p75 rather than the max. One slow response — a cold model
29201
+ * load, a retried connection — otherwise poisons admission control for the rest
29202
+ * of the run: every later call is refused against a worst case that already
29203
+ * passed. A recorded run tripped RUN_DEADLINE_INSUFFICIENT 6.4s into a 45s
29204
+ * budget for exactly that reason. p75 still errs slow, so a genuinely slow
29205
+ * provider is still respected.
29206
+ */
29207
+ export function predictDispatchMs(observed, assumedMs = ASSUMED_PROVIDER_DISPATCH_MS) {
29208
+ if (observed.length === 0)
29209
+ return assumedMs;
29210
+ const sorted = [...observed].sort((left, right) => left - right);
29211
+ if (sorted.length < 3)
29212
+ return sorted[sorted.length - 1];
29213
+ const index = Math.max(0, Math.min(sorted.length - 1, Math.ceil(sorted.length * 0.75) - 1));
29214
+ return sorted[index];
29215
+ }
29216
+ const AGENT_MEANING_TIMEOUT_BASE_MS = 10_000;
29217
+ const AGENT_MEANING_TIMEOUT_MS = AGENT_MEANING_TIMEOUT_BASE_MS * deadlineScale();
28241
29218
  export function boundedAgentMeaningSignal(signal, timeoutMs = AGENT_MEANING_TIMEOUT_MS) {
28242
29219
  const timeout = AbortSignal.timeout(Math.max(1, timeoutMs));
28243
29220
  return signal ? AbortSignal.any([signal, timeout]) : timeout;
@@ -28888,26 +29865,25 @@ function notebookResearchContextPreview(contextPack) {
28888
29865
  };
28889
29866
  }
28890
29867
  function normalizeNotebookAgentResult(result) {
28891
- const columns = Array.isArray(result.columns)
28892
- ? result.columns.map((column) => {
28893
- if (typeof column === 'string')
28894
- return column;
28895
- if (column && typeof column === 'object' && typeof column.name === 'string') {
28896
- return String(column.name);
28897
- }
28898
- return String(column);
28899
- })
28900
- : [];
28901
- const rows = Array.isArray(result.rows)
28902
- ? result.rows
28903
- .filter((row) => Boolean(row && typeof row === 'object' && !Array.isArray(row)))
28904
- .map((row) => row)
28905
- : [];
29868
+ const canonical = normalizeCanonicalQueryResult({
29869
+ columns: result.columns,
29870
+ rows: result.rows,
29871
+ rowCount: result.rowCount,
29872
+ executionTime: result.executionTime,
29873
+ resultFingerprint: result.resultFingerprint,
29874
+ executionReceipt: result.executionReceipt,
29875
+ trustState: result.executableArtifact?.trustState,
29876
+ answerTier: result.answerTier,
29877
+ });
28906
29878
  return {
28907
- columns,
28908
- rows,
28909
- rowCount: typeof result.rowCount === 'number' ? result.rowCount : rows.length,
28910
- executionTime: typeof result.executionTime === 'number' ? result.executionTime : 0,
29879
+ columns: canonical.columns,
29880
+ rows: canonical.rows,
29881
+ rowCount: canonical.rowCount,
29882
+ resultFingerprint: canonical.resultFingerprint,
29883
+ executionTime: canonical.executionTime ?? 0,
29884
+ ...(canonical.truncated ? { truncated: true } : {}),
29885
+ ...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
29886
+ ...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
28911
29887
  };
28912
29888
  }
28913
29889
  function notebookResearchSummary(question, result, error) {
@@ -29623,49 +30599,9 @@ function scoreAgentValueProbeColumn(table, column) {
29623
30599
  return score;
29624
30600
  }
29625
30601
  export function isAgentValueProbeColumn(column) {
29626
- const name = column.name.toLowerCase();
29627
- // Tokenize underscore/camel names before applying the hard deny-list. This is
29628
- // intentionally independent of an allowlist: secrets and free-text payloads
29629
- // can never be probed through automatic grounding.
29630
- const normalizedName = column.name
29631
- .replace(/([a-z0-9])([A-Z])/g, '$1 $2')
29632
- .replace(/[_-]+/g, ' ')
29633
- .toLowerCase();
29634
- if (/\b(password|secret|token|credential|hash|salt|notes?|comments?|description|message|body|payload|content)\b/.test(normalizedName))
29635
- return false;
29636
- if (/\bemail\b/.test(normalizedName))
29637
- return false;
29638
- if (!hasAgentSchemaToken(name, [
29639
- 'account',
29640
- 'category',
29641
- 'channel',
29642
- 'city',
29643
- 'code',
29644
- 'country',
29645
- 'customer',
29646
- 'email',
29647
- 'full',
29648
- 'id',
29649
- 'key',
29650
- 'member',
29651
- 'name',
29652
- 'number',
29653
- 'product',
29654
- 'region',
29655
- 'segment',
29656
- 'sku',
29657
- 'state',
29658
- 'status',
29659
- 'subscriber',
29660
- 'type',
29661
- 'user',
29662
- ])) {
29663
- return false;
29664
- }
29665
- const type = column.type?.toLowerCase() ?? '';
29666
- if (!type)
29667
- return true;
29668
- return /\b(char|character|clob|email|string|text|uuid|varchar)\b/.test(type);
30602
+ // Delegates to the canonical predicate in dql-agent. Two copies of a security
30603
+ // rule drift, and the one that drifts is the one nobody is looking at.
30604
+ return isProbeSafeColumn({ name: column.name, ...(column.type ? { type: column.type } : {}) });
29669
30605
  }
29670
30606
  export function buildAgentValueProbeSql(table, column, searchTerms, connection) {
29671
30607
  const relation = quoteAgentRelation(table.relation, connection);