@duckcodeailabs/dql-cli 1.13.5 → 1.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/args.d.ts +17 -0
- package/dist/args.d.ts.map +1 -1
- package/dist/args.js +30 -0
- package/dist/args.js.map +1 -1
- package/dist/assets/dql-notebook/assets/{AgentLogPage-BzQOKjyV.js → AgentLogPage-BPz-UWFh.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildDialog-CTxha499.js → AiBuildDialog-BEl53WA_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiBuildResult--MuD_I6g.js → AiBuildResult-B4yfGTTZ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AiSidePanel-DcI4PMJ1.js → AiSidePanel-CSZAAvuD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AnalyticsHome-Dq54vjNz.js → AnalyticsHome-D5P6Ujwi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{AppsView-0SlwWYex.js → AppsView-DQOwU9Cg.js} +4 -4
- package/dist/assets/dql-notebook/assets/{BlockStudio-CxxXZD3M.js → BlockStudio-B4ap0GdY.js} +1 -1
- package/dist/assets/dql-notebook/assets/{BusinessArtifactView-DxdAmeUN.js → BusinessArtifactView-BIfNI-0S.js} +1 -1
- package/dist/assets/dql-notebook/assets/{DbtFirstModelingPage-y81gFg_b.js → DbtFirstModelingPage-D72byb2g.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GitPage-GQtcwncb.js → GitPage-lQb1uXH2.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GlobalAiRail-DM4wkxR_.js → GlobalAiRail-CVECf6Xj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{GovernedContextPage-B8Ft6JES.js → GovernedContextPage-trOyMCY6.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HelpDocsPage-DnD4nMuF.js → HelpDocsPage-D8hLS5lE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{HomePage-Ehb0ITj-.js → HomePage-eIkfBIep.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDAG-Lvjc2AQX.js → LineageDAG-CSqcbDrE.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDetailView-DX32pfbp.js → LineageDetailView-DJZZjZu-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineageDrawer-CFq3jPfk.js → LineageDrawer-BcIipIc3.js} +1 -1
- package/dist/assets/dql-notebook/assets/{LineagePathBreadcrumb-CVHXuHls.js → LineagePathBreadcrumb-CviIf8PN.js} +1 -1
- package/dist/assets/dql-notebook/assets/{MiniLineageGraph-BKaeT5Kt.js → MiniLineageGraph-vH_MY_Ju.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewBlockModal-DI-JDzFH.js → NewBlockModal-DbCQg-pj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NewNotebookModal-Csw6eF74.js → NewNotebookModal-5Vp6XiuK.js} +1 -1
- package/dist/assets/dql-notebook/assets/{NotebookEditor-FEqs3789.js → NotebookEditor-DjqJS44s.js} +1 -1
- package/dist/assets/dql-notebook/assets/{ReadinessPage-Bk0S_MEA.js → ReadinessPage-BHGrC5ho.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SetupOnboarding-DyLaPFGn.js → SetupOnboarding-BfS9Tdx-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{SkillsPage-DLHyvhci.js → SkillsPage-BFH21wSj.js} +1 -1
- package/dist/assets/dql-notebook/assets/{TrustBadge-CIDLj2A6.js → TrustBadge-zm6g_SxZ.js} +1 -1
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel--oxmjlgr.js +88 -0
- package/dist/assets/dql-notebook/assets/{answer-to-notebook-CFEJHLvs.js → answer-to-notebook-DPhxIEzF.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-left-B-Zdcyvm.js → arrow-left-DNEb86Xc.js} +1 -1
- package/dist/assets/dql-notebook/assets/{arrow-right-Cn8TM7cp.js → arrow-right-DpPWbwaD.js} +1 -1
- package/dist/assets/dql-notebook/assets/{book-open-text-Bko2NNUs.js → book-open-text-D7s5jo4X.js} +1 -1
- package/dist/assets/dql-notebook/assets/{circle-x-AWJAmUBB.js → circle-x-Db3dXKg7.js} +1 -1
- package/dist/assets/dql-notebook/assets/{dagre.esm-Cl_ucrRh.js → dagre.esm-C7pppQ1a.js} +1 -1
- package/dist/assets/dql-notebook/assets/{external-link-DN57tb5f.js → external-link-BxwXitO_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{grip-vertical-BOWwFFva.js → grip-vertical-Dt4lkWRi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{index-Ck-wqvV2.js → index-DKo-bwNw.js} +4 -4
- package/dist/assets/dql-notebook/assets/{link-2-xUhbfBs8.js → link-2-Dfo2P6wi.js} +1 -1
- package/dist/assets/dql-notebook/assets/{list-tree-BDWcBr67.js → list-tree-DiTmIWAL.js} +1 -1
- package/dist/assets/dql-notebook/assets/{minimize-2-CEeNoMTS.js → minimize-2-CMTAkPzL.js} +1 -1
- package/dist/assets/dql-notebook/assets/{panel-right-open-DnSeUvOY.js → panel-right-open-DwYr7FW4.js} +1 -1
- package/dist/assets/dql-notebook/assets/{play-EoDj8-fa.js → play-BXhHYQ4x.js} +1 -1
- package/dist/assets/dql-notebook/assets/{rotate-ccw-BAMxWT9A.js → rotate-ccw-BNi6F8pl.js} +1 -1
- package/dist/assets/dql-notebook/assets/{semantic-fields-BwLOs1kl.js → semantic-fields-CNOGysAy.js} +1 -1
- package/dist/assets/dql-notebook/assets/{sliders-horizontal-BYkm8SWW.js → sliders-horizontal-Ec5MUMUW.js} +1 -1
- package/dist/assets/dql-notebook/assets/{star-sVexqNCs.js → star-B9leDkp_.js} +1 -1
- package/dist/assets/dql-notebook/assets/{triangle-alert-D2njv4o4.js → triangle-alert-BefTYCzx.js} +1 -1
- package/dist/assets/dql-notebook/assets/{upload-CXTMxIA8.js → upload-SPiOM2tQ.js} +1 -1
- package/dist/assets/dql-notebook/assets/{usePersistedAgentThreadId-CPDeaAgL.js → usePersistedAgentThreadId-C4foXeiQ.js} +4 -4
- package/dist/assets/dql-notebook/assets/{user-round-zVsI_uud.js → user-round-comGmyw-.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wand-sparkles-BMIwIzXT.js → wand-sparkles-BffR4dF8.js} +1 -1
- package/dist/assets/dql-notebook/assets/{workflow-Dj9y1sNk.js → workflow-ChmPTEzH.js} +1 -1
- package/dist/assets/dql-notebook/assets/{wrench-DPQi6zrs.js → wrench-DovaG_ze.js} +1 -1
- package/dist/assets/dql-notebook/assets/{x-XhhbtinL.js → x-nRx91AgW.js} +1 -1
- package/dist/assets/dql-notebook/index.html +1 -1
- package/dist/commands/agent-eval-cassette.d.ts +73 -0
- package/dist/commands/agent-eval-cassette.d.ts.map +1 -0
- package/dist/commands/agent-eval-cassette.js +170 -0
- package/dist/commands/agent-eval-cassette.js.map +1 -0
- package/dist/commands/agent-eval-runtime.d.ts +97 -0
- package/dist/commands/agent-eval-runtime.d.ts.map +1 -0
- package/dist/commands/agent-eval-runtime.js +155 -0
- package/dist/commands/agent-eval-runtime.js.map +1 -0
- package/dist/commands/agent.d.ts +115 -1
- package/dist/commands/agent.d.ts.map +1 -1
- package/dist/commands/agent.js +289 -62
- package/dist/commands/agent.js.map +1 -1
- package/dist/commands/eval.d.ts +15 -1
- package/dist/commands/eval.d.ts.map +1 -1
- package/dist/commands/eval.js +32 -3
- package/dist/commands/eval.js.map +1 -1
- package/dist/index.js +5 -1
- package/dist/index.js.map +1 -1
- package/dist/llm/analyst-loop-tools.d.ts +19 -0
- package/dist/llm/analyst-loop-tools.d.ts.map +1 -0
- package/dist/llm/analyst-loop-tools.js +56 -0
- package/dist/llm/analyst-loop-tools.js.map +1 -0
- package/dist/llm/providers/dql-agent-provider.d.ts +24 -1
- package/dist/llm/providers/dql-agent-provider.d.ts.map +1 -1
- package/dist/llm/providers/dql-agent-provider.js +343 -10
- package/dist/llm/providers/dql-agent-provider.js.map +1 -1
- package/dist/llm/types.d.ts +19 -1
- package/dist/llm/types.d.ts.map +1 -1
- package/dist/local-runtime.d.ts +132 -2
- package/dist/local-runtime.d.ts.map +1 -1
- package/dist/local-runtime.js +1120 -184
- package/dist/local-runtime.js.map +1 -1
- package/dist/package.json +10 -10
- package/package.json +10 -10
- package/dist/assets/dql-notebook/assets/UnifiedAgentRunPanel-C0oKTU6G.js +0 -88
package/dist/local-runtime.js
CHANGED
|
@@ -23,9 +23,10 @@ import { getRunner as getLLMRunner } from './llm/index.js';
|
|
|
23
23
|
import { rethrowIfCancelled } from './llm/cancellation.js';
|
|
24
24
|
import { fetchLatestPublishedDqlVersion, resolveDqlRuntimeVersionStatus } from './version-status.js';
|
|
25
25
|
import { resolveRetrievalHealthStatus } from './retrieval-health.js';
|
|
26
|
-
import {
|
|
26
|
+
import { applyFinding, createResearchState, narrationMaxTokensForFacts, nextHypothesis, rerankCandidates, synthesizeResearchNarrative, AgenticExecutionCapabilityGate, mintFinalSqlAuthorization, verifyAgenticSqlExecutionCapability, qualifyAuthorizationReferences, validateSqlAgainstLocalContext as validateAuthorizedSqlReferences, verifyFinalSql, } from '@duckcodeailabs/dql-agent';
|
|
27
|
+
import { applyEvalCassette, createDqlAgentProviderRunner, createGovernedTextProvider, resolveAgentFollowUpContext } from './llm/providers/dql-agent-provider.js';
|
|
27
28
|
import { listRemoteMcpSettings, saveRemoteMcpSettings } from './llm/mcp-config.js';
|
|
28
|
-
import { ClaudeProvider, ConversationStore, advanceThreadState, buildConversationSnapshot, conversationHistoryFromContext, recallRelevantTurns, renderConversationEnvelopeForPrompt, GeminiProvider, MemoryStore, OllamaProvider, OpenAIProvider, buildBlockBusinessFingerprint, buildBlockSqlFingerprints, buildAnalysisQuestionPlan, composeSemanticQueryForQuestion, aggregationIntegrityIssuesForSql, buildAggregationSafetyProof, buildLocalContextPack, applyContextPackCompatibility, toAgentRetrievalEvidence, prepareConversationPath, defaultMemoryPath, ensureDefaultMemoryFiles, ensureAgentProjectReady, isAgentProjectIndexReady, currentMetadataFingerprint, ensureMetadataCatalogFresh, readIndexedDomainKnowledge, readIndexedKnowledge360, compactSemanticRuntimeFailure, classifyAnalyticalFailure, normalizeWarehouseSqlFailure, parseProposal, propose, proposePlan, recordGovernedCorrection, HintStore, defaultHintIndexPath, ensureHintIndexFresh, listHintsFromGit, getHintEvaluationFromGit, getCorrectionTraceFromGit, inspectGovernedHint, editGovernedHintCandidate, reopenGovernedHint, retireHint, supersedeHint, hintsConflict, mineJoinPatterns, reviewGovernedHint, AgentRunEngine, SqliteAgentRunStore, defaultAgentRunGates, createLlmAgentRunPlanner, createHybridRouter, computeResultStats, buildDeterministicDashboardStory, synthesizeAnswer, streamOrGenerate, narrateResult, buildProposePreview, buildFromPrompt, internalRelationIdsInSql, defaultAgentRunStorePath, defaultAgentRunSqlitePath, resolveLocalOwner, resolveProposeConfig, recordQueryRun, recordRuntimeSchemaSnapshot, latestRuntimeSchemaSnapshotForProject, loadSkills, migrateLegacySkills, configuredSkillsPath, skillsDir, draftDomainSkillBootstrap, buildDomainSkillBootstrapPrompt, mergeDomainSkillBootstrapEnrichment, writeSkill, previewSkillChange, buildContextAuthoringProposal, contextAuthoringDependencyClosure, FileContextAuthoringProposalStore, deleteSkill, deriveGeneratedDraftSlug, deriveAnalyticalRepair, reindexProject, invalidateAgentProjectState, recordAgentRuntimeVersion, resolveDomainContextEnvelope, projectEmbeddingProvider, isHashedEmbeddingProvider, clearProjectEmbeddingCache, upgradeVectorIndexForProject, openMetadataCatalog, defaultKgPath, planAppFromPrompt, KGStore, planResearch, loadSemanticMetrics, cascadeTraceToEvidenceRouteSteps, createCascadeAnswerResult, createCascadeTrace, routeReasoningEffort, createAgentRunBudget, routeForCascadeAnswerTier, clampReasoningEffort, bumpReasoningEffort, resolveThinkingMode, coerceThinkingMode, upsertGeneratedDqlArtifactDraft, loadAgentSemanticLayer, isTrustedConversationTurn, resolveInternalRelationIds, analyticalError, tagAnalyticalError, withAnalyticalErrorOrigin, withAnalyticalErrorOriginSync, assertProviderPayloadAllowed, createProviderDispatchEgressReceipt, prepareProviderWireEnvelopeForDispatch, markProviderMetadataArray, createProviderEgressReceipt, redactProviderResultRows, composeVerifiedAnalyticalNarrative, DEFAULT_ASK_ROW_EGRESS_POLICY, ZERO_ROW_EGRESS_POLICY, resolveProviderResultRowEgressPolicy, } from '@duckcodeailabs/dql-agent';
|
|
29
|
+
import { composeBusinessExplanation, ClaudeProvider, ConversationStore, advanceThreadState, buildConversationSnapshot, conversationHistoryFromContext, recallRelevantTurns, renderConversationEnvelopeForPrompt, GeminiProvider, MemoryStore, OllamaProvider, OpenAIProvider, buildBlockBusinessFingerprint, buildBlockSqlFingerprints, buildAnalysisQuestionPlan, composeSemanticQueryForQuestion, aggregationIntegrityIssuesForSql, buildAggregationSafetyProof, buildLocalContextPack, applyContextPackCompatibility, toAgentRetrievalEvidence, prepareConversationPath, defaultMemoryPath, ensureDefaultMemoryFiles, ensureAgentProjectReady, isAgentProjectIndexReady, currentMetadataFingerprint, ensureMetadataCatalogFresh, readIndexedDomainKnowledge, readIndexedKnowledge360, compactSemanticRuntimeFailure, classifyAnalyticalFailure, normalizeWarehouseSqlFailure, parseProposal, propose, proposePlan, recordGovernedCorrection, HintStore, defaultHintIndexPath, ensureHintIndexFresh, listHintsFromGit, getHintEvaluationFromGit, getCorrectionTraceFromGit, inspectGovernedHint, editGovernedHintCandidate, reopenGovernedHint, retireHint, supersedeHint, hintsConflict, mineJoinPatterns, reviewGovernedHint, AgentRunEngine, SqliteAgentRunStore, defaultAgentRunGates, createLlmAgentRunPlanner, createHybridRouter, computeResultStats, buildDeterministicDashboardStory, synthesizeAnswer, streamOrGenerate, narrateResult, buildProposePreview, buildFromPrompt, internalRelationIdsInSql, defaultAgentRunStorePath, defaultAgentRunSqlitePath, resolveLocalOwner, resolveProposeConfig, recordQueryRun, recordRuntimeSchemaSnapshot, latestRuntimeSchemaSnapshotForProject, loadSkills, migrateLegacySkills, configuredSkillsPath, skillsDir, draftDomainSkillBootstrap, buildDomainSkillBootstrapPrompt, mergeDomainSkillBootstrapEnrichment, writeSkill, previewSkillChange, buildContextAuthoringProposal, contextAuthoringDependencyClosure, FileContextAuthoringProposalStore, deleteSkill, deriveGeneratedDraftSlug, deriveAnalyticalRepair, reindexProject, invalidateAgentProjectState, recordAgentRuntimeVersion, resolveDomainContextEnvelope, projectEmbeddingProvider, isHashedEmbeddingProvider, clearProjectEmbeddingCache, upgradeVectorIndexForProject, openMetadataCatalog, defaultKgPath, planAppFromPrompt, KGStore, planResearch, loadSemanticMetrics, cascadeTraceToEvidenceRouteSteps, createCascadeAnswerResult, createCascadeTrace, routeReasoningEffort, createAgentRunBudget, isProbeSafeColumn, deadlineScale, routeForCascadeAnswerTier, clampReasoningEffort, bumpReasoningEffort, resolveThinkingMode, coerceThinkingMode, upsertGeneratedDqlArtifactDraft, loadAgentSemanticLayer, isTrustedConversationTurn, resolveInternalRelationIds, analyticalError, tagAnalyticalError, withAnalyticalErrorOrigin, withAnalyticalErrorOriginSync, assertProviderPayloadAllowed, createProviderDispatchEgressReceipt, prepareProviderWireEnvelopeForDispatch, markProviderMetadataArray, createProviderEgressReceipt, redactProviderResultRows, composeVerifiedAnalyticalNarrative, buildCoverageGap, capResearchBranches, buildResearchEvidenceLedger, buildAnalyticalTurnPlan, resolveTopRankedRegionDependency, DEFAULT_ASK_ROW_EGRESS_POLICY, ZERO_ROW_EGRESS_POLICY, resolveProviderResultRowEgressPolicy, normalizeCanonicalQueryResult, normalizeAnalyticalExecutionFingerprint, normalizeAnalyticalExecutionReceipt, createAgentRunCancellationError, } from '@duckcodeailabs/dql-agent';
|
|
29
30
|
import { addSqlResultFilter, dashboardFilterableResultColumns, filterableResultColumns, replaceBlockStudioSql } from './sql-result-filter.js';
|
|
30
31
|
import { gatherProposeEnrichment } from './propose-enrich.js';
|
|
31
32
|
import { handleAppsApi, proposeAppAiBuild, recommendVisualization, } from './apps-api.js';
|
|
@@ -328,6 +329,10 @@ const CLIENT_PLAN_AUTHORITY_KEYS = new Set([
|
|
|
328
329
|
'priorResolvedAnalyticalPlan',
|
|
329
330
|
'resolvedAnalyticalPlan',
|
|
330
331
|
'analyticalFrame',
|
|
332
|
+
// Only the local compound executor may inject this after it has derived a
|
|
333
|
+
// canonical parent result binding. A browser-provided lookalike cannot become
|
|
334
|
+
// a child filter or skip ordinary member validation.
|
|
335
|
+
'analyticalTaskDependencyBinding',
|
|
331
336
|
]);
|
|
332
337
|
/**
|
|
333
338
|
* Browser/embedding context is useful retrieval and history input, but it is
|
|
@@ -378,6 +383,52 @@ export function agentRunDeadlineMs(request, env = process.env, activeProviderId)
|
|
|
378
383
|
? AGENT_RESEARCH_DEADLINE_MS
|
|
379
384
|
: AGENT_LOOKUP_DEADLINE_MS;
|
|
380
385
|
}
|
|
386
|
+
/**
|
|
387
|
+
* Run ready independent compound clauses concurrently, but wait for a typed
|
|
388
|
+
* parent result before executing a declared dependent clause. The scheduler
|
|
389
|
+
* itself has no authority to query or filter; callers supply both execution and
|
|
390
|
+
* a dependency resolver so immutable-plan and SQL guards remain unchanged.
|
|
391
|
+
*/
|
|
392
|
+
export async function scheduleCompoundAnalyticalTasks(input) {
|
|
393
|
+
const pending = [...input.tasks];
|
|
394
|
+
const settled = new Map();
|
|
395
|
+
while (pending.length > 0) {
|
|
396
|
+
const ready = pending.filter((task) => task.dependencies.every((dependencyId) => settled.has(dependencyId)));
|
|
397
|
+
if (ready.length === 0) {
|
|
398
|
+
for (const task of pending.splice(0)) {
|
|
399
|
+
settled.set(task.id, {
|
|
400
|
+
task,
|
|
401
|
+
error: 'The compound task dependency graph could not be resolved.',
|
|
402
|
+
dependencyError: {
|
|
403
|
+
ok: false,
|
|
404
|
+
code: 'RESULT_CONTRACT_MISMATCH',
|
|
405
|
+
message: 'The dependent task could not run because its parent dependency was unresolved.',
|
|
406
|
+
},
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
break;
|
|
410
|
+
}
|
|
411
|
+
for (const task of ready)
|
|
412
|
+
pending.splice(pending.indexOf(task), 1);
|
|
413
|
+
const batch = await Promise.all(ready.map(async (task) => {
|
|
414
|
+
if (!task.dependency || task.dependency.kind !== 'top_ranked_region')
|
|
415
|
+
return input.runTask(task);
|
|
416
|
+
const parent = settled.get(task.dependency.sourceTaskId);
|
|
417
|
+
const resolution = input.resolveDependency(task, parent);
|
|
418
|
+
if (!resolution.ok) {
|
|
419
|
+
const dependencyError = resolution;
|
|
420
|
+
return { task, error: dependencyError.message, dependencyError };
|
|
421
|
+
}
|
|
422
|
+
return input.runTask(task, resolution.binding);
|
|
423
|
+
}));
|
|
424
|
+
for (const result of batch)
|
|
425
|
+
settled.set(result.task.id, result);
|
|
426
|
+
}
|
|
427
|
+
return input.tasks.map((task) => settled.get(task.id) ?? {
|
|
428
|
+
task,
|
|
429
|
+
error: 'The compound task did not produce an outcome.',
|
|
430
|
+
});
|
|
431
|
+
}
|
|
381
432
|
/**
|
|
382
433
|
* Decide how a settled answer gets its business-facing prose.
|
|
383
434
|
*
|
|
@@ -420,11 +471,64 @@ export function shouldSynthesizeAgentRunAnswer(governedAnswer, requestedMode = '
|
|
|
420
471
|
rowEgress: DEFAULT_ASK_ROW_EGRESS_POLICY,
|
|
421
472
|
}).mode !== 'skip';
|
|
422
473
|
}
|
|
474
|
+
/**
|
|
475
|
+
* A receipt may be rendered in the local inspector and exported into evaluation
|
|
476
|
+
* output. Keep only stable validation codes there; provider error messages can
|
|
477
|
+
* contain a prompt excerpt, result value, or connector detail and must not
|
|
478
|
+
* become user-visible durable data.
|
|
479
|
+
*/
|
|
480
|
+
function narrationIntegrityFailureCodes(failures) {
|
|
481
|
+
return [...new Set(failures.map((failure) => {
|
|
482
|
+
const match = failure.trim().match(/^([A-Z][A-Z0-9_]{1,80})/);
|
|
483
|
+
return match?.[1] ?? 'NARRATION_VALIDATION_FAILED';
|
|
484
|
+
}).filter(Boolean))].slice(0, 8);
|
|
485
|
+
}
|
|
423
486
|
/**
|
|
424
487
|
* AGT-010 — the semantic route label is descriptive, while the exact
|
|
425
488
|
* route-specific aggregation proof is authoritative for governed trust.
|
|
426
489
|
* Missing proof remains blocked for legacy or malformed results.
|
|
427
490
|
*/
|
|
491
|
+
/**
|
|
492
|
+
* Trust for ONE answer, by the same rule the single-answer path uses: a route
|
|
493
|
+
* label is not authority, and a semantic route earns `governed` only when its
|
|
494
|
+
* aggregation proof actually passed.
|
|
495
|
+
*/
|
|
496
|
+
export function trustStateForAgentAnswer(answer) {
|
|
497
|
+
if (answer.certification === 'certified' || answer.kind === 'certified')
|
|
498
|
+
return 'certified';
|
|
499
|
+
return semanticAnswerHasPassedAggregationProof(answer) ? 'governed' : 'review_required';
|
|
500
|
+
}
|
|
501
|
+
const TRUST_RANK = {
|
|
502
|
+
certified: 3,
|
|
503
|
+
governed: 2,
|
|
504
|
+
grounded: 1,
|
|
505
|
+
review_required: 0,
|
|
506
|
+
};
|
|
507
|
+
/**
|
|
508
|
+
* A compound answer is exactly as trustworthy as its WEAKEST successful child.
|
|
509
|
+
*
|
|
510
|
+
* The previous rule was `every child completed ? 'governed' : 'review_required'`,
|
|
511
|
+
* which stamped `governed` on a parent whose children were review-required
|
|
512
|
+
* generated SQL — completion is not proof. That is a governance violation and
|
|
513
|
+
* the worst possible failure for this product: the reader is told a number
|
|
514
|
+
* carries governed authority when nothing proved it.
|
|
515
|
+
*
|
|
516
|
+
* `certified` is deliberately NOT reachable here. Certified trust is granted
|
|
517
|
+
* only by executing the exact certified artifact; a parent that merely
|
|
518
|
+
* assembled certified children did not execute one, so it caps at `governed`.
|
|
519
|
+
*/
|
|
520
|
+
export function compoundTrustState(childTrust) {
|
|
521
|
+
if (childTrust.length === 0)
|
|
522
|
+
return 'review_required';
|
|
523
|
+
const weakest = childTrust.reduce((low, current) => (TRUST_RANK[current] ?? 0) < (TRUST_RANK[low] ?? 0) ? current : low);
|
|
524
|
+
return weakest === 'certified' ? 'governed' : weakest;
|
|
525
|
+
}
|
|
526
|
+
/** Neutral parent outcome: governed only when every child completed governed. */
|
|
527
|
+
export function compoundStopReason(completedCount, childCount, trustState) {
|
|
528
|
+
return completedCount === childCount && childCount > 0 && trustState === 'governed'
|
|
529
|
+
? 'governed_compound_answer'
|
|
530
|
+
: 'human_review_required';
|
|
531
|
+
}
|
|
428
532
|
export function semanticAnswerHasPassedAggregationProof(governedAnswer) {
|
|
429
533
|
return governedAnswer.route?.tier === 'semantic_metric'
|
|
430
534
|
&& governedAnswer.aggregationSafetyProof?.status === 'safe';
|
|
@@ -564,7 +668,7 @@ export function validAskRepairDqlWrapper(source) {
|
|
|
564
668
|
}
|
|
565
669
|
/** Build retained automatic-repair authority from analytical failure state only. */
|
|
566
670
|
export function analyticalRepairCapabilityForAgentRun(run, resolvedTargetFingerprint) {
|
|
567
|
-
if (run.status !== 'blocked')
|
|
671
|
+
if (run.status !== 'blocked' || run.stopReason !== 'blocked')
|
|
568
672
|
return undefined;
|
|
569
673
|
const failedRun = analyticalFailedRunFromAgentRun(run);
|
|
570
674
|
const failure = failedRun?.failure;
|
|
@@ -963,6 +1067,33 @@ export function slimAgentRunForTransport(run) {
|
|
|
963
1067
|
: {}),
|
|
964
1068
|
};
|
|
965
1069
|
}
|
|
1070
|
+
/**
|
|
1071
|
+
* INDEX projection for `GET /api/agent-runs` — strictly lighter than the
|
|
1072
|
+
* presentation projection above.
|
|
1073
|
+
*
|
|
1074
|
+
* A run history list renders one ROW per run: question, route, status, trust,
|
|
1075
|
+
* timing, summary. It never renders an answer body, an event stream, or a step
|
|
1076
|
+
* trace. Shipping those anyway dominated the response: on 300 real stored runs
|
|
1077
|
+
* a 20-run page was 47.61 MB whole and still 6.80 MB under the presentation
|
|
1078
|
+
* projection, of which 6.35 MB was `artifacts[].payload` alone.
|
|
1079
|
+
*
|
|
1080
|
+
* Artifact identity (`id`/`kind`/`title`/`trustState`) is kept so a row can say
|
|
1081
|
+
* what it produced; only the payload body is dropped. The complete immutable
|
|
1082
|
+
* record stays available from `GET /api/agent-runs/:id`.
|
|
1083
|
+
*
|
|
1084
|
+
* Acceptance: PERF-003, E2E-022.
|
|
1085
|
+
*/
|
|
1086
|
+
export function agentRunListEntryForTransport(run) {
|
|
1087
|
+
const slim = slimAgentRunForTransport(run);
|
|
1088
|
+
const artifacts = (slim.artifacts ?? []).map((artifact) => {
|
|
1089
|
+
const record = agentRunRecord(artifact);
|
|
1090
|
+
if (!record)
|
|
1091
|
+
return artifact;
|
|
1092
|
+
const { payload: _payload, ...rest } = record;
|
|
1093
|
+
return rest;
|
|
1094
|
+
});
|
|
1095
|
+
return { ...slim, artifacts, steps: [], events: [] };
|
|
1096
|
+
}
|
|
966
1097
|
export function conversationTurnInputFromRun(run) {
|
|
967
1098
|
const artifact = run.artifacts.find((candidate) => candidate.kind === 'answer')
|
|
968
1099
|
?? run.artifacts.find((candidate) => candidate.kind === 'research_run')
|
|
@@ -970,7 +1101,9 @@ export function conversationTurnInputFromRun(run) {
|
|
|
970
1101
|
const payload = agentRunRecord(artifact?.payload);
|
|
971
1102
|
// A blocked run may retain diagnostic artifacts, but their result-shaped
|
|
972
1103
|
// payload is never accepted conversation evidence or prose authority.
|
|
973
|
-
const result = run.status === 'blocked'
|
|
1104
|
+
const result = run.status === 'blocked' || run.status === 'cancelled'
|
|
1105
|
+
? undefined
|
|
1106
|
+
: agentRunRecord(payload?.result);
|
|
974
1107
|
const columns = conversationResultColumns(result?.columns);
|
|
975
1108
|
// The visual preview stays tiny, but member resolution needs a wider bounded
|
|
976
1109
|
// value window. Deriving dimensions from only the eight preview rows caused a
|
|
@@ -1010,6 +1143,7 @@ export function conversationTurnInputFromRun(run) {
|
|
|
1010
1143
|
sql: agentRunString(payload?.proposedSql) ?? agentRunString(payload?.sql),
|
|
1011
1144
|
dqlArtifact: agentRunRecord(payload?.dqlArtifact),
|
|
1012
1145
|
cascade: agentRunRecord(payload?.cascade),
|
|
1146
|
+
narrationIntegrityReceipt: run.narrationIntegrityReceipt,
|
|
1013
1147
|
result: columns.length > 0 || rows.length > 0
|
|
1014
1148
|
? {
|
|
1015
1149
|
columns,
|
|
@@ -1137,12 +1271,7 @@ export class RunScopedProviderDispatchEvidence {
|
|
|
1137
1271
|
* from what it already has.
|
|
1138
1272
|
*/
|
|
1139
1273
|
expectedDispatchMs() {
|
|
1140
|
-
|
|
1141
|
-
return ASSUMED_PROVIDER_DISPATCH_MS;
|
|
1142
|
-
const sorted = [...this.observedDispatchDurations].sort((left, right) => left - right);
|
|
1143
|
-
// The slowest observed call is the honest predictor: an optimistic median
|
|
1144
|
-
// still admits a dispatch that the deadline then kills.
|
|
1145
|
-
return sorted[sorted.length - 1];
|
|
1274
|
+
return predictDispatchMs(this.observedDispatchDurations);
|
|
1146
1275
|
}
|
|
1147
1276
|
/** True when the remaining wall clock cannot fit another provider call. */
|
|
1148
1277
|
cannotFitAnotherDispatch() {
|
|
@@ -1335,6 +1464,50 @@ function mergeRunScopedProviderDispatchEvidence(run, evidence) {
|
|
|
1335
1464
|
diagnosticReceiptV2,
|
|
1336
1465
|
};
|
|
1337
1466
|
}
|
|
1467
|
+
/**
|
|
1468
|
+
* Final physical generated-SQL boundary.
|
|
1469
|
+
*
|
|
1470
|
+
* This receives the exact prepared statement immediately before the connector
|
|
1471
|
+
* callback. It intentionally validates before invoking `execute`: a bad
|
|
1472
|
+
* capability or unproven prepared reference must result in zero warehouse
|
|
1473
|
+
* calls, not a post-execution warning. It is module-exported only for the
|
|
1474
|
+
* local-runtime boundary harness; it is never an HTTP API or durable artifact.
|
|
1475
|
+
*
|
|
1476
|
+
* @internal
|
|
1477
|
+
*/
|
|
1478
|
+
export async function executePreparedAgenticSqlBoundary(input) {
|
|
1479
|
+
const capability = input.capability;
|
|
1480
|
+
if (capability) {
|
|
1481
|
+
const authorization = mintFinalSqlAuthorization({
|
|
1482
|
+
sql: input.preparedSql,
|
|
1483
|
+
proven: capability.provenIdentifiers.map((identifier) => ({
|
|
1484
|
+
identifier,
|
|
1485
|
+
evidence: capability.evidence[identifier] ?? 'catalog',
|
|
1486
|
+
})),
|
|
1487
|
+
runId: capability.runId,
|
|
1488
|
+
executionId: capability.executionId,
|
|
1489
|
+
snapshotId: capability.snapshotId,
|
|
1490
|
+
planId: capability.planId,
|
|
1491
|
+
targetFingerprint: capability.targetFingerprint,
|
|
1492
|
+
bindings: input.bindings,
|
|
1493
|
+
});
|
|
1494
|
+
const validation = validateAuthorizedSqlReferences(input.preparedSql, undefined);
|
|
1495
|
+
const verdict = verifyFinalSql(authorization, input.preparedSql, qualifyAuthorizationReferences(input.preparedSql, {
|
|
1496
|
+
relations: validation.referencedRelations ?? [],
|
|
1497
|
+
columns: validation.referencedColumns ?? [],
|
|
1498
|
+
}), {
|
|
1499
|
+
...input.scope,
|
|
1500
|
+
bindings: input.bindings,
|
|
1501
|
+
});
|
|
1502
|
+
if (process.env.DQL_ORCHESTRATOR_TRACE) {
|
|
1503
|
+
console.warn(`[dql] execution authorization: ${verdict.ok ? 'admitted' : 'REFUSED'} proven=${authorization.provenIdentifiers.length}${verdict.ok ? '' : ` reason=${verdict.reason}`}`);
|
|
1504
|
+
}
|
|
1505
|
+
if (!verdict.ok) {
|
|
1506
|
+
throw analyticalError(verdict.reason ?? 'The statement was not authorized for execution.', { origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql' });
|
|
1507
|
+
}
|
|
1508
|
+
}
|
|
1509
|
+
return input.execute();
|
|
1510
|
+
}
|
|
1338
1511
|
export async function startLocalServer(opts) {
|
|
1339
1512
|
const { rootDir, executor, connection: rawConnection, preferredPort, projectRoot = process.cwd() } = opts;
|
|
1340
1513
|
const bindHost = opts.host ?? process.env.DQL_HOST ?? '127.0.0.1';
|
|
@@ -2032,6 +2205,9 @@ export async function startLocalServer(opts) {
|
|
|
2032
2205
|
});
|
|
2033
2206
|
};
|
|
2034
2207
|
async function runGovernedAgentAnswerForRun(request, repair, route = 'generated_answer', onProgress, routeDecision) {
|
|
2208
|
+
return runGovernedAgentAnswerForRunInner(request, repair, route, onProgress, routeDecision);
|
|
2209
|
+
}
|
|
2210
|
+
async function runGovernedAgentAnswerForRunInner(request, repair, route = 'generated_answer', onProgress, routeDecision) {
|
|
2035
2211
|
const governed = resolveGovernedAnswerRunner(projectRoot);
|
|
2036
2212
|
let resolvedProvider = governed?.provider ?? null;
|
|
2037
2213
|
let runner = governed?.runner ?? null;
|
|
@@ -2134,6 +2310,9 @@ export async function startLocalServer(opts) {
|
|
|
2134
2310
|
snapshotId: runProjectSnapshot.snapshotId,
|
|
2135
2311
|
})
|
|
2136
2312
|
: undefined;
|
|
2313
|
+
// Local to this exact answer invocation. Compound children each enter this
|
|
2314
|
+
// function separately, so no child can consume another child's capability.
|
|
2315
|
+
const agenticExecutionCapabilityGate = new AgenticExecutionCapabilityGate();
|
|
2137
2316
|
await runner.run({
|
|
2138
2317
|
provider: resolvedProvider,
|
|
2139
2318
|
...(agentRunProviderEvidenceContext.getStore()
|
|
@@ -2158,9 +2337,13 @@ export async function startLocalServer(opts) {
|
|
|
2158
2337
|
},
|
|
2159
2338
|
reasoningEffort,
|
|
2160
2339
|
...(analysisDepth ? { analysisDepth } : {}),
|
|
2340
|
+
orchestrationMode: route === 'research' ? 'research' : 'ask',
|
|
2161
2341
|
allowProviderSemanticMemberSelection: route === 'research',
|
|
2162
2342
|
researchResultRowsOptIn: route === 'research' && request.researchResultRowsOptIn === true,
|
|
2163
2343
|
projectRoot,
|
|
2344
|
+
// Keys the execution authorization, so the proofs the analyst loop
|
|
2345
|
+
// gathers can be checked against the statement this run executes.
|
|
2346
|
+
...(request.runId ? { agentRunId: request.runId } : {}),
|
|
2164
2347
|
preparedContextPack: preparedAgentContextPacks.get(request),
|
|
2165
2348
|
domainContext,
|
|
2166
2349
|
projectSnapshot: { snapshotId: runProjectSnapshot.snapshotId, manifest: runProjectSnapshot.manifest },
|
|
@@ -2328,6 +2511,20 @@ export async function startLocalServer(opts) {
|
|
|
2328
2511
|
analyticalReferenceInstant: new Date().toISOString(),
|
|
2329
2512
|
executeCertifiedBlock: (node, invocation) => executeCertifiedBlockForAgent(node, invocation, semanticConnection, semanticConnectionName),
|
|
2330
2513
|
executeGeneratedSql: (sql, artifact) => executeGeneratedArtifactForAgent(request.question, sql, artifact, semanticConnection, semanticConnectionName),
|
|
2514
|
+
executeAgenticGeneratedSql: async (capability, sql, artifact) => {
|
|
2515
|
+
if (!agenticExecutionCapabilityGate.consume(capability)) {
|
|
2516
|
+
throw analyticalError('This analyst execution capability was already consumed; DQL did not retry it with stale proof.', {
|
|
2517
|
+
origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
|
|
2518
|
+
});
|
|
2519
|
+
}
|
|
2520
|
+
return executeGeneratedArtifactForAgent(request.question, sql, artifact, semanticConnection, semanticConnectionName, capability, {
|
|
2521
|
+
runId: request.runId,
|
|
2522
|
+
executionId: capability.executionId,
|
|
2523
|
+
snapshotId: runProjectSnapshot.snapshotId,
|
|
2524
|
+
planId: routeDecision?.resolvedAnalyticalPlan?.planId,
|
|
2525
|
+
targetFingerprint: generatedProposalTargetIdentity?.identityFingerprint,
|
|
2526
|
+
});
|
|
2527
|
+
},
|
|
2331
2528
|
executeDqlArtifact: (artifact) => executeArtifactReferenceForAgent(artifact, request.question, semanticConnection, semanticConnectionName),
|
|
2332
2529
|
getSchemaContext: (question, preparedContextPack) => getSchemaContextForAgent(question, preparedContextPack, semanticConnection, request.executionTarget?.target === 'connection'
|
|
2333
2530
|
? request.executionTarget.connectionName
|
|
@@ -2598,9 +2795,198 @@ export async function startLocalServer(opts) {
|
|
|
2598
2795
|
}
|
|
2599
2796
|
const answerRunExecutor = async ({ request, route, routeDecision, attempt, repairHint, emit }) => {
|
|
2600
2797
|
const runStartedAtMs = Date.now();
|
|
2798
|
+
const turnPlan = buildAnalyticalTurnPlan({
|
|
2799
|
+
question: request.question,
|
|
2800
|
+
mode: route === 'research' ? 'research' : 'ask',
|
|
2801
|
+
turnId: request.runId,
|
|
2802
|
+
candidateIds: routeDecision?.retrievalEvidence?.candidateIds ?? [],
|
|
2803
|
+
frozen: routeDecision?.resolvedAnalyticalPlan?.mode === 'authoritative',
|
|
2804
|
+
snapshotId: routeDecision?.resolvedAnalyticalPlan?.snapshotId ?? routeDecision?.retrievalEvidence?.snapshotId,
|
|
2805
|
+
sourceFingerprint: routeDecision?.resolvedAnalyticalPlan?.sourceFingerprint ?? routeDecision?.retrievalEvidence?.sourceFingerprint,
|
|
2806
|
+
});
|
|
2807
|
+
const childTurn = Boolean(request.workspaceContext && typeof request.workspaceContext === 'object'
|
|
2808
|
+
&& request.workspaceContext.analyticalTaskChild === true);
|
|
2809
|
+
// Compound questions are a bounded task graph, not one query whose answer
|
|
2810
|
+
// is copied into several labels. Independent children share the parent's
|
|
2811
|
+
// signal/deadline and return truthful partial success.
|
|
2812
|
+
if (turnPlan.tasks.length > 1 && !childTurn && (attempt ?? 0) === 0) {
|
|
2813
|
+
const runChildTask = async (task, dependencyBinding) => {
|
|
2814
|
+
if (request.signal?.aborted)
|
|
2815
|
+
rethrowIfCancelled(request.signal.reason, request.signal);
|
|
2816
|
+
try {
|
|
2817
|
+
const childRequest = {
|
|
2818
|
+
...request,
|
|
2819
|
+
question: task.question,
|
|
2820
|
+
...(dependencyBinding ? {
|
|
2821
|
+
conversationContext: {
|
|
2822
|
+
...(request.conversationContext ?? {}),
|
|
2823
|
+
// This is the complete parent-to-child data boundary: no
|
|
2824
|
+
// parent prose, SQL, or rows cross into a dependent clause.
|
|
2825
|
+
analyticalTaskDependencyBinding: dependencyBinding,
|
|
2826
|
+
},
|
|
2827
|
+
} : {}),
|
|
2828
|
+
workspaceContext: {
|
|
2829
|
+
...(request.workspaceContext && typeof request.workspaceContext === 'object' ? request.workspaceContext : {}),
|
|
2830
|
+
analyticalTaskChild: true,
|
|
2831
|
+
analyticalParentRunId: request.runId,
|
|
2832
|
+
analyticalTaskId: task.id,
|
|
2833
|
+
...(dependencyBinding ? { analyticalTaskDependencyBinding: dependencyBinding } : {}),
|
|
2834
|
+
},
|
|
2835
|
+
};
|
|
2836
|
+
const answer = await runGovernedAgentAnswerForRun(childRequest, { attempt: 0, repairHint }, route, (message) => emit({ type: 'executor.started', message: `Task ${task.id}: ${message}`, route }), undefined);
|
|
2837
|
+
if (answer.result) {
|
|
2838
|
+
const canonical = normalizeCanonicalQueryResult({
|
|
2839
|
+
...answer.result,
|
|
2840
|
+
resultFingerprint: answer.result.resultFingerprint ?? answer.result.executionReceipt?.resultFingerprint,
|
|
2841
|
+
executionReceipt: answer.result.executionReceipt,
|
|
2842
|
+
answerTier: answer.route?.tier ?? answer.sourceTier,
|
|
2843
|
+
});
|
|
2844
|
+
answer.result = {
|
|
2845
|
+
...answer.result,
|
|
2846
|
+
columns: canonical.columns,
|
|
2847
|
+
rows: canonical.rows,
|
|
2848
|
+
rowCount: canonical.rowCount,
|
|
2849
|
+
resultFingerprint: canonical.resultFingerprint,
|
|
2850
|
+
...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
|
|
2851
|
+
...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
|
|
2852
|
+
};
|
|
2853
|
+
}
|
|
2854
|
+
return { task, answer };
|
|
2855
|
+
}
|
|
2856
|
+
catch (error) {
|
|
2857
|
+
rethrowIfCancelled(error, request.signal);
|
|
2858
|
+
return { task, error: error instanceof Error ? error.message : String(error) };
|
|
2859
|
+
}
|
|
2860
|
+
};
|
|
2861
|
+
const scheduledChildren = await scheduleCompoundAnalyticalTasks({
|
|
2862
|
+
tasks: turnPlan.tasks.slice(0, 6),
|
|
2863
|
+
runTask: async (task, binding) => {
|
|
2864
|
+
const child = await runChildTask(task, binding);
|
|
2865
|
+
return { task, value: child.answer, error: child.error };
|
|
2866
|
+
},
|
|
2867
|
+
resolveDependency: (task, parent) => {
|
|
2868
|
+
const sourceTaskId = task.dependency?.sourceTaskId ?? '';
|
|
2869
|
+
return resolveTopRankedRegionDependency(sourceTaskId, parent?.value?.result
|
|
2870
|
+
? normalizeCanonicalQueryResult({
|
|
2871
|
+
...parent.value.result,
|
|
2872
|
+
resultFingerprint: parent.value.result.resultFingerprint ?? parent.value.result.executionReceipt?.resultFingerprint,
|
|
2873
|
+
executionReceipt: parent.value.result.executionReceipt,
|
|
2874
|
+
answerTier: parent.value.route?.tier ?? parent.value.sourceTier,
|
|
2875
|
+
})
|
|
2876
|
+
: undefined, parent?.task);
|
|
2877
|
+
},
|
|
2878
|
+
});
|
|
2879
|
+
const childResults = scheduledChildren.map(({ task, value, error, dependencyError }) => ({
|
|
2880
|
+
task,
|
|
2881
|
+
answer: value,
|
|
2882
|
+
error,
|
|
2883
|
+
...(dependencyError ? {
|
|
2884
|
+
dependencyGap: buildCoverageGap({
|
|
2885
|
+
code: dependencyError.code,
|
|
2886
|
+
phase: 'planning',
|
|
2887
|
+
message: dependencyError.message,
|
|
2888
|
+
searchedSources: routeDecision?.retrievalEvidence?.candidateIds ?? [],
|
|
2889
|
+
attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
|
|
2890
|
+
missing: ['unambiguous_top_region'],
|
|
2891
|
+
recoverable: true,
|
|
2892
|
+
planFrozen: turnPlan.frozen,
|
|
2893
|
+
nextActions: ['Ask for a single top region or review the parent result before retrying the customer task.'],
|
|
2894
|
+
}),
|
|
2895
|
+
} : {}),
|
|
2896
|
+
}));
|
|
2897
|
+
const outcomes = childResults.map(({ task, answer, error }) => ({
|
|
2898
|
+
version: 1,
|
|
2899
|
+
taskId: task.id,
|
|
2900
|
+
status: error || answer?.kind === 'no_answer' ? 'gap' : 'completed',
|
|
2901
|
+
...(answer?.answer || answer?.text ? { summary: answer.answer ?? answer.text } : {}),
|
|
2902
|
+
...(answer?.result?.resultFingerprint ? { resultFingerprint: answer.result.resultFingerprint } : {}),
|
|
2903
|
+
...(error || answer?.kind === 'no_answer' ? {
|
|
2904
|
+
gap: childResults.find((candidate) => candidate.task.id === task.id)?.dependencyGap ?? buildCoverageGap({
|
|
2905
|
+
code: answer?.refusalCode === 'ambiguous' ? 'AMBIGUOUS_MEANING' : 'EXECUTION_FAILED',
|
|
2906
|
+
phase: answer?.executionError ? 'execution' : 'meaning',
|
|
2907
|
+
message: error ?? answer?.answer ?? answer?.text ?? 'The task did not produce an accepted analytical result.',
|
|
2908
|
+
searchedSources: routeDecision?.retrievalEvidence?.candidateIds ?? [],
|
|
2909
|
+
attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
|
|
2910
|
+
missing: [],
|
|
2911
|
+
recoverable: false,
|
|
2912
|
+
planFrozen: turnPlan.frozen,
|
|
2913
|
+
nextActions: ['Review the task context and retry the clause.'],
|
|
2914
|
+
}),
|
|
2915
|
+
} : {}),
|
|
2916
|
+
}));
|
|
2917
|
+
const completedCount = outcomes.filter((outcome) => outcome.status === 'completed').length;
|
|
2918
|
+
const tasks = turnPlan.tasks.map((task) => ({
|
|
2919
|
+
...task,
|
|
2920
|
+
status: outcomes.find((outcome) => outcome.taskId === task.id)?.status === 'completed' ? 'completed' : 'gap',
|
|
2921
|
+
}));
|
|
2922
|
+
const answerText = childResults.map(({ task, answer, error }) => `${task.question}: ${error ?? answer?.answer ?? answer?.text ?? 'No accepted result was produced.'}`).join('\n\n');
|
|
2923
|
+
const compoundTrust = compoundTrustState(childResults
|
|
2924
|
+
.filter(({ answer, error }) => !error && answer && answer.kind !== 'no_answer')
|
|
2925
|
+
.map(({ answer }) => trustStateForAgentAnswer(answer)));
|
|
2926
|
+
return {
|
|
2927
|
+
summary: completedCount === outcomes.length
|
|
2928
|
+
? `Answered ${completedCount} analytical clauses.`
|
|
2929
|
+
: `Answered ${completedCount} of ${outcomes.length} analytical clauses; the remaining clauses need review.`,
|
|
2930
|
+
answer: answerText,
|
|
2931
|
+
status: completedCount === outcomes.length ? 'completed' : completedCount > 0 ? 'needs_review' : 'needs_clarification',
|
|
2932
|
+
// The parent is only as trustworthy as its weakest SUCCESSFUL child.
|
|
2933
|
+
// Completion is not proof: the previous rule stamped `governed` on a
|
|
2934
|
+
// parent assembled from review-required generated SQL.
|
|
2935
|
+
trustState: compoundTrust,
|
|
2936
|
+
// The stop reason has to agree with the trust it reports. Claiming a
|
|
2937
|
+
// governed semantic answer over generated/certified children
|
|
2938
|
+
// misrepresents provenance as much as the trust label does.
|
|
2939
|
+
stopReason: compoundStopReason(completedCount, outcomes.length, compoundTrust),
|
|
2940
|
+
artifacts: childResults.map(({ task, answer, error }) => agentRunArtifact('answer', `Task: ${task.question}`, {
|
|
2941
|
+
taskId: task.id,
|
|
2942
|
+
question: task.question,
|
|
2943
|
+
answer: answer?.answer ?? answer?.text,
|
|
2944
|
+
resultFingerprint: answer?.result?.resultFingerprint,
|
|
2945
|
+
error,
|
|
2946
|
+
})),
|
|
2947
|
+
evaluations: outcomes.map((outcome) => agentRunEvaluation(`analytical-task-${outcome.taskId}`, `Analytical task ${outcome.taskId}`, outcome.status === 'completed', outcome.status === 'completed' ? 'info' : 'warning', outcome.summary ?? outcome.gap?.message ?? 'Task outcome recorded.', outcome)),
|
|
2948
|
+
analyticalTurnPlan: { ...turnPlan, tasks, frozen: true },
|
|
2949
|
+
analyticalTaskOutcomes: outcomes,
|
|
2950
|
+
};
|
|
2951
|
+
}
|
|
2601
2952
|
let governedAnswer;
|
|
2602
2953
|
try {
|
|
2603
2954
|
governedAnswer = await runGovernedAgentAnswerForRun(request, { attempt, repairHint }, route, (message) => emit({ type: 'executor.started', message, route }), routeDecision);
|
|
2955
|
+
// Keep the canonical result contract on the answer itself, not only on
|
|
2956
|
+
// the narration preview. Conversation persistence, follow-up member
|
|
2957
|
+
// resolution, Apply, and the notebook table all read this payload; if a
|
|
2958
|
+
// connector returned positional rows, dropping normalization here makes
|
|
2959
|
+
// the next question lose its customer/product/region binding (AGT-032).
|
|
2960
|
+
if (governedAnswer.result) {
|
|
2961
|
+
const canonical = normalizeCanonicalQueryResult({
|
|
2962
|
+
...governedAnswer.result,
|
|
2963
|
+
resultFingerprint: governedAnswer.result.resultFingerprint
|
|
2964
|
+
?? governedAnswer.result.executionReceipt?.resultFingerprint,
|
|
2965
|
+
executionReceipt: governedAnswer.result.executionReceipt,
|
|
2966
|
+
answerTier: governedAnswer.route?.tier ?? governedAnswer.sourceTier,
|
|
2967
|
+
trustState: governedAnswer.certification === 'certified'
|
|
2968
|
+
? 'certified'
|
|
2969
|
+
: governedAnswer.kind === 'no_answer'
|
|
2970
|
+
? 'blocked'
|
|
2971
|
+
: governedAnswer.reviewStatus === 'analyst_review_required'
|
|
2972
|
+
? 'review_required'
|
|
2973
|
+
: 'governed',
|
|
2974
|
+
});
|
|
2975
|
+
governedAnswer.result = {
|
|
2976
|
+
...governedAnswer.result,
|
|
2977
|
+
columns: canonical.columns,
|
|
2978
|
+
rows: canonical.rows,
|
|
2979
|
+
rowCount: canonical.rowCount,
|
|
2980
|
+
// Preserve the execution-service/receipt fingerprint. A local UI
|
|
2981
|
+
// digest is only a legacy fallback in normalizeCanonicalQueryResult;
|
|
2982
|
+
// it must never replace the cryptographic identity of this run.
|
|
2983
|
+
resultFingerprint: canonical.resultFingerprint,
|
|
2984
|
+
...(canonical.executionTime !== undefined ? { executionTime: canonical.executionTime } : {}),
|
|
2985
|
+
...(canonical.truncated ? { truncated: true } : {}),
|
|
2986
|
+
...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
|
|
2987
|
+
...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
|
|
2988
|
+
};
|
|
2989
|
+
}
|
|
2604
2990
|
// Surface the approved Hint-Graph corrections that shaped this answer so the
|
|
2605
2991
|
// UI can show an "applied learnings" chip (memoryContext is already on the answer).
|
|
2606
2992
|
if (!governedAnswer.appliedHints) {
|
|
@@ -2818,6 +3204,38 @@ export async function startLocalServer(opts) {
|
|
|
2818
3204
|
// has already spent its one evidence-aware repair. Keep this terminal and
|
|
2819
3205
|
// inspectable; an ordinary Ask must never silently become a second Research run.
|
|
2820
3206
|
const isModelDeclined = governedAnswer.kind === 'no_answer' && governedAnswer.refusalCode === 'model_declined';
|
|
3207
|
+
// Before an analytical plan is frozen, a generated Ask gap is a recoverable
|
|
3208
|
+
// coverage result rather than a terminal governed error. Let the run engine
|
|
3209
|
+
// consume the typed evaluation and make one bounded cascade decision. Frozen
|
|
3210
|
+
// certified/semantic plans and provider/policy/execution failures remain
|
|
3211
|
+
// fail-closed.
|
|
3212
|
+
const canRecoverPreFreezeGap = (isGroundingGap || isModelDeclined)
|
|
3213
|
+
&& route === 'generated_answer'
|
|
3214
|
+
&& (request.requestedMode === undefined || request.requestedMode === 'auto' || request.requestedMode === 'ask')
|
|
3215
|
+
&& routeDecision?.resolvedAnalyticalPlan?.mode !== 'authoritative';
|
|
3216
|
+
const typedCoverageGap = (isGroundingGap || isModelDeclined)
|
|
3217
|
+
? buildCoverageGap({
|
|
3218
|
+
code: governedAnswer.refusalCode === 'modeling_gap' ? 'MISSING_RELATIONSHIP' : 'MISSING_RUNTIME_CAPABILITY',
|
|
3219
|
+
phase: 'planning',
|
|
3220
|
+
message: governedAnswer.refusalDetails?.message
|
|
3221
|
+
?? (isModelDeclined
|
|
3222
|
+
? 'The available context did not produce a safe analytical tuple.'
|
|
3223
|
+
: 'The retrieved context did not prove the metadata required for this analytical question.'),
|
|
3224
|
+
searchedSources: ['certified_blocks', 'semantic_metrics', 'dbt_manifest', 'relationship_graph', 'warehouse_metadata'],
|
|
3225
|
+
attemptedRoutes: ['certified', 'semantic', 'governed_relational', 'generated'],
|
|
3226
|
+
missing: governedAnswer.refusalDetails?.offending
|
|
3227
|
+
? [
|
|
3228
|
+
governedAnswer.refusalDetails.offending.relation,
|
|
3229
|
+
governedAnswer.refusalDetails.offending.column,
|
|
3230
|
+
].filter((value) => Boolean(value))
|
|
3231
|
+
: ['an executable metric/dimension/relationship tuple'],
|
|
3232
|
+
recoverable: canRecoverPreFreezeGap,
|
|
3233
|
+
planFrozen: routeDecision?.resolvedAnalyticalPlan?.mode === 'authoritative',
|
|
3234
|
+
nextActions: canRecoverPreFreezeGap
|
|
3235
|
+
? ['continue through DBT-grounded relational context', 'run review-required generated SQL', 'start bounded Research if coverage remains incomplete']
|
|
3236
|
+
: ['review the retained metadata gap', 'select a governed metric or dimension', 'repair modeling if the relationship is missing'],
|
|
3237
|
+
})
|
|
3238
|
+
: undefined;
|
|
2821
3239
|
// Only a genuinely AMBIGUOUS question is surfaced as "needs clarification".
|
|
2822
3240
|
// Grounding/compiler gaps are terminal review states with their evidence trace;
|
|
2823
3241
|
// provider outages are blocked so the UI can offer an explicit retry.
|
|
@@ -2867,7 +3285,34 @@ export async function startLocalServer(opts) {
|
|
|
2867
3285
|
? { ...preview, rows: redactProviderResultRows(preview.rows, narrationMaxRows) }
|
|
2868
3286
|
: undefined;
|
|
2869
3287
|
let narrationSource;
|
|
2870
|
-
|
|
3288
|
+
// The receipt is the durable source of truth for evaluation and inspector
|
|
3289
|
+
// display. Do not reconstruct this later from rows or the reader-facing
|
|
3290
|
+
// fallback sentence: both are presentation artifacts, not evidence that a
|
|
3291
|
+
// fact-grounded narrator actually ran.
|
|
3292
|
+
const verifiedFactNarration = narrationPlan.mode === 'verified_facts'
|
|
3293
|
+
&& Boolean(governedAnswer.analyticalFacts && governedAnswer.resolvedAnalyticalPlan?.analyticalFrame);
|
|
3294
|
+
let narrationIntegrityReceipt = narrationPlan.mode === 'skip'
|
|
3295
|
+
? {
|
|
3296
|
+
version: 1,
|
|
3297
|
+
mode: 'skip',
|
|
3298
|
+
outcome: 'skipped',
|
|
3299
|
+
attempted: false,
|
|
3300
|
+
factCount: 0,
|
|
3301
|
+
maxRows: 0,
|
|
3302
|
+
validationFailures: [],
|
|
3303
|
+
skipReason: narrationPlan.reason,
|
|
3304
|
+
}
|
|
3305
|
+
: {
|
|
3306
|
+
version: 1,
|
|
3307
|
+
mode: verifiedFactNarration ? 'verified_facts' : 'preview_grounded',
|
|
3308
|
+
// This is deliberately pessimistic until a narration outcome is
|
|
3309
|
+
// observed, so an exception cannot be persisted as a silent skip.
|
|
3310
|
+
outcome: 'error',
|
|
3311
|
+
attempted: true,
|
|
3312
|
+
factCount: verifiedFactNarration ? governedAnswer.analyticalFacts?.facts.length ?? 0 : 0,
|
|
3313
|
+
maxRows: narrationPlan.maxRows,
|
|
3314
|
+
validationFailures: [],
|
|
3315
|
+
};
|
|
2871
3316
|
if (narrationPlan.mode !== 'skip' && narrationProvider) {
|
|
2872
3317
|
const narrationStartedAtMs = Date.now();
|
|
2873
3318
|
const draft = governedAnswer.answer ?? governedAnswer.text;
|
|
@@ -2880,8 +3325,16 @@ export async function startLocalServer(opts) {
|
|
|
2880
3325
|
columnCount: providerPreview?.columns.length ?? 0,
|
|
2881
3326
|
},
|
|
2882
3327
|
});
|
|
2883
|
-
const narrationDispatchOptions = () => ({
|
|
2884
|
-
|
|
3328
|
+
const narrationDispatchOptions = (factCount = 0) => ({
|
|
3329
|
+
// The ceiling has to grow with the result. Every claim must echo the
|
|
3330
|
+
// fact ids it rests on, and a fact id is a long hex string that
|
|
3331
|
+
// tokenizes badly — ten of them consume most of the budget before a
|
|
3332
|
+
// word of prose is written. At a flat 350 a ten-row answer was
|
|
3333
|
+
// truncated mid-sentence ("...Elizabeth Shea (875), and Dyl"), so the
|
|
3334
|
+
// JSON never closed, BOTH attempts failed as UNPARSEABLE_CLAIMS, and
|
|
3335
|
+
// the reader got "Verified narration was unavailable" above a robot
|
|
3336
|
+
// dump of the very rows the model had just described correctly.
|
|
3337
|
+
maxTokens: narrationMaxTokensForFacts(factCount),
|
|
2885
3338
|
temperature: 0.3,
|
|
2886
3339
|
maxProviderDispatches: 2,
|
|
2887
3340
|
...(agentRunProviderEvidenceContext.getStore()
|
|
@@ -2898,10 +3351,19 @@ export async function startLocalServer(opts) {
|
|
|
2898
3351
|
factSet: governedAnswer.analyticalFacts,
|
|
2899
3352
|
question: request.question,
|
|
2900
3353
|
maxRows: narrationPlan.maxRows,
|
|
2901
|
-
complete: async ({ system, user }) => streamOrGenerate(narrationProvider, [{ role: 'system', content: system }, { role: 'user', content: user }], narrationDispatchOptions(), () => { }),
|
|
3354
|
+
complete: async ({ system, user }) => streamOrGenerate(narrationProvider, [{ role: 'system', content: system }, { role: 'user', content: user }], narrationDispatchOptions(governedAnswer.analyticalFacts?.facts.length ?? 0), () => { }),
|
|
2902
3355
|
});
|
|
2903
3356
|
narrationSource = composed.source;
|
|
2904
|
-
|
|
3357
|
+
narrationIntegrityReceipt = {
|
|
3358
|
+
...narrationIntegrityReceipt,
|
|
3359
|
+
outcome: composed.source === 'llm' ? 'success' : 'deterministic_fallback',
|
|
3360
|
+
validationFailures: narrationIntegrityFailureCodes(composed.validationFailures),
|
|
3361
|
+
};
|
|
3362
|
+
if (process.env.DQL_ORCHESTRATOR_TRACE) {
|
|
3363
|
+
console.warn(`[dql] narration: source=${composed.source}${narrationIntegrityReceipt.validationFailures.length > 0
|
|
3364
|
+
? ` rejected=${narrationIntegrityReceipt.validationFailures.join(',')}`
|
|
3365
|
+
: ''}`);
|
|
3366
|
+
}
|
|
2905
3367
|
synthesizedAnswer = composed.source === 'llm'
|
|
2906
3368
|
? composed.narrative.text
|
|
2907
3369
|
// A verification failure is not silent: the deterministic join is a
|
|
@@ -2931,11 +3393,22 @@ export async function startLocalServer(opts) {
|
|
|
2931
3393
|
narrationSource = result.source;
|
|
2932
3394
|
if (result.text)
|
|
2933
3395
|
synthesizedAnswer = result.text;
|
|
3396
|
+
narrationIntegrityReceipt = {
|
|
3397
|
+
...narrationIntegrityReceipt,
|
|
3398
|
+
outcome: result.source === 'llm' ? 'success' : 'deterministic_fallback',
|
|
3399
|
+
validationFailures: [],
|
|
3400
|
+
};
|
|
2934
3401
|
}
|
|
2935
3402
|
}
|
|
2936
3403
|
catch {
|
|
2937
3404
|
// Keep the governed draft on any narration failure.
|
|
2938
3405
|
synthesizedAnswer = undefined;
|
|
3406
|
+
narrationIntegrityReceipt = {
|
|
3407
|
+
...narrationIntegrityReceipt,
|
|
3408
|
+
outcome: 'error',
|
|
3409
|
+
validationFailures: [],
|
|
3410
|
+
errorCode: 'narration_error',
|
|
3411
|
+
};
|
|
2939
3412
|
}
|
|
2940
3413
|
finally {
|
|
2941
3414
|
narrationDurationMs = Date.now() - narrationStartedAtMs;
|
|
@@ -2971,7 +3444,7 @@ export async function startLocalServer(opts) {
|
|
|
2971
3444
|
? 'blocked'
|
|
2972
3445
|
: needsClarification
|
|
2973
3446
|
? 'needs_clarification'
|
|
2974
|
-
: isGroundingGap || isModelDeclined
|
|
3447
|
+
: (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
|
|
2975
3448
|
? 'blocked'
|
|
2976
3449
|
: isCertified || isSemantic
|
|
2977
3450
|
? 'completed'
|
|
@@ -2980,7 +3453,7 @@ export async function startLocalServer(opts) {
|
|
|
2980
3453
|
? 'blocked'
|
|
2981
3454
|
: needsClarification
|
|
2982
3455
|
? 'not_applicable'
|
|
2983
|
-
: isGroundingGap || isModelDeclined
|
|
3456
|
+
: (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
|
|
2984
3457
|
? 'blocked'
|
|
2985
3458
|
: isCertified
|
|
2986
3459
|
? 'certified'
|
|
@@ -2994,7 +3467,7 @@ export async function startLocalServer(opts) {
|
|
|
2994
3467
|
// A gap is terminal, but it is NOT a provider outage: it still owes the
|
|
2995
3468
|
// user its evidence trace, and its next-actions below stay the
|
|
2996
3469
|
// grounding-gap set rather than "retry the provider".
|
|
2997
|
-
: isGroundingGap || isModelDeclined
|
|
3470
|
+
: (isGroundingGap || isModelDeclined) && !canRecoverPreFreezeGap
|
|
2998
3471
|
? 'human_review_required'
|
|
2999
3472
|
: isCertified
|
|
3000
3473
|
? 'certified_answer_found'
|
|
@@ -3075,7 +3548,11 @@ export async function startLocalServer(opts) {
|
|
|
3075
3548
|
trustState,
|
|
3076
3549
|
stopReason,
|
|
3077
3550
|
artifacts: isTerminalFailure
|
|
3078
|
-
? [agentRunArtifact('answer',
|
|
3551
|
+
? [agentRunArtifact('answer',
|
|
3552
|
+
// Was 'Failed governed analytical run' — internal orchestration state
|
|
3553
|
+
// used as the card heading a user reads. It names our pipeline, not
|
|
3554
|
+
// what happened to their question.
|
|
3555
|
+
terminalFailureTitle(governedAnswer), governedAnswer, governedAnswer.sourceCertifiedBlock ?? governedAnswer.block?.name, 'blocked')]
|
|
3079
3556
|
: governedAnswer.kind === 'no_answer'
|
|
3080
3557
|
// A refusal still keeps the DQL draft the answer loop produced (when any),
|
|
3081
3558
|
// so the "Review DQL draft" next-action isn't a dead link and the user can
|
|
@@ -3083,7 +3560,7 @@ export async function startLocalServer(opts) {
|
|
|
3083
3560
|
? (governedAnswer.semanticExecutionTrace
|
|
3084
3561
|
? [agentRunArtifact('answer', governedAnswer.refusalCode === 'ambiguous'
|
|
3085
3562
|
? 'Semantic path selection required'
|
|
3086
|
-
: 'Semantic compilation details', governedAnswer, undefined, needsClarification ? 'not_applicable' : 'review_required')]
|
|
3563
|
+
: 'Semantic compilation details', typedCoverageGap ? { ...governedAnswer, coverageGap: typedCoverageGap } : governedAnswer, undefined, needsClarification ? 'not_applicable' : 'review_required')]
|
|
3087
3564
|
: governedAnswer.dqlArtifact && !isProviderError && !isGroundingGap && !isModelDeclined && !isPolicyBlocked
|
|
3088
3565
|
? [agentRunArtifact('dql_block_draft', 'DQL draft (review required)', governedAnswer.dqlArtifact, undefined, 'review_required')]
|
|
3089
3566
|
// Every refusal still owes the user an account of itself. Emitting no
|
|
@@ -3092,7 +3569,7 @@ export async function startLocalServer(opts) {
|
|
|
3092
3569
|
// answered", the DQL draft and the compiled SQL all become
|
|
3093
3570
|
// unreachable exactly when the user most needs to see why it stopped
|
|
3094
3571
|
// and carry the query into a notebook.
|
|
3095
|
-
: [agentRunArtifact('answer', 'No answer was accepted', governedAnswer, undefined, needsClarification ? 'not_applicable' : 'blocked')])
|
|
3572
|
+
: [agentRunArtifact('answer', 'No answer was accepted', typedCoverageGap ? { ...governedAnswer, coverageGap: typedCoverageGap } : governedAnswer, undefined, needsClarification ? 'not_applicable' : 'blocked')])
|
|
3096
3573
|
: [agentRunArtifact('answer', isCertified ? 'Certified answer' : isSemantic ? 'Governed semantic answer' : isExploratory ? 'Exploratory DBT-grounded answer' : 'Review-required answer', governedAnswer, governedAnswer.sourceCertifiedBlock ?? governedAnswer.block?.name, isCertified ? 'certified' : isSemantic ? 'governed' : 'review_required')],
|
|
3097
3574
|
evaluations: [
|
|
3098
3575
|
agentRunEvaluation('route-decision', 'Route decision', true, 'info', routeDecision?.reason ?? 'Routed request to governed answer.', {
|
|
@@ -3127,22 +3604,34 @@ export async function startLocalServer(opts) {
|
|
|
3127
3604
|
...(isExecutionFailure ? [agentRunEvaluation('query-execution', 'Query execution', false, 'blocking', `The governed query failed before it produced a result: ${governedAnswer.executionError}`)] : []),
|
|
3128
3605
|
...(isGroundingGap ? [
|
|
3129
3606
|
{
|
|
3130
|
-
...agentRunEvaluation('grounding-gap', 'Metadata grounding', false, 'warning',
|
|
3607
|
+
...agentRunEvaluation('grounding-gap', 'Metadata grounding', false, 'warning', canRecoverPreFreezeGap
|
|
3608
|
+
? 'The first governed lookup did not prove the required metadata grounding. Continue through the bounded relational/generated cascade before asking the user to repair modeling.'
|
|
3609
|
+
: 'The bounded lookup could not prove the required metadata grounding. This plan is already frozen, so no automatic Research escalation was started.', {
|
|
3131
3610
|
refusalCode: governedAnswer.refusalCode,
|
|
3132
3611
|
refusalDetails: governedAnswer.refusalDetails,
|
|
3133
3612
|
validationWarnings: governedAnswer.validationWarnings,
|
|
3134
3613
|
route: governedAnswer.route,
|
|
3614
|
+
coverageGap: typedCoverageGap,
|
|
3135
3615
|
}),
|
|
3616
|
+
...(canRecoverPreFreezeGap ? {
|
|
3617
|
+
suggestedRepair: 'Continue the unfrozen Ask cascade through DBT-grounded relational context and review-required generated SQL.',
|
|
3618
|
+
} : {}),
|
|
3136
3619
|
},
|
|
3137
3620
|
] : []),
|
|
3138
3621
|
...(isModelDeclined ? [
|
|
3139
3622
|
{
|
|
3140
|
-
...agentRunEvaluation('declined-despite-context', 'Answer grounding', false, 'blocking',
|
|
3623
|
+
...agentRunEvaluation('declined-despite-context', 'Answer grounding', false, 'blocking', canRecoverPreFreezeGap
|
|
3624
|
+
? 'The governed lookup could not compose a query after its bounded in-lane repair. Continue with the Research context ledger before returning a typed gap.'
|
|
3625
|
+
: 'The bounded lookup could not compose a governed query after its in-lane repair. Start Research explicitly to investigate beyond this lookup budget.', {
|
|
3141
3626
|
refusalCode: governedAnswer.refusalCode,
|
|
3142
3627
|
refusalDetails: governedAnswer.refusalDetails,
|
|
3143
3628
|
validationWarnings: governedAnswer.validationWarnings,
|
|
3144
3629
|
route: governedAnswer.route,
|
|
3630
|
+
coverageGap: typedCoverageGap,
|
|
3145
3631
|
}),
|
|
3632
|
+
...(canRecoverPreFreezeGap ? {
|
|
3633
|
+
suggestedRepair: 'Search the surrounding metadata and relationship context before giving up on the analytical turn.',
|
|
3634
|
+
} : {}),
|
|
3146
3635
|
},
|
|
3147
3636
|
] : []),
|
|
3148
3637
|
...(isPolicyBlocked ? [
|
|
@@ -3160,12 +3649,38 @@ export async function startLocalServer(opts) {
|
|
|
3160
3649
|
...(governedAnswer.executionError ? [
|
|
3161
3650
|
agentRunEvaluation('execution-error', 'Execution error', false, 'warning', governedAnswer.executionError),
|
|
3162
3651
|
] : []),
|
|
3652
|
+
// Keep only the content-free receipt codes in the durable inspection
|
|
3653
|
+
// record. Raw verifier prose can contain a result value, prompt excerpt,
|
|
3654
|
+
// or provider error and is not safe evidence to surface or persist.
|
|
3655
|
+
...(narrationIntegrityReceipt.outcome === 'deterministic_fallback'
|
|
3656
|
+
&& narrationIntegrityReceipt.validationFailures.length > 0 ? [
|
|
3657
|
+
agentRunEvaluation('narration-verification', 'Narration verification', false, 'warning', `The drafted narration was rejected against the result fact set, so the deterministic record was shown instead: ${narrationIntegrityReceipt.validationFailures.join(', ')}.`, { narrationSource, validationFailures: narrationIntegrityReceipt.validationFailures }),
|
|
3658
|
+
] : []),
|
|
3163
3659
|
],
|
|
3164
3660
|
nextActions,
|
|
3165
3661
|
providerEgressReceipts: finalProviderEgressReceipts,
|
|
3166
3662
|
telemetry: finalTelemetry,
|
|
3663
|
+
narrationIntegrityReceipt,
|
|
3167
3664
|
};
|
|
3168
3665
|
};
|
|
3666
|
+
/**
|
|
3667
|
+
* A heading for a run that ended without an answer, in the user's terms.
|
|
3668
|
+
*
|
|
3669
|
+
* Says WHICH stage stopped, because "it failed" and "it was stopped before
|
|
3670
|
+
* running" call for different next moves: one is worth retrying, the other
|
|
3671
|
+
* needs the question or the model changed.
|
|
3672
|
+
*/
|
|
3673
|
+
const terminalFailureTitle = (answer) => {
|
|
3674
|
+
switch (answer.refusalCode) {
|
|
3675
|
+
case 'policy_blocked': return 'Blocked by a governance policy';
|
|
3676
|
+
case 'modeling_gap': return 'Not modeled yet';
|
|
3677
|
+
case 'grounding_gap': return 'Not enough context to answer safely';
|
|
3678
|
+
case 'model_declined': return 'The assistant declined to answer';
|
|
3679
|
+
case 'provider_error': return 'The AI provider did not respond';
|
|
3680
|
+
case 'ambiguous': return 'Needs one detail before running';
|
|
3681
|
+
default: return 'No answer was produced';
|
|
3682
|
+
}
|
|
3683
|
+
};
|
|
3169
3684
|
const conversationRunExecutor = async ({ request, routeDecision, emitAnswerDelta }) => {
|
|
3170
3685
|
const kind = routeDecision?.conversationalKind ?? 'smalltalk';
|
|
3171
3686
|
const isGeneralKnowledge = routeDecision?.category === 'general_knowledge';
|
|
@@ -3182,6 +3697,15 @@ export async function startLocalServer(opts) {
|
|
|
3182
3697
|
let text = kind === 'answer_explanation'
|
|
3183
3698
|
? buildPriorAnswerExplanation(request.question, request.conversationContext)
|
|
3184
3699
|
: undefined;
|
|
3700
|
+
// A definitional question that NAMES a governed artifact is answerable from
|
|
3701
|
+
// the catalog: the description, domain, and dimensions are already recorded.
|
|
3702
|
+
// Reaching for a provider to paraphrase facts we hold can only add drift, and
|
|
3703
|
+
// the generic conversational reply this replaces used none of them.
|
|
3704
|
+
//
|
|
3705
|
+
// Returns undefined unless the question names something real, so a turn that
|
|
3706
|
+
// does not match keeps today's behaviour exactly.
|
|
3707
|
+
if (!text)
|
|
3708
|
+
text = buildGovernedObjectExplanation(request.question);
|
|
3185
3709
|
if (text) {
|
|
3186
3710
|
emitAnswerDelta?.(text);
|
|
3187
3711
|
}
|
|
@@ -3648,10 +4172,18 @@ export async function startLocalServer(opts) {
|
|
|
3648
4172
|
const conversationHistory = request.history?.length
|
|
3649
4173
|
? request.history
|
|
3650
4174
|
: conversationHistoryFromContext(request.conversationContext);
|
|
4175
|
+
// The provider that will plan the investigation as hypotheses. Absent or
|
|
4176
|
+
// unreachable, `planResearch` keeps its deterministic template, so
|
|
4177
|
+
// research never depends on a model being available.
|
|
4178
|
+
const researchPlanner = resolveGovernedAnswerRunner(projectRoot);
|
|
4179
|
+
const researchPlannerProvider = researchPlanner
|
|
4180
|
+
? createGovernedTextProvider(researchPlanner.provider, projectRoot)
|
|
4181
|
+
: undefined;
|
|
3651
4182
|
const plan = await planResearch({
|
|
3652
4183
|
question: request.question,
|
|
3653
4184
|
metrics,
|
|
3654
4185
|
blocks,
|
|
4186
|
+
...(researchPlannerProvider ? { provider: researchPlannerProvider } : {}),
|
|
3655
4187
|
intent: request.intent,
|
|
3656
4188
|
isFollowUp: conversationHistory.length > 0,
|
|
3657
4189
|
history: conversationHistory,
|
|
@@ -3687,6 +4219,7 @@ export async function startLocalServer(opts) {
|
|
|
3687
4219
|
plan,
|
|
3688
4220
|
};
|
|
3689
4221
|
let researchRun;
|
|
4222
|
+
const researchRuns = [];
|
|
3690
4223
|
let researchWorkspaceError;
|
|
3691
4224
|
if (!needsClarification) {
|
|
3692
4225
|
try {
|
|
@@ -3711,26 +4244,155 @@ export async function startLocalServer(opts) {
|
|
|
3711
4244
|
});
|
|
3712
4245
|
emit({
|
|
3713
4246
|
type: 'artifact.created',
|
|
3714
|
-
message: 'Saved
|
|
4247
|
+
message: 'Saved the immutable root research plan; executing bounded child branches.',
|
|
3715
4248
|
route: 'research',
|
|
3716
4249
|
trustState: 'review_required',
|
|
3717
|
-
payload: { researchRunId: created.id, notebookPath },
|
|
3718
|
-
});
|
|
3719
|
-
const executed = await runNotebookResearch(storage, created, {
|
|
3720
|
-
domain: agentRunWorkspaceValue(request, 'domain'),
|
|
3721
|
-
owner: agentRunWorkspaceValue(request, 'owner'),
|
|
3722
|
-
sourceCellFingerprint,
|
|
3723
|
-
question: request.question,
|
|
3724
|
-
intent: researchIntent,
|
|
3725
|
-
context: researchContextEnvelope,
|
|
3726
|
-
executionConnection: researchExecutionConnection,
|
|
3727
|
-
executionConnectionName: researchExecutionConnectionName,
|
|
3728
|
-
signal: request.signal,
|
|
3729
|
-
baselineSql: agentRunString(researchSource?.sql),
|
|
3730
|
-
baselineDqlArtifact: researchSource?.dqlArtifact,
|
|
3731
|
-
baselineRunId: agentRunString(researchSource?.runId),
|
|
4250
|
+
payload: { researchRunId: created.id, notebookPath, branchCap: 6 },
|
|
3732
4251
|
});
|
|
3733
|
-
|
|
4252
|
+
// The root record is a plan/dossier parent. Only child runs are
|
|
4253
|
+
// observed research executions. This prevents one root result from
|
|
4254
|
+
// being copied into six fabricated ledger entries (AGT-016/033).
|
|
4255
|
+
// An explicit Research request still gets one real child when the
|
|
4256
|
+
// catalog planner has no grounded step (for example, an empty
|
|
4257
|
+
// starter project). The child is an observed metadata/baseline
|
|
4258
|
+
// attempt, not a fabricated successful finding; its durable status
|
|
4259
|
+
// and receipt determine the ledger entry.
|
|
4260
|
+
const fallbackBranch = {
|
|
4261
|
+
thought: 'Inspect the requested analytical question against the frozen root context.',
|
|
4262
|
+
action: {
|
|
4263
|
+
kind: 'lookup_metric',
|
|
4264
|
+
target: routeDecision?.resolvedAnalyticalPlan?.executionId ?? request.question,
|
|
4265
|
+
},
|
|
4266
|
+
expectation: 'Whether the frozen context contains enough evidence for a bounded answer.',
|
|
4267
|
+
};
|
|
4268
|
+
const branches = capResearchBranches(plan.steps.length > 0 ? plan.steps : [fallbackBranch], 6);
|
|
4269
|
+
// The replan edge. Each branch tests one hypothesis; folding its
|
|
4270
|
+
// outcome back into the state is what lets the investigation stop
|
|
4271
|
+
// when the question is settled instead of grinding through a plan
|
|
4272
|
+
// frozen before any observation. `nextHypothesis` returning
|
|
4273
|
+
// undefined is how the loop learns to stop — it enforces the hop
|
|
4274
|
+
// budget and reports when nothing is open.
|
|
4275
|
+
let researchState = createResearchState(request.question, branches.map((branch, position) => ({
|
|
4276
|
+
id: `h${position + 1}`,
|
|
4277
|
+
statement: branch.thought,
|
|
4278
|
+
priorConfidence: 1 - position / (branches.length + 1),
|
|
4279
|
+
})));
|
|
4280
|
+
for (let index = 0; index < branches.length; index += 1) {
|
|
4281
|
+
const step = branches[index];
|
|
4282
|
+
if (request.signal?.aborted)
|
|
4283
|
+
rethrowIfCancelled(request.signal.reason, request.signal);
|
|
4284
|
+
// A hypothesis an earlier finding already closed is not
|
|
4285
|
+
// re-investigated, and an exhausted hop budget stops the run.
|
|
4286
|
+
const stillOpen = nextHypothesis(researchState);
|
|
4287
|
+
if (!stillOpen) {
|
|
4288
|
+
emit({
|
|
4289
|
+
type: 'executor.started',
|
|
4290
|
+
message: `Stopping early: ${researchState.hopsUsed} of ${branches.length} branches settled what could be settled.`,
|
|
4291
|
+
route: 'research',
|
|
4292
|
+
});
|
|
4293
|
+
break;
|
|
4294
|
+
}
|
|
4295
|
+
const branchId = `${step.action.kind}:${step.action.target}`;
|
|
4296
|
+
const branchQuestion = `${request.question}\nResearch branch ${index + 1} (${branchId}): ${step.expectation}`;
|
|
4297
|
+
const childId = `${created.id}:research:${index + 1}`;
|
|
4298
|
+
const child = storage.createRun({
|
|
4299
|
+
id: childId,
|
|
4300
|
+
notebookPath,
|
|
4301
|
+
title: `${agentRunTitle(request.question, 'Agent research')} · branch ${index + 1}`,
|
|
4302
|
+
question: branchQuestion,
|
|
4303
|
+
sourceCell,
|
|
4304
|
+
sourceCellId,
|
|
4305
|
+
sourceCellName,
|
|
4306
|
+
sourceCellFingerprint,
|
|
4307
|
+
intent: researchIntent,
|
|
4308
|
+
domain: agentRunWorkspaceValue(request, 'domain'),
|
|
4309
|
+
owner: agentRunWorkspaceValue(request, 'owner'),
|
|
4310
|
+
context: {
|
|
4311
|
+
...researchContextEnvelope,
|
|
4312
|
+
rootRunId: created.id,
|
|
4313
|
+
rootPlanId: plan.rootPlanId,
|
|
4314
|
+
branch: {
|
|
4315
|
+
id: branchId,
|
|
4316
|
+
index: index + 1,
|
|
4317
|
+
expectation: step.expectation,
|
|
4318
|
+
action: step.action,
|
|
4319
|
+
},
|
|
4320
|
+
},
|
|
4321
|
+
});
|
|
4322
|
+
emit({
|
|
4323
|
+
type: 'artifact.created',
|
|
4324
|
+
message: `Started research branch ${index + 1} of ${branches.length}.`,
|
|
4325
|
+
route: 'research',
|
|
4326
|
+
trustState: 'review_required',
|
|
4327
|
+
payload: { researchRunId: child.id, parentResearchRunId: created.id, branchId },
|
|
4328
|
+
});
|
|
4329
|
+
try {
|
|
4330
|
+
const executed = await runNotebookResearch(storage, child, {
|
|
4331
|
+
domain: agentRunWorkspaceValue(request, 'domain'),
|
|
4332
|
+
owner: agentRunWorkspaceValue(request, 'owner'),
|
|
4333
|
+
sourceCellFingerprint,
|
|
4334
|
+
question: branchQuestion,
|
|
4335
|
+
intent: researchIntent,
|
|
4336
|
+
context: {
|
|
4337
|
+
...researchContextEnvelope,
|
|
4338
|
+
rootRunId: created.id,
|
|
4339
|
+
rootPlanId: plan.rootPlanId,
|
|
4340
|
+
branch: { id: branchId, index: index + 1, expectation: step.expectation, action: step.action },
|
|
4341
|
+
},
|
|
4342
|
+
executionConnection: researchExecutionConnection,
|
|
4343
|
+
executionConnectionName: researchExecutionConnectionName,
|
|
4344
|
+
signal: request.signal,
|
|
4345
|
+
baselineSql: agentRunString(researchSource?.sql),
|
|
4346
|
+
baselineDqlArtifact: researchSource?.dqlArtifact,
|
|
4347
|
+
baselineRunId: agentRunString(researchSource?.runId),
|
|
4348
|
+
});
|
|
4349
|
+
const branchRun = withNotebookResearchChecklist(executed);
|
|
4350
|
+
researchRuns.push(branchRun);
|
|
4351
|
+
// Observe, then decide. A branch that produced rows is evidence
|
|
4352
|
+
// for its hypothesis; one that did not is inconclusive, which is
|
|
4353
|
+
// a real outcome and not a failure.
|
|
4354
|
+
// Rows are not support. A branch that returned data has been
|
|
4355
|
+
// OBSERVED, not confirmed — deciding whether the observation
|
|
4356
|
+
// matches what the hypothesis predicted needs the expectation,
|
|
4357
|
+
// and nothing available at this layer can judge it. Recording
|
|
4358
|
+
// rows as `supports` would let the dossier report a driver the
|
|
4359
|
+
// evidence never established, which is the failure mode the
|
|
4360
|
+
// whole verified-fact chain exists to prevent.
|
|
4361
|
+
researchState = applyFinding(researchState, {
|
|
4362
|
+
id: `f${index + 1}`,
|
|
4363
|
+
hypothesisId: `h${index + 1}`,
|
|
4364
|
+
verdict: 'inconclusive',
|
|
4365
|
+
summary: branchRun.summary ?? '',
|
|
4366
|
+
strength: (branchRun.resultPreview?.rows?.length ?? 0) > 0
|
|
4367
|
+
? 0.5
|
|
4368
|
+
: 0.1,
|
|
4369
|
+
});
|
|
4370
|
+
}
|
|
4371
|
+
catch (error) {
|
|
4372
|
+
// A child is a real durable run even when cancellation stops the
|
|
4373
|
+
// shared branch budget. Persist the truthful stop before
|
|
4374
|
+
// propagating cancellation to the parent run.
|
|
4375
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
4376
|
+
storage.updateRun(child.id, {
|
|
4377
|
+
status: 'error',
|
|
4378
|
+
error: message,
|
|
4379
|
+
summary: 'Research branch stopped before producing a result.',
|
|
4380
|
+
reviewStatus: 'needs_review',
|
|
4381
|
+
});
|
|
4382
|
+
const stopped = storage.getRun(child.id);
|
|
4383
|
+
if (stopped)
|
|
4384
|
+
researchRuns.push(withNotebookResearchChecklist(stopped));
|
|
4385
|
+
researchState = applyFinding(researchState, {
|
|
4386
|
+
id: `f${index + 1}`,
|
|
4387
|
+
hypothesisId: `h${index + 1}`,
|
|
4388
|
+
verdict: 'inconclusive',
|
|
4389
|
+
summary: message,
|
|
4390
|
+
strength: 0,
|
|
4391
|
+
});
|
|
4392
|
+
rethrowIfCancelled(error, request.signal);
|
|
4393
|
+
}
|
|
4394
|
+
}
|
|
4395
|
+
researchRun = researchRuns[0];
|
|
3734
4396
|
}
|
|
3735
4397
|
finally {
|
|
3736
4398
|
storage.close();
|
|
@@ -3741,15 +4403,81 @@ export async function startLocalServer(opts) {
|
|
|
3741
4403
|
researchWorkspaceError = formatNotebookResearchStorageError(error);
|
|
3742
4404
|
}
|
|
3743
4405
|
}
|
|
3744
|
-
const researchResultPreview =
|
|
4406
|
+
const researchResultPreview = researchRuns
|
|
4407
|
+
.map((run) => run.resultPreview)
|
|
4408
|
+
// Keep zero-row executions in the proof path. `coerceNarrateResultData`
|
|
4409
|
+
// intentionally omits empty row sets for prose, but an empty result is
|
|
4410
|
+
// still an executed result when it carries its execution identity.
|
|
4411
|
+
.find((preview) => {
|
|
4412
|
+
const record = agentRunRecord(preview);
|
|
4413
|
+
return Boolean(record && Array.isArray(record.rows));
|
|
4414
|
+
})
|
|
4415
|
+
?? researchRun?.resultPreview;
|
|
3745
4416
|
const researchResultData = coerceNarrateResultData(researchResultPreview);
|
|
3746
4417
|
const researchResultRecord = agentRunRecord(researchResultPreview);
|
|
4418
|
+
// Keep each bounded research branch inspectable as an evidence ledger.
|
|
4419
|
+
// The notebook workspace remains the durable execution record; this
|
|
4420
|
+
// additive projection gives Ask/Research narration a stable list of
|
|
4421
|
+
// observations, failures, facts, and receipts without exposing provider
|
|
4422
|
+
// chain-of-thought. Six is the hard branch budget for one root question.
|
|
4423
|
+
const researchLedger = buildResearchEvidenceLedger({
|
|
4424
|
+
rootQuestion: request.question,
|
|
4425
|
+
planId: plan.rootPlanId,
|
|
4426
|
+
snapshotId: routeDecision?.resolvedAnalyticalPlan?.snapshotId,
|
|
4427
|
+
entries: researchRuns.slice(0, 6).map((branch, index) => {
|
|
4428
|
+
const branchContext = agentRunRecord(branch.context)?.branch;
|
|
4429
|
+
const branchPreviewRecord = agentRunRecord(branch.resultPreview);
|
|
4430
|
+
const branchPreview = coerceNarrateResultData(branchPreviewRecord);
|
|
4431
|
+
const previewRecord = branchPreviewRecord;
|
|
4432
|
+
const executionReceipt = normalizeAnalyticalExecutionReceipt(previewRecord?.executionReceipt);
|
|
4433
|
+
const resultFingerprint = normalizeAnalyticalExecutionFingerprint(previewRecord?.resultFingerprint)
|
|
4434
|
+
?? executionReceipt?.resultFingerprint;
|
|
4435
|
+
// A child run ID or context-pack ID proves that planning happened,
|
|
4436
|
+
// not that a query executed. Only the canonical result fingerprint
|
|
4437
|
+
// (on the result or its execution receipt) can make a branch
|
|
4438
|
+
// observed (AGT-016/033).
|
|
4439
|
+
const executionProof = resultFingerprint;
|
|
4440
|
+
const observed = branch.status === 'ready' && Boolean(executionProof);
|
|
4441
|
+
return {
|
|
4442
|
+
id: branch.id,
|
|
4443
|
+
branchId: agentRunString(branchContext?.id) ?? `branch:${index + 1}`,
|
|
4444
|
+
question: branch.question,
|
|
4445
|
+
status: observed ? 'observed' : branch.status === 'error' ? 'failed' : 'skipped',
|
|
4446
|
+
...(branchPreviewRecord && Array.isArray(branchPreviewRecord.rows)
|
|
4447
|
+
? { rowCount: branchPreviewRecord.rows.length }
|
|
4448
|
+
: {}),
|
|
4449
|
+
...(resultFingerprint ? { resultFingerprint } : {}),
|
|
4450
|
+
...(executionReceipt ? { executionReceipt } : {}),
|
|
4451
|
+
facts: [agentRunString(branchContext?.expectation) ?? branch.question, ...(branch.evidence?.citations ?? []).flatMap((item) => typeof item === 'string' ? [item] : [])].slice(0, 8),
|
|
4452
|
+
// A context-pack ID or child run ID is not execution evidence. Keep
|
|
4453
|
+
// the receipt list empty until the execution service supplies a
|
|
4454
|
+
// receipt/fingerprint (AGT-016/033).
|
|
4455
|
+
receipts: executionProof ? [executionProof] : [],
|
|
4456
|
+
...(!observed ? {
|
|
4457
|
+
error: branch.error
|
|
4458
|
+
?? (branch.status === 'error'
|
|
4459
|
+
? branch.summary
|
|
4460
|
+
: 'Research branch did not produce an execution receipt or result fingerprint.'),
|
|
4461
|
+
} : {}),
|
|
4462
|
+
};
|
|
4463
|
+
}),
|
|
4464
|
+
stoppingReason: needsClarification
|
|
4465
|
+
? 'not_started'
|
|
4466
|
+
: researchRuns.some((run) => run.status === 'error')
|
|
4467
|
+
? 'insufficient_evidence'
|
|
4468
|
+
: plan.steps.length > 6
|
|
4469
|
+
? 'budget'
|
|
4470
|
+
: 'completed',
|
|
4471
|
+
});
|
|
3747
4472
|
// A query that ran and matched 0 rows STILL executed — treat it as a clean,
|
|
3748
4473
|
// grounded execution (not "no result"), so an empty answer is surfaced as
|
|
3749
4474
|
// "0 rows matched" rather than silently downgraded to review-required.
|
|
3750
4475
|
const researchDidExecute = Boolean(researchResultData) || Boolean(researchResultRecord && Array.isArray(researchResultRecord.rows));
|
|
3751
4476
|
const researchZeroRows = !researchResultData && researchDidExecute;
|
|
3752
|
-
const researchExecutedCleanly = researchDidExecute
|
|
4477
|
+
const researchExecutedCleanly = researchDidExecute
|
|
4478
|
+
&& !researchWorkspaceError
|
|
4479
|
+
&& researchRuns.length > 0
|
|
4480
|
+
&& researchRuns.every((run) => run.status === 'ready');
|
|
3753
4481
|
const narration = !needsClarification && researchResultData
|
|
3754
4482
|
? await narrateForAgentRun({
|
|
3755
4483
|
question: request.question,
|
|
@@ -3759,20 +4487,40 @@ export async function startLocalServer(opts) {
|
|
|
3759
4487
|
reviewRequired: true,
|
|
3760
4488
|
}, request.researchResultRowsOptIn === true)
|
|
3761
4489
|
: undefined;
|
|
4490
|
+
// The cross-branch story. Every branch tested a hypothesis and produced a
|
|
4491
|
+
// finding; narrating only the one result the executor happened to carry
|
|
4492
|
+
// reported a single fact and discarded the rest, which is the visible
|
|
4493
|
+
// half of "research answers one question instead of telling a story".
|
|
4494
|
+
const researchStory = !needsClarification && plan.steps.length > 0
|
|
4495
|
+
? synthesizeResearchNarrative({
|
|
4496
|
+
question: request.question,
|
|
4497
|
+
branches: researchRuns.map((branch, index) => ({
|
|
4498
|
+
statement: plan.steps[index]?.thought ?? branch.question ?? '',
|
|
4499
|
+
produced: branch.status === 'ready'
|
|
4500
|
+
&& (branch.resultPreview?.rows?.length ?? 0) > 0,
|
|
4501
|
+
...(branch.summary ? { summary: branch.summary } : {}),
|
|
4502
|
+
...(branch.status ? { status: branch.status } : {}),
|
|
4503
|
+
})),
|
|
4504
|
+
})
|
|
4505
|
+
: undefined;
|
|
3762
4506
|
const summary = needsClarification
|
|
3763
4507
|
? 'Needs clarification before running deeper research.'
|
|
3764
|
-
|
|
3765
|
-
|
|
3766
|
-
|
|
3767
|
-
|
|
3768
|
-
|
|
3769
|
-
|
|
3770
|
-
|
|
3771
|
-
|
|
3772
|
-
|
|
3773
|
-
|
|
3774
|
-
|
|
3775
|
-
|
|
4508
|
+
// The story leads; the verified-fact narration follows it, so the
|
|
4509
|
+
// numbers still come from the narrator that checks them.
|
|
4510
|
+
: researchStory
|
|
4511
|
+
? `${researchStory}${narration?.summary ? `\n\n${narration.summary}` : ''}`
|
|
4512
|
+
: narration?.summary
|
|
4513
|
+
?? (researchZeroRows
|
|
4514
|
+
? 'The query executed cleanly against real data and matched 0 rows.'
|
|
4515
|
+
: researchRun?.status === 'ready'
|
|
4516
|
+
? 'Saved a grounded research dossier with context evidence and next review actions.'
|
|
4517
|
+
: researchRun?.status === 'error'
|
|
4518
|
+
? 'Saved a research dossier, but the preview needs review before promotion.'
|
|
4519
|
+
: researchWorkspaceError
|
|
4520
|
+
? 'Prepared a grounded research plan; durable research storage is unavailable in this runtime.'
|
|
4521
|
+
: plan.done
|
|
4522
|
+
? 'Prepared a direct grounded-answer plan.'
|
|
4523
|
+
: 'Prepared a grounded research plan over real DQL assets.');
|
|
3776
4524
|
return {
|
|
3777
4525
|
summary,
|
|
3778
4526
|
answer: plan.followUp?.question ?? narration?.summary
|
|
@@ -3785,8 +4533,11 @@ export async function startLocalServer(opts) {
|
|
|
3785
4533
|
? []
|
|
3786
4534
|
: [agentRunArtifact('research_run', 'Research plan', {
|
|
3787
4535
|
plan,
|
|
4536
|
+
researchLedger,
|
|
3788
4537
|
researchRun,
|
|
4538
|
+
researchRuns,
|
|
3789
4539
|
researchRunId: researchRun?.id,
|
|
4540
|
+
researchRunIds: researchRuns.map((run) => run.id),
|
|
3790
4541
|
notebookPath,
|
|
3791
4542
|
workspaceError: researchWorkspaceError,
|
|
3792
4543
|
routeDecision,
|
|
@@ -3817,7 +4568,7 @@ export async function startLocalServer(opts) {
|
|
|
3817
4568
|
: [
|
|
3818
4569
|
...(researchRun?.id ? [{ id: 'open-research', label: 'Open research dossier', artifactKind: 'research_run' }] : []),
|
|
3819
4570
|
{ id: 'create-block', label: 'Review DQL draft', route: 'dql_block_draft', artifactKind: 'dql_block_draft' },
|
|
3820
|
-
...(
|
|
4571
|
+
...(researchRuns.some((run) => run.generatedSql || run.reviewedSql) ? [{ id: 'insert-sql', label: 'Insert SQL preview', route: 'sql_cell', artifactKind: 'sql_cell' }] : []),
|
|
3821
4572
|
],
|
|
3822
4573
|
};
|
|
3823
4574
|
},
|
|
@@ -4117,6 +4868,22 @@ export async function startLocalServer(opts) {
|
|
|
4117
4868
|
// lookup, and governed execution for the lifetime of a request. This removes
|
|
4118
4869
|
// both positional catalog truncation and the previous duplicate retrieval pass.
|
|
4119
4870
|
const preparedAgentContextPacks = new WeakMap();
|
|
4871
|
+
/**
|
|
4872
|
+
* Cross-encoder pass over the fused candidates, when a provider is available.
|
|
4873
|
+
* Advisory throughout: it may only reorder ids retrieval returned, and any
|
|
4874
|
+
* failure leaves retrieval's own ordering in place.
|
|
4875
|
+
*/
|
|
4876
|
+
const agentRerankCandidates = (() => {
|
|
4877
|
+
const governed = resolveGovernedAnswerRunner(projectRoot);
|
|
4878
|
+
const provider = governed
|
|
4879
|
+
? createGovernedTextProvider(governed.provider, projectRoot)
|
|
4880
|
+
: undefined;
|
|
4881
|
+
if (!provider)
|
|
4882
|
+
return undefined;
|
|
4883
|
+
return (question, candidates) => rerankCandidates(provider, question, candidates, {
|
|
4884
|
+
timeoutMs: Math.round(2_500 * deadlineScale()),
|
|
4885
|
+
});
|
|
4886
|
+
})();
|
|
4120
4887
|
const pendingAgentContextPacks = new WeakMap();
|
|
4121
4888
|
const buildAgentRunContextPack = async (request) => {
|
|
4122
4889
|
const prepared = preparedAgentContextPacks.get(request);
|
|
@@ -4184,6 +4951,10 @@ export async function startLocalServer(opts) {
|
|
|
4184
4951
|
},
|
|
4185
4952
|
strictness: request.analysisDepth === 'deep' ? 'exploratory' : 'balanced',
|
|
4186
4953
|
limit: request.analysisDepth === 'deep' ? 120 : 80,
|
|
4954
|
+
// The runtime PRE-BUILDS this pack, so wiring the reranker only at the
|
|
4955
|
+
// provider's own `buildLocalContextPack` left it unreachable on the
|
|
4956
|
+
// common path — the prepared pack is used and that call never happens.
|
|
4957
|
+
...(agentRerankCandidates ? { rerankCandidates: agentRerankCandidates } : {}),
|
|
4187
4958
|
domainContext: requestedDomain
|
|
4188
4959
|
? resolveUiDomainContext({
|
|
4189
4960
|
manifest: snapshot.manifest,
|
|
@@ -4239,6 +5010,56 @@ export async function startLocalServer(opts) {
|
|
|
4239
5010
|
};
|
|
4240
5011
|
// Compact fallback used only for plain conversational replies. Analytical
|
|
4241
5012
|
// turns use the structured, question-ranked evidence path above.
|
|
5013
|
+
/**
|
|
5014
|
+
* Explain a governed artifact the question names, from catalog metadata alone.
|
|
5015
|
+
*
|
|
5016
|
+
* Certified blocks are offered first: when a concept exists both as a
|
|
5017
|
+
* certified block and a raw model, the certified one is the authored
|
|
5018
|
+
* definition and the other is an implementation detail.
|
|
5019
|
+
*/
|
|
5020
|
+
const buildGovernedObjectExplanation = (question) => {
|
|
5021
|
+
try {
|
|
5022
|
+
const blocks = collectPlanBlocks(projectRoot, { certifiedOnly: true });
|
|
5023
|
+
const certifiedNames = new Set(blocks.map((block) => block.name));
|
|
5024
|
+
const all = [
|
|
5025
|
+
...blocks.map((block) => ({ block, status: 'certified' })),
|
|
5026
|
+
...collectPlanBlocks(projectRoot, { certifiedOnly: false })
|
|
5027
|
+
.filter((block) => !certifiedNames.has(block.name))
|
|
5028
|
+
.map((block) => ({ block, status: 'draft' })),
|
|
5029
|
+
];
|
|
5030
|
+
// Metrics as well as blocks. "How is revenue defined here?" names a
|
|
5031
|
+
// semantic metric, not a block, and answering it from the metric's own
|
|
5032
|
+
// description is the whole point of holding one.
|
|
5033
|
+
const metricObjects = loadSemanticMetrics(projectRoot).map((metric) => ({
|
|
5034
|
+
objectKey: `semantic:metric:${metric.name}`,
|
|
5035
|
+
objectType: 'semantic_metric',
|
|
5036
|
+
name: metric.name,
|
|
5037
|
+
...(metric.description ? { description: metric.description } : {}),
|
|
5038
|
+
...(metric.domain ? { domain: metric.domain } : {}),
|
|
5039
|
+
status: 'governed',
|
|
5040
|
+
payload: {},
|
|
5041
|
+
}));
|
|
5042
|
+
const explanation = composeBusinessExplanation(question, [
|
|
5043
|
+
...all.map(({ block, status }) => ({
|
|
5044
|
+
objectKey: `dql:block:${block.name}`,
|
|
5045
|
+
objectType: 'dql_block',
|
|
5046
|
+
name: block.name,
|
|
5047
|
+
...(block.description ? { description: block.description } : {}),
|
|
5048
|
+
...(block.domain ? { domain: block.domain } : {}),
|
|
5049
|
+
status,
|
|
5050
|
+
payload: {
|
|
5051
|
+
...(block.dimensions?.length ? { dimensions: block.dimensions } : {}),
|
|
5052
|
+
},
|
|
5053
|
+
})),
|
|
5054
|
+
...metricObjects,
|
|
5055
|
+
]);
|
|
5056
|
+
return explanation?.text;
|
|
5057
|
+
}
|
|
5058
|
+
catch {
|
|
5059
|
+
// Never let an explanation attempt break a conversational turn.
|
|
5060
|
+
return undefined;
|
|
5061
|
+
}
|
|
5062
|
+
};
|
|
4242
5063
|
const buildAgentRunCatalogContext = () => {
|
|
4243
5064
|
try {
|
|
4244
5065
|
const blocks = collectPlanBlocks(projectRoot, { certifiedOnly: true });
|
|
@@ -4288,11 +5109,12 @@ export async function startLocalServer(opts) {
|
|
|
4288
5109
|
},
|
|
4289
5110
|
getCatalogContext: buildRankedAgentRunCatalogContext,
|
|
4290
5111
|
});
|
|
4291
|
-
//
|
|
4292
|
-
//
|
|
4293
|
-
//
|
|
4294
|
-
//
|
|
5112
|
+
// Explicit selections and conversation-only turns remain deterministic;
|
|
5113
|
+
// each fresh natural-language analytical turn gets one bounded candidate-ID
|
|
5114
|
+
// interpretation call by default. The rollback is host-owned and cannot be
|
|
5115
|
+
// supplied by an HTTP/MCP client.
|
|
4295
5116
|
const agentRunRouter = createHybridRouter({
|
|
5117
|
+
requireMeaningCallForNaturalLanguage: opts.requireMeaningCallForNaturalLanguage ?? true,
|
|
4296
5118
|
complete: async ({ system, user, signal }) => {
|
|
4297
5119
|
const provider = await createBlockStudioAssistProvider(projectRoot);
|
|
4298
5120
|
if (!provider)
|
|
@@ -4332,6 +5154,25 @@ export async function startLocalServer(opts) {
|
|
|
4332
5154
|
// A run may outlive its streaming browser connection, so cancellation is
|
|
4333
5155
|
// server-owned and keyed by run id rather than relying on fetch abort alone.
|
|
4334
5156
|
const activeAgentRunControllers = new Map();
|
|
5157
|
+
const cancelActiveAgentRun = (id) => {
|
|
5158
|
+
const controller = activeAgentRunControllers.get(id);
|
|
5159
|
+
if (!controller)
|
|
5160
|
+
return false;
|
|
5161
|
+
const progress = agentRunStore.getProgress(id);
|
|
5162
|
+
if (progress) {
|
|
5163
|
+
agentRunStore.saveProgress({
|
|
5164
|
+
...progress,
|
|
5165
|
+
lifecycle: {
|
|
5166
|
+
...progress.lifecycle,
|
|
5167
|
+
state: 'cancelling',
|
|
5168
|
+
revision: progress.lifecycle.revision + 1,
|
|
5169
|
+
updatedAt: new Date().toISOString(),
|
|
5170
|
+
},
|
|
5171
|
+
});
|
|
5172
|
+
}
|
|
5173
|
+
controller.abort(createAgentRunCancellationError());
|
|
5174
|
+
return true;
|
|
5175
|
+
};
|
|
4335
5176
|
const activeAnalyticalRepairReservations = new Set();
|
|
4336
5177
|
const consumedAnalyticalRepairCapabilities = new Set();
|
|
4337
5178
|
// Server-side conversation threads: persisted multi-turn state (survives refresh).
|
|
@@ -4770,12 +5611,43 @@ export async function startLocalServer(opts) {
|
|
|
4770
5611
|
* string, so the only remaining reasons to differ are the connection and the
|
|
4771
5612
|
* governance gates, both of which report themselves honestly.
|
|
4772
5613
|
*/
|
|
4773
|
-
const executeGeneratedSqlDirect = async (question, sql, seed, executionConnection, bindings, executionConnectionName) => {
|
|
5614
|
+
const executeGeneratedSqlDirect = async (question, sql, seed, executionConnection, bindings, executionConnectionName, agenticCapability, agenticScope) => {
|
|
4774
5615
|
const activeConnection = requireActiveConnection(executionConnection);
|
|
4775
5616
|
const rowBound = clampAnalyticalRowBound(seed?.limit ?? 200);
|
|
4776
5617
|
const trimmed = sql.trim().replace(/;\s*$/, '').trim();
|
|
4777
5618
|
if (!trimmed)
|
|
4778
5619
|
throw analyticalError('The generated SQL was empty.', { origin: 'host', stage: 'execute' });
|
|
5620
|
+
const bindingValue = {
|
|
5621
|
+
sqlParams: bindings?.sqlParams ?? [],
|
|
5622
|
+
variables: bindings?.variables ?? {},
|
|
5623
|
+
};
|
|
5624
|
+
if (agenticCapability) {
|
|
5625
|
+
// The proposal itself is immutable. A changed literal, comment, or
|
|
5626
|
+
// whitespace is drift here rather than a benign formatting change.
|
|
5627
|
+
const currentSnapshotId = projectSnapshot().snapshotId;
|
|
5628
|
+
let targetFingerprint;
|
|
5629
|
+
if (agenticCapability.targetFingerprint) {
|
|
5630
|
+
try {
|
|
5631
|
+
targetFingerprint = (await observeWarehouseTargetIdentity(executor, activeConnection)).identityFingerprint;
|
|
5632
|
+
}
|
|
5633
|
+
catch {
|
|
5634
|
+
throw analyticalError('DQL could not re-confirm the selected execution target, so the analyst-approved query was not run.', {
|
|
5635
|
+
origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
|
|
5636
|
+
});
|
|
5637
|
+
}
|
|
5638
|
+
}
|
|
5639
|
+
const capabilityVerdict = verifyAgenticSqlExecutionCapability(agenticCapability, sql, {
|
|
5640
|
+
...agenticScope,
|
|
5641
|
+
bindings: bindingValue,
|
|
5642
|
+
snapshotId: currentSnapshotId,
|
|
5643
|
+
...(targetFingerprint ? { targetFingerprint } : {}),
|
|
5644
|
+
});
|
|
5645
|
+
if (!capabilityVerdict.ok) {
|
|
5646
|
+
throw analyticalError(capabilityVerdict.reason ?? 'The analyst-approved SQL no longer matches this execution.', {
|
|
5647
|
+
origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
|
|
5648
|
+
});
|
|
5649
|
+
}
|
|
5650
|
+
}
|
|
4779
5651
|
// RESOLVE FIRST, VALIDATE SECOND. Executing verbatim means executing the
|
|
4780
5652
|
// statement the notebook would execute — and the notebook resolves
|
|
4781
5653
|
// `@metric()` / `@dim()` refs and dbt macros before it runs anything.
|
|
@@ -4800,8 +5672,6 @@ export async function startLocalServer(opts) {
|
|
|
4800
5672
|
// release denied; a governance boundary must not move as a side effect.
|
|
4801
5673
|
const sourceDomain = seed?.source?.match(/\bdomain\s*=\s*"([^"]+)"/i)?.[1] ?? 'uncategorized';
|
|
4802
5674
|
assertAppAccess({ app, domain: sourceDomain, level: 'execute' });
|
|
4803
|
-
// A semantic query the loop already compiled (MetricFlow / dbt Cloud) must
|
|
4804
|
-
// execute through its pinned target binding, not as loose SQL.
|
|
4805
5675
|
const semanticExecutionHolder = { value: null };
|
|
4806
5676
|
const execution = await analyticalExecutionService.execute({
|
|
4807
5677
|
sql: semantic.sql,
|
|
@@ -4813,27 +5683,46 @@ export async function startLocalServer(opts) {
|
|
|
4813
5683
|
variables: bindings?.variables,
|
|
4814
5684
|
semanticRefs: semantic.semanticRefs,
|
|
4815
5685
|
executePrepared: async (preparation) => {
|
|
4816
|
-
|
|
4817
|
-
|
|
4818
|
-
|
|
4819
|
-
|
|
4820
|
-
|
|
4821
|
-
|
|
4822
|
-
|
|
4823
|
-
|
|
4824
|
-
|
|
4825
|
-
|
|
4826
|
-
|
|
4827
|
-
|
|
4828
|
-
|
|
4829
|
-
|
|
4830
|
-
|
|
4831
|
-
|
|
4832
|
-
|
|
4833
|
-
|
|
5686
|
+
// Both the native executor and the target-bound semantic adapter can
|
|
5687
|
+
// reach the warehouse from this callback. Admit the exact prepared
|
|
5688
|
+
// bytes before either path so a matching cached semantic compile cannot
|
|
5689
|
+
// bypass the generated proposal's capability.
|
|
5690
|
+
return executePreparedAgenticSqlBoundary({
|
|
5691
|
+
capability: agenticCapability,
|
|
5692
|
+
preparedSql: preparation.executedSql,
|
|
5693
|
+
bindings: bindingValue,
|
|
5694
|
+
scope: {
|
|
5695
|
+
...agenticScope,
|
|
5696
|
+
snapshotId: projectSnapshot().snapshotId,
|
|
5697
|
+
...(agenticCapability ? { targetFingerprint: agenticCapability.targetFingerprint } : {}),
|
|
5698
|
+
},
|
|
5699
|
+
execute: async () => {
|
|
5700
|
+
// A semantic query the loop already compiled (MetricFlow / dbt
|
|
5701
|
+
// Cloud) must execute through its pinned target binding, not as
|
|
5702
|
+
// loose SQL.
|
|
5703
|
+
const pinnedSemanticCompile = compiledSemanticQueries.get(executionFingerprint(preparation.preparedSql))
|
|
5704
|
+
?? compiledSemanticQueries.get(executionFingerprint(trimmed));
|
|
5705
|
+
if (pinnedSemanticCompile) {
|
|
5706
|
+
const semanticExecution = await executeTargetBoundSemanticQuery({
|
|
5707
|
+
executor,
|
|
5708
|
+
connection: activeConnection,
|
|
5709
|
+
projectRoot,
|
|
5710
|
+
plannedAdapter: pinnedSemanticCompile.engine,
|
|
5711
|
+
metricFlow: pinnedSemanticCompile.engine === 'metricflow-cli'
|
|
5712
|
+
? resolveMetricFlowTargetMetadata(projectRoot, projectConfig)
|
|
5713
|
+
: undefined,
|
|
5714
|
+
compile: async () => pinnedSemanticCompile,
|
|
5715
|
+
prepareSql: () => ({ sql: preparation.executedSql, connection: preparation.connection }),
|
|
5716
|
+
rowBound,
|
|
5717
|
+
});
|
|
5718
|
+
if (semanticExecution) {
|
|
5719
|
+
semanticExecutionHolder.value = semanticExecution;
|
|
5720
|
+
return semanticExecution.result;
|
|
5721
|
+
}
|
|
5722
|
+
}
|
|
5723
|
+
return executor.executeQuery(preparation.executedSql, bindings?.sqlParams ?? [], runtimeVariables(bindings?.variables ?? {}), preparation.connection);
|
|
4834
5724
|
}
|
|
4835
|
-
}
|
|
4836
|
-
return executor.executeQuery(preparation.executedSql, bindings?.sqlParams ?? [], runtimeVariables(bindings?.variables ?? {}), preparation.connection);
|
|
5725
|
+
});
|
|
4837
5726
|
},
|
|
4838
5727
|
});
|
|
4839
5728
|
const semanticExecution = semanticExecutionHolder.value;
|
|
@@ -4891,14 +5780,19 @@ export async function startLocalServer(opts) {
|
|
|
4891
5780
|
executableArtifact,
|
|
4892
5781
|
};
|
|
4893
5782
|
};
|
|
4894
|
-
const executeGeneratedArtifactForAgent = async (question, sql, seed, executionConnection, executionConnectionName) => {
|
|
5783
|
+
const executeGeneratedArtifactForAgent = async (question, sql, seed, executionConnection, executionConnectionName, agenticCapability, agenticScope) => {
|
|
4895
5784
|
// A seed that is already a certified/saved artifact keeps the DQL-first
|
|
4896
5785
|
// path: there the `.dql` source IS the contract, and its parameters and
|
|
4897
5786
|
// semantic refs must be compiled, not bypassed.
|
|
4898
5787
|
if (seed && seed.kind !== 'sql_block') {
|
|
5788
|
+
if (agenticCapability) {
|
|
5789
|
+
throw analyticalError('The analyst-approved SQL cannot be redirected through a saved artifact.', {
|
|
5790
|
+
origin: 'governance_gate', stage: 'execute', code: 'unauthorized_sql',
|
|
5791
|
+
});
|
|
5792
|
+
}
|
|
4899
5793
|
return executeArtifactReferenceForAgent({ ...seed, limit: seed.limit ?? 200 }, question, executionConnection, executionConnectionName);
|
|
4900
5794
|
}
|
|
4901
|
-
return executeGeneratedSqlDirect(question, sql, seed, executionConnection, undefined, executionConnectionName);
|
|
5795
|
+
return executeGeneratedSqlDirect(question, sql, seed, executionConnection, undefined, executionConnectionName, agenticCapability, agenticScope);
|
|
4902
5796
|
};
|
|
4903
5797
|
/**
|
|
4904
5798
|
* EXP-001: execution host for the deliberately narrow non-governed lane.
|
|
@@ -5506,7 +6400,10 @@ export async function startLocalServer(opts) {
|
|
|
5506
6400
|
let invocation;
|
|
5507
6401
|
let plan;
|
|
5508
6402
|
try {
|
|
5509
|
-
|
|
6403
|
+
// The source label is echoed into parse errors, which reach the user.
|
|
6404
|
+
// `<bounded-dql-repair>` is an internal artifact id and tells them nothing;
|
|
6405
|
+
// it appeared verbatim in a reported failure card.
|
|
6406
|
+
const program = new Parser(repairedSource, 'repaired query').parse();
|
|
5510
6407
|
const blocks = program.statements.filter((statement) => statement.kind === NodeKind.BlockDecl);
|
|
5511
6408
|
if (blocks.length !== 1 || program.statements.length !== 1) {
|
|
5512
6409
|
throw new Error('The repaired source must contain exactly one DQL block.');
|
|
@@ -6113,6 +7010,19 @@ export async function startLocalServer(opts) {
|
|
|
6113
7010
|
: notebookResearchString(governedAnswer?.answer)
|
|
6114
7011
|
?? notebookResearchString(governedAnswer?.text)
|
|
6115
7012
|
?? notebookResearchSummary(question, resultPreview, previewError);
|
|
7013
|
+
const previewRecord = agentRunRecord(resultPreview);
|
|
7014
|
+
const executionReceipt = normalizeAnalyticalExecutionReceipt(previewRecord?.executionReceipt);
|
|
7015
|
+
// Do not treat a child run ID as execution evidence. The canonical
|
|
7016
|
+
// fingerprint is the only proof that this Research branch produced a
|
|
7017
|
+
// result; planning, SQL text, and a durable run record remain review
|
|
7018
|
+
// required until that proof exists (AGT-016/033).
|
|
7019
|
+
const executionProof = normalizeAnalyticalExecutionFingerprint(previewRecord?.resultFingerprint)
|
|
7020
|
+
?? executionReceipt?.resultFingerprint;
|
|
7021
|
+
const executionUnavailable = !previewError && !executionProof;
|
|
7022
|
+
const terminalError = previewError
|
|
7023
|
+
?? (executionUnavailable
|
|
7024
|
+
? 'Research did not produce an executed result or execution receipt; the branch remains review-required.'
|
|
7025
|
+
: undefined);
|
|
6116
7026
|
const recommendation = previewError
|
|
6117
7027
|
? 'Review the SQL, selected metadata, and connection context before rerunning.'
|
|
6118
7028
|
: dqlArtifact && !reviewedSql
|
|
@@ -6137,8 +7047,14 @@ export async function startLocalServer(opts) {
|
|
|
6137
7047
|
question,
|
|
6138
7048
|
intent,
|
|
6139
7049
|
context,
|
|
6140
|
-
|
|
6141
|
-
|
|
7050
|
+
// A plan, generated SQL, or DQL artifact is not an observed execution.
|
|
7051
|
+
// Only a result carrying an execution fingerprint/receipt may become
|
|
7052
|
+
// `ready`; otherwise persist a failed review state with no fabricated
|
|
7053
|
+
// evidence (AGT-016/033).
|
|
7054
|
+
status: terminalError ? 'error' : 'ready',
|
|
7055
|
+
summary: terminalError && executionUnavailable
|
|
7056
|
+
? 'Research branch stopped without an executed result; no observation was recorded.'
|
|
7057
|
+
: summary,
|
|
6142
7058
|
recommendation,
|
|
6143
7059
|
resultPreview,
|
|
6144
7060
|
evidence,
|
|
@@ -6158,7 +7074,7 @@ export async function startLocalServer(opts) {
|
|
|
6158
7074
|
...(display && display.ok ? display.warnings : []),
|
|
6159
7075
|
],
|
|
6160
7076
|
reviewStatus: 'needs_review',
|
|
6161
|
-
error:
|
|
7077
|
+
error: terminalError,
|
|
6162
7078
|
lastRunAt: startedAt,
|
|
6163
7079
|
}) ?? run;
|
|
6164
7080
|
}
|
|
@@ -6985,7 +7901,12 @@ export async function startLocalServer(opts) {
|
|
|
6985
7901
|
return;
|
|
6986
7902
|
}
|
|
6987
7903
|
if (operationMatch && req.method === 'DELETE') {
|
|
6988
|
-
const
|
|
7904
|
+
const operationId = decodeURIComponent(operationMatch[1]);
|
|
7905
|
+
const existingOperation = operationCoordinator.get(operationId);
|
|
7906
|
+
if (existingOperation?.type === 'agent_run' && existingOperation.scope.startsWith('agent-run:')) {
|
|
7907
|
+
cancelActiveAgentRun(existingOperation.scope.slice('agent-run:'.length));
|
|
7908
|
+
}
|
|
7909
|
+
const operation = operationCoordinator.cancel(operationId);
|
|
6989
7910
|
res.writeHead(operation ? 200 : 404, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
6990
7911
|
res.end(serializeJSON(operation ?? { error: 'Operation not found.' }));
|
|
6991
7912
|
return;
|
|
@@ -8098,12 +9019,17 @@ export async function startLocalServer(opts) {
|
|
|
8098
9019
|
const limit = Number.isFinite(rawLimit) && rawLimit > 0
|
|
8099
9020
|
? Math.min(200, Math.floor(rawLimit))
|
|
8100
9021
|
: 50;
|
|
8101
|
-
const
|
|
8102
|
-
|
|
9022
|
+
const stored = agentRunStore.list();
|
|
9023
|
+
// This is an INDEX payload: one row per run, no answer bodies. Shipping
|
|
9024
|
+
// stored runs whole meant a measured 47.61 MB for 20 real runs.
|
|
9025
|
+
// `GET /api/agent-runs/:id` still serves the complete immutable record.
|
|
9026
|
+
const runs = stored
|
|
9027
|
+
.slice()
|
|
8103
9028
|
.sort((a, b) => b.startedAt.localeCompare(a.startedAt))
|
|
8104
|
-
.slice(0, limit)
|
|
9029
|
+
.slice(0, limit)
|
|
9030
|
+
.map(agentRunListEntryForTransport);
|
|
8105
9031
|
res.writeHead(200, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
8106
|
-
res.end(serializeJSON({ runs, total:
|
|
9032
|
+
res.end(serializeJSON({ runs, total: stored.length, limit }));
|
|
8107
9033
|
return;
|
|
8108
9034
|
}
|
|
8109
9035
|
/**
|
|
@@ -8641,6 +9567,14 @@ export async function startLocalServer(opts) {
|
|
|
8641
9567
|
res.end(serializeJSON({ error: 'Agent run not found.' }));
|
|
8642
9568
|
return;
|
|
8643
9569
|
}
|
|
9570
|
+
if (run.status !== 'blocked' || run.stopReason !== 'blocked') {
|
|
9571
|
+
res.writeHead(409, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
9572
|
+
res.end(serializeJSON({
|
|
9573
|
+
code: 'REPAIR_CAPABILITY_REQUIRED',
|
|
9574
|
+
error: 'Only a terminal blocked run with a blocked stop reason may derive an analytical repair.',
|
|
9575
|
+
}));
|
|
9576
|
+
return;
|
|
9577
|
+
}
|
|
8644
9578
|
const source = analyticalFailedRunFromAgentRun(run);
|
|
8645
9579
|
if (!source) {
|
|
8646
9580
|
res.writeHead(409, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
@@ -8679,25 +9613,11 @@ export async function startLocalServer(opts) {
|
|
|
8679
9613
|
if (req.method === 'POST' && /^\/api\/agent-runs\/[^/]+\/cancel$/.test(path)) {
|
|
8680
9614
|
const match = path.match(/^\/api\/agent-runs\/([^/]+)\/cancel$/);
|
|
8681
9615
|
const id = decodeURIComponent(match?.[1] ?? '');
|
|
8682
|
-
|
|
8683
|
-
if (!controller) {
|
|
9616
|
+
if (!cancelActiveAgentRun(id)) {
|
|
8684
9617
|
res.writeHead(404, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
8685
9618
|
res.end(serializeJSON({ ok: false, error: 'This run is no longer active.' }));
|
|
8686
9619
|
return;
|
|
8687
9620
|
}
|
|
8688
|
-
const progress = agentRunStore.getProgress(id);
|
|
8689
|
-
if (progress) {
|
|
8690
|
-
agentRunStore.saveProgress({
|
|
8691
|
-
...progress,
|
|
8692
|
-
lifecycle: {
|
|
8693
|
-
...progress.lifecycle,
|
|
8694
|
-
state: 'cancelling',
|
|
8695
|
-
revision: progress.lifecycle.revision + 1,
|
|
8696
|
-
updatedAt: new Date().toISOString(),
|
|
8697
|
-
},
|
|
8698
|
-
});
|
|
8699
|
-
}
|
|
8700
|
-
controller.abort(new Error('Stopped by user.'));
|
|
8701
9621
|
res.writeHead(202, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
8702
9622
|
res.end(serializeJSON({ ok: true, id }));
|
|
8703
9623
|
return;
|
|
@@ -8842,7 +9762,10 @@ export async function startLocalServer(opts) {
|
|
|
8842
9762
|
return;
|
|
8843
9763
|
}
|
|
8844
9764
|
res.writeHead(201, { 'Content-Type': 'application/json; charset=utf-8' });
|
|
8845
|
-
|
|
9765
|
+
// Same run, same reader as the stream above, so it ships the same
|
|
9766
|
+
// projection. Sending the stored record here instead made an ordinary
|
|
9767
|
+
// completed Ask a 4.66 MB reply.
|
|
9768
|
+
res.end(serializeJSON({ run: slimAgentRunForTransport(completedRun) }));
|
|
8846
9769
|
}
|
|
8847
9770
|
finally {
|
|
8848
9771
|
if (runId)
|
|
@@ -17527,21 +18450,31 @@ function connectionDriverLabel(connection) {
|
|
|
17527
18450
|
* The notebook SPA expects columns as string[] (just names).
|
|
17528
18451
|
*/
|
|
17529
18452
|
function normalizeQueryResult(result, semanticRefs) {
|
|
17530
|
-
const
|
|
17531
|
-
|
|
18453
|
+
const canonical = normalizeCanonicalQueryResult({
|
|
18454
|
+
columns: result?.columns,
|
|
18455
|
+
rows: result?.rows,
|
|
18456
|
+
rowCount: result?.rowCount,
|
|
18457
|
+
executionTime: result?.executionTime,
|
|
18458
|
+
executionTimeMs: result?.executionTimeMs,
|
|
18459
|
+
truncated: result?.truncated,
|
|
18460
|
+
resultFingerprint: result?.resultFingerprint,
|
|
18461
|
+
executionReceipt: result?.executionReceipt,
|
|
18462
|
+
trustState: result?.trustState,
|
|
18463
|
+
answerTier: result?.answerTier,
|
|
18464
|
+
});
|
|
17532
18465
|
const rawRows = Array.isArray(result?.rows) ? result.rows : [];
|
|
17533
|
-
const rows =
|
|
18466
|
+
const rows = canonical.rows.slice(0, NOTEBOOK_EXECUTE_PREVIEW_ROW_LIMIT);
|
|
17534
18467
|
const hasRefs = semanticRefs && (semanticRefs.metrics.length > 0 || semanticRefs.dimensions.length > 0);
|
|
17535
18468
|
return {
|
|
17536
|
-
columns,
|
|
18469
|
+
columns: canonical.columns,
|
|
17537
18470
|
rows,
|
|
17538
|
-
rowCount:
|
|
17539
|
-
|
|
17540
|
-
|
|
17541
|
-
|
|
17542
|
-
|
|
17543
|
-
|
|
17544
|
-
...(
|
|
18471
|
+
rowCount: canonical.rowCount,
|
|
18472
|
+
resultFingerprint: canonical.resultFingerprint,
|
|
18473
|
+
executionTime: canonical.executionTime ?? 0,
|
|
18474
|
+
...(rawRows.length > rows.length || canonical.truncated ? { truncated: true } : {}),
|
|
18475
|
+
...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
|
|
18476
|
+
...(canonical.trustState ? { trustState: canonical.trustState } : {}),
|
|
18477
|
+
...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
|
|
17545
18478
|
...(hasRefs ? { semanticRefs } : {}),
|
|
17546
18479
|
};
|
|
17547
18480
|
}
|
|
@@ -25400,7 +26333,13 @@ async function createBlockStudioAssistProvider(projectRoot, requestedProvider) {
|
|
|
25400
26333
|
default:
|
|
25401
26334
|
return null;
|
|
25402
26335
|
}
|
|
25403
|
-
|
|
26336
|
+
// Route through the eval cassette too. This constructor serves the MEANING
|
|
26337
|
+
// call and narration — the two dispatches that decide routing and wording —
|
|
26338
|
+
// so leaving it unwrapped meant a recorded suite still hit a live model for
|
|
26339
|
+
// exactly the calls whose non-determinism it was recorded to remove. A local
|
|
26340
|
+
// baseline reproduced that: the same question blocked on one run and answered
|
|
26341
|
+
// on the next, and zero cassettes were written.
|
|
26342
|
+
return await provider.available() ? applyEvalCassette(provider) : null;
|
|
25404
26343
|
}
|
|
25405
26344
|
/** Convert a governed answer's result payload into a bounded synthesis preview. */
|
|
25406
26345
|
function agentResultToSynthesisPreview(result) {
|
|
@@ -28237,7 +29176,45 @@ async function buildAgentSchemaContextFromCatalog(projectRoot, question, prepare
|
|
|
28237
29176
|
const RUNTIME_SNAPSHOT_MAX_AGE_MS = 60 * 60 * 1000; // 1 hour
|
|
28238
29177
|
// A resolver compares at most 12 compact cards and never performs tool calls;
|
|
28239
29178
|
// ten seconds is the full allowance, not the start of another planning loop.
|
|
28240
|
-
|
|
29179
|
+
/**
|
|
29180
|
+
* Ceiling on the one bounded meaning-resolution call.
|
|
29181
|
+
*
|
|
29182
|
+
* 10s assumes a hosted model. A local Ollama model needs ~7s for a ONE-WORD
|
|
29183
|
+
* reply, so a 600-token resolution over a dozen candidates never lands: it
|
|
29184
|
+
* aborts, the router falls back to its evidence-only decision, and
|
|
29185
|
+
* `mayAssumeInterpretation` goes false — which sends every ambiguous question to
|
|
29186
|
+
* the clarification gate (AGT-017). The effect is that a local model cannot
|
|
29187
|
+
* answer anything ambiguous, in a product whose whole positioning is local-first.
|
|
29188
|
+
*
|
|
29189
|
+
* Scaled by the same `DQL_AGENT_DEADLINE_SCALE` as the run budget, so one
|
|
29190
|
+
* setting moves the provider's whole time envelope together rather than leaving
|
|
29191
|
+
* an inner bound to silently cap an outer one.
|
|
29192
|
+
*/
|
|
29193
|
+
/**
|
|
29194
|
+
* Predict how long the next provider call will take, for admission control.
|
|
29195
|
+
*
|
|
29196
|
+
* With fewer than three samples the MAX is the only honest predictor: there is
|
|
29197
|
+
* no distribution yet, and admitting a call the deadline then kills wastes the
|
|
29198
|
+
* whole remaining budget.
|
|
29199
|
+
*
|
|
29200
|
+
* With a real sample, p75 rather than the max. One slow response — a cold model
|
|
29201
|
+
* load, a retried connection — otherwise poisons admission control for the rest
|
|
29202
|
+
* of the run: every later call is refused against a worst case that already
|
|
29203
|
+
* passed. A recorded run tripped RUN_DEADLINE_INSUFFICIENT 6.4s into a 45s
|
|
29204
|
+
* budget for exactly that reason. p75 still errs slow, so a genuinely slow
|
|
29205
|
+
* provider is still respected.
|
|
29206
|
+
*/
|
|
29207
|
+
export function predictDispatchMs(observed, assumedMs = ASSUMED_PROVIDER_DISPATCH_MS) {
|
|
29208
|
+
if (observed.length === 0)
|
|
29209
|
+
return assumedMs;
|
|
29210
|
+
const sorted = [...observed].sort((left, right) => left - right);
|
|
29211
|
+
if (sorted.length < 3)
|
|
29212
|
+
return sorted[sorted.length - 1];
|
|
29213
|
+
const index = Math.max(0, Math.min(sorted.length - 1, Math.ceil(sorted.length * 0.75) - 1));
|
|
29214
|
+
return sorted[index];
|
|
29215
|
+
}
|
|
29216
|
+
const AGENT_MEANING_TIMEOUT_BASE_MS = 10_000;
|
|
29217
|
+
const AGENT_MEANING_TIMEOUT_MS = AGENT_MEANING_TIMEOUT_BASE_MS * deadlineScale();
|
|
28241
29218
|
export function boundedAgentMeaningSignal(signal, timeoutMs = AGENT_MEANING_TIMEOUT_MS) {
|
|
28242
29219
|
const timeout = AbortSignal.timeout(Math.max(1, timeoutMs));
|
|
28243
29220
|
return signal ? AbortSignal.any([signal, timeout]) : timeout;
|
|
@@ -28888,26 +29865,25 @@ function notebookResearchContextPreview(contextPack) {
|
|
|
28888
29865
|
};
|
|
28889
29866
|
}
|
|
28890
29867
|
function normalizeNotebookAgentResult(result) {
|
|
28891
|
-
const
|
|
28892
|
-
|
|
28893
|
-
|
|
28894
|
-
|
|
28895
|
-
|
|
28896
|
-
|
|
28897
|
-
|
|
28898
|
-
|
|
28899
|
-
|
|
28900
|
-
|
|
28901
|
-
const rows = Array.isArray(result.rows)
|
|
28902
|
-
? result.rows
|
|
28903
|
-
.filter((row) => Boolean(row && typeof row === 'object' && !Array.isArray(row)))
|
|
28904
|
-
.map((row) => row)
|
|
28905
|
-
: [];
|
|
29868
|
+
const canonical = normalizeCanonicalQueryResult({
|
|
29869
|
+
columns: result.columns,
|
|
29870
|
+
rows: result.rows,
|
|
29871
|
+
rowCount: result.rowCount,
|
|
29872
|
+
executionTime: result.executionTime,
|
|
29873
|
+
resultFingerprint: result.resultFingerprint,
|
|
29874
|
+
executionReceipt: result.executionReceipt,
|
|
29875
|
+
trustState: result.executableArtifact?.trustState,
|
|
29876
|
+
answerTier: result.answerTier,
|
|
29877
|
+
});
|
|
28906
29878
|
return {
|
|
28907
|
-
columns,
|
|
28908
|
-
rows,
|
|
28909
|
-
rowCount:
|
|
28910
|
-
|
|
29879
|
+
columns: canonical.columns,
|
|
29880
|
+
rows: canonical.rows,
|
|
29881
|
+
rowCount: canonical.rowCount,
|
|
29882
|
+
resultFingerprint: canonical.resultFingerprint,
|
|
29883
|
+
executionTime: canonical.executionTime ?? 0,
|
|
29884
|
+
...(canonical.truncated ? { truncated: true } : {}),
|
|
29885
|
+
...(canonical.executionReceipt ? { executionReceipt: canonical.executionReceipt } : {}),
|
|
29886
|
+
...(canonical.answerTier ? { answerTier: canonical.answerTier } : {}),
|
|
28911
29887
|
};
|
|
28912
29888
|
}
|
|
28913
29889
|
function notebookResearchSummary(question, result, error) {
|
|
@@ -29623,49 +30599,9 @@ function scoreAgentValueProbeColumn(table, column) {
|
|
|
29623
30599
|
return score;
|
|
29624
30600
|
}
|
|
29625
30601
|
export function isAgentValueProbeColumn(column) {
|
|
29626
|
-
|
|
29627
|
-
//
|
|
29628
|
-
|
|
29629
|
-
// can never be probed through automatic grounding.
|
|
29630
|
-
const normalizedName = column.name
|
|
29631
|
-
.replace(/([a-z0-9])([A-Z])/g, '$1 $2')
|
|
29632
|
-
.replace(/[_-]+/g, ' ')
|
|
29633
|
-
.toLowerCase();
|
|
29634
|
-
if (/\b(password|secret|token|credential|hash|salt|notes?|comments?|description|message|body|payload|content)\b/.test(normalizedName))
|
|
29635
|
-
return false;
|
|
29636
|
-
if (/\bemail\b/.test(normalizedName))
|
|
29637
|
-
return false;
|
|
29638
|
-
if (!hasAgentSchemaToken(name, [
|
|
29639
|
-
'account',
|
|
29640
|
-
'category',
|
|
29641
|
-
'channel',
|
|
29642
|
-
'city',
|
|
29643
|
-
'code',
|
|
29644
|
-
'country',
|
|
29645
|
-
'customer',
|
|
29646
|
-
'email',
|
|
29647
|
-
'full',
|
|
29648
|
-
'id',
|
|
29649
|
-
'key',
|
|
29650
|
-
'member',
|
|
29651
|
-
'name',
|
|
29652
|
-
'number',
|
|
29653
|
-
'product',
|
|
29654
|
-
'region',
|
|
29655
|
-
'segment',
|
|
29656
|
-
'sku',
|
|
29657
|
-
'state',
|
|
29658
|
-
'status',
|
|
29659
|
-
'subscriber',
|
|
29660
|
-
'type',
|
|
29661
|
-
'user',
|
|
29662
|
-
])) {
|
|
29663
|
-
return false;
|
|
29664
|
-
}
|
|
29665
|
-
const type = column.type?.toLowerCase() ?? '';
|
|
29666
|
-
if (!type)
|
|
29667
|
-
return true;
|
|
29668
|
-
return /\b(char|character|clob|email|string|text|uuid|varchar)\b/.test(type);
|
|
30602
|
+
// Delegates to the canonical predicate in dql-agent. Two copies of a security
|
|
30603
|
+
// rule drift, and the one that drifts is the one nobody is looking at.
|
|
30604
|
+
return isProbeSafeColumn({ name: column.name, ...(column.type ? { type: column.type } : {}) });
|
|
29669
30605
|
}
|
|
29670
30606
|
export function buildAgentValueProbeSql(table, column, searchTerms, connection) {
|
|
29671
30607
|
const relation = quoteAgentRelation(table.relation, connection);
|