@tangle-network/agent-eval 0.138.0 → 0.139.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/README.md +2 -1
  3. package/dist/analyst/index.d.ts +41 -94
  4. package/dist/analyst/index.d.ts.map +1 -1
  5. package/dist/analyst/index.js +9 -24
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
  8. package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
  9. package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
  10. package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
  11. package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
  12. package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +1 -1
  14. package/dist/benchmarks/index.js +1 -1
  15. package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
  16. package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
  17. package/dist/campaign/index.d.ts +5 -5
  18. package/dist/campaign/index.js +4 -3
  19. package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
  20. package/dist/campaign-DrS6_hLd.js.map +1 -0
  21. package/dist/canonical-D011XM8r.js +86 -0
  22. package/dist/canonical-D011XM8r.js.map +1 -0
  23. package/dist/cli.js +3 -3
  24. package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
  25. package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
  26. package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
  27. package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
  28. package/dist/contract/index.d.ts +10 -10
  29. package/dist/contract/index.js +8 -7
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/control.d.ts +2 -2
  32. package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
  33. package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
  34. package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
  35. package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
  36. package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
  37. package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
  38. package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
  39. package/dist/default-registry-BgJJItGr.js.map +1 -0
  40. package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
  41. package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
  42. package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
  43. package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
  44. package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
  45. package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
  46. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
  47. package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
  48. package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
  49. package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
  50. package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
  51. package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
  52. package/dist/fuzz.d.ts +1 -1
  53. package/dist/fuzz.js +1 -1
  54. package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
  55. package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
  56. package/dist/hosted/index.d.ts +3 -3
  57. package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
  58. package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
  59. package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
  60. package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
  61. package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
  62. package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
  63. package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
  64. package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
  65. package/dist/index.d.ts +35 -56
  66. package/dist/index.d.ts.map +1 -1
  67. package/dist/index.js +51 -176
  68. package/dist/index.js.map +1 -1
  69. package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
  70. package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
  71. package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
  72. package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
  73. package/dist/kind-factory-CFxA0JQX.js +2133 -0
  74. package/dist/kind-factory-CFxA0JQX.js.map +1 -0
  75. package/dist/ledger-core/index.js +2 -1
  76. package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
  77. package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
  78. package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
  79. package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
  80. package/dist/meta-eval/index.d.ts +2 -2
  81. package/dist/multishot/index.d.ts +2 -2
  82. package/dist/openapi.json +1 -1
  83. package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
  84. package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
  85. package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
  86. package/dist/replay-DjG4IG60.d.ts.map +1 -0
  87. package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
  88. package/dist/replay-SA4OB7O7.js.map +1 -0
  89. package/dist/reporting.d.ts +4 -4
  90. package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
  91. package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
  92. package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
  93. package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
  94. package/dist/rl.d.ts +5 -5
  95. package/dist/rl.js +1 -1
  96. package/dist/rollout/index.d.ts +1 -1
  97. package/dist/rollout/index.js +2 -2
  98. package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
  99. package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
  100. package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
  101. package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
  102. package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
  103. package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
  104. package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
  105. package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
  106. package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
  107. package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
  108. package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
  109. package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
  110. package/dist/single-run-lock-BTTtPZ9N.js +989 -0
  111. package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
  112. package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
  113. package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
  114. package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
  115. package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
  116. package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
  117. package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
  118. package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
  119. package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
  120. package/dist/store-otlp-DX4fGIcf.js +757 -0
  121. package/dist/store-otlp-DX4fGIcf.js.map +1 -0
  122. package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
  123. package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
  124. package/dist/tool-groups-CdYq22lX.d.ts +258 -0
  125. package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
  126. package/dist/traces.d.ts +7 -6
  127. package/dist/traces.js +5 -5
  128. package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
  129. package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
  130. package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
  131. package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
  132. package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
  133. package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
  134. package/dist/usage-receipt-CgxMEBZq.js +134 -0
  135. package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
  136. package/dist/wire/index.d.ts +3 -3
  137. package/dist/wire/index.js +1 -1
  138. package/docs/trace-analysis.md +170 -484
  139. package/package.json +1 -2
  140. package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
  141. package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
  142. package/dist/benchmark-command-CMqVqReF.js.map +0 -1
  143. package/dist/campaign-BIBS-NHV.js.map +0 -1
  144. package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
  145. package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
  146. package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
  147. package/dist/default-registry-PUhIVRWz.d.ts +0 -215
  148. package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
  149. package/dist/default-registry-lp5R0lve.js.map +0 -1
  150. package/dist/ledger-core-C0Yx1I14.js.map +0 -1
  151. package/dist/registry-C4yJTza7.d.ts +0 -178
  152. package/dist/registry-C4yJTza7.d.ts.map +0 -1
  153. package/dist/replay-Cb-4Vf0k.js.map +0 -1
  154. package/dist/replay-DbIYwso6.d.ts.map +0 -1
  155. package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
  156. package/dist/single-run-lock-D_bS5xhj.js +0 -318
  157. package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
  158. package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
  159. package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
  160. package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
  161. package/dist/store-otlp-BenKynPE.js +0 -1688
  162. package/dist/store-otlp-BenKynPE.js.map +0 -1
  163. package/dist/tools-DZGdROtG.js +0 -255
  164. package/dist/tools-DZGdROtG.js.map +0 -1
@@ -1,58 +1,13 @@
1
- import { i as CostLedger } from "./cost-ledger-CHDLA0Ss.js";
2
- import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, n as LlmClient, s as callLlm, u as costReceiptFromLlmError } from "./llm-client-Cj3c7PEm.js";
1
+ import { n as LlmClient } from "./llm-client-bkztEfIx.js";
3
2
  import { LLM_CONTEXT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, TOOL_NAME_ATTR_KEYS } from "./trace-attributes.js";
4
3
  import { t as executionTrackByLane } from "./execution-tracks-CpgFPpS5.js";
5
- import { D as spanEpochMillis } from "./store-otlp-BenKynPE.js";
6
- import { f as validateUsageSettlementTimeout, l as assertValidAnalystUsageReceipt, m as makeFinding, u as settleUsageReceiptFromCostLedger } from "./single-run-lock-D_bS5xhj.js";
7
- import { _ as canonicalString, v as hashCanonical } from "./ledger-core-C0Yx1I14.js";
8
- import { a as runTraceAnalysisLoop, r as buildTraceAnalystTools } from "./tools-DZGdROtG.js";
4
+ import { $ as spanEpochMillis, H as snapshotExactExecutionPlan, R as findingSubjectGrammarPromptFor, U as deepFreezeCanonicalJson, V as snapshotExactExecutionComponentIdentity, t as createTraceAnalyst } from "./kind-factory-CFxA0JQX.js";
5
+ import { i as validateUsageSettlementTimeout, o as makeFinding, t as assertValidAnalystUsageReceipt } from "./usage-receipt-CgxMEBZq.js";
6
+ import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
9
7
  import { t as analyzeSupervisorRunIntegrity } from "./integrity-CCXTftiL.js";
10
8
  import { o as combineAbortSignals } from "./proposal-findings-2GIUo1et.js";
11
- import { ai } from "@ax-llm/ax";
12
9
  import { z } from "zod";
13
10
  import { randomUUID } from "node:crypto";
14
- //#region src/analyst/ax-service.ts
15
- const configuredModels = /* @__PURE__ */ new WeakMap();
16
- /**
17
- * Construct the `AxAIService` an analyst kind calls through
18
- * (`createTraceAnalystKind({ ai })`).
19
- *
20
- * Ax's `ai()` pins `config.model` to the OpenAI catalog enum, but every
21
- * OpenAI-compatible router an analyst points at (router.tangle.tools,
22
- * cli-bridge) accepts arbitrary model ids (claude-code/sonnet, openai/gpt-5.4,
23
- * …). Consumers were each re-rolling `ai({ name, apiKey, apiURL, config })`
24
- * behind an `as (a: any) => any` cast to dodge the enum; this is the one
25
- * canonical constructor so they don't have to — and don't take a direct
26
- * `@ax-llm/ax` dependency for it.
27
- */
28
- function createAnalystAi(config) {
29
- const model = config.model.trim();
30
- if (!model) throw new TypeError("createAnalystAi: model must be a non-empty string");
31
- const service = ai({
32
- name: config.provider ?? "openai",
33
- apiKey: config.apiKey,
34
- ...config.baseUrl ? { apiURL: config.baseUrl } : {},
35
- ...config.headers ? { headers: config.headers } : {},
36
- config: { model }
37
- });
38
- configuredModels.set(service, model);
39
- return service;
40
- }
41
- function getConfiguredAnalystModel(service) {
42
- return configuredModels.get(service);
43
- }
44
- /** Resolve the model before paid work so every request can be bounded and attributed. */
45
- function resolveAnalystModel(service, override) {
46
- if (override !== void 0) {
47
- const model = override.trim();
48
- if (!model) throw new TypeError("createTraceAnalystKind: model must be a non-empty string");
49
- return model;
50
- }
51
- const model = getConfiguredAnalystModel(service)?.trim();
52
- if (!model) throw new TypeError("createTraceAnalystKind: model is required for Ax services not created by createAnalystAi()");
53
- return model;
54
- }
55
- //#endregion
56
11
  //#region src/analyst/chat-client.ts
57
12
  /**
58
13
  * Provider-neutral chat contract for every model call made by agent-eval.
@@ -608,1362 +563,6 @@ function positiveInteger(value, name) {
608
563
  return value;
609
564
  }
610
565
  //#endregion
611
- //#region src/analyst/ax-cost-service.ts
612
- /**
613
- * Meter every chat call an Ax program makes through the shared paid-call ledger.
614
- * The wrapper disables provider streaming because a stream has no complete usage
615
- * receipt until it is consumed, while Ax's analyst output is not streamed to users.
616
- */
617
- function meterAxChatService(ai, options) {
618
- assertPositiveInteger(options.maxOutputTokens, "maxOutputTokens");
619
- const source = ai;
620
- if (typeof source.chat !== "function") throw new TypeError("meterAxChatService: Ax service must implement chat()");
621
- const providerChat = source.chat.bind(ai);
622
- const chat = async (request, callOptions = {}) => {
623
- const boundedRequest = boundOutputTokens(request, options.maxOutputTokens);
624
- const model = modelName(boundedRequest.model) || options.defaultModel || "";
625
- const canTurnOffThinking = canDisableThinking(ai, model);
626
- const combined = combineSignals(options.signal, callOptions.abortSignal);
627
- try {
628
- const paid = await options.ledger.runPaidCall({
629
- channel: "analyst",
630
- phase: options.phase ?? "analyst.ax.chat",
631
- actor: options.actor,
632
- model,
633
- tags: options.tags,
634
- signal: combined.signal,
635
- maximumCharge: maximumChargeForAxChatRequest(boundedRequest, model),
636
- execute: async (executionSignal) => {
637
- const providerOptions = {
638
- ...callOptions,
639
- abortSignal: executionSignal,
640
- retry: {
641
- ...callOptions.retry,
642
- maxRetries: 0
643
- },
644
- stream: false,
645
- showThoughts: false
646
- };
647
- if (canTurnOffThinking) providerOptions.thinkingTokenBudget = "none";
648
- else Reflect.deleteProperty(providerOptions, "thinkingTokenBudget");
649
- const response = await providerChat(boundedRequest, providerOptions);
650
- if (response instanceof ReadableStream) throw new Error("meterAxChatService: provider returned a stream after stream:false");
651
- return response;
652
- },
653
- receipt: (response) => costReceiptFromAxResponse(response, model)
654
- });
655
- if (!paid.succeeded) throw paid.error;
656
- return paid.value;
657
- } finally {
658
- combined.dispose();
659
- }
660
- };
661
- return new Proxy(ai, { get(target, property) {
662
- if (property === "chat") return chat;
663
- const value = Reflect.get(target, property, target);
664
- return typeof value === "function" ? value.bind(target) : value;
665
- } });
666
- }
667
- /** Conservative priced bound for one Ax text chat request. */
668
- function maximumChargeForAxChatRequest(request, defaultModel) {
669
- const model = modelName(request.model) || defaultModel || "";
670
- const maxTokens = request.modelConfig?.maxTokens;
671
- const completions = request.modelConfig?.n;
672
- if (!model || maxTokens === void 0 || completions === void 0) return void 0;
673
- assertPositiveInteger(maxTokens, "request.modelConfig.maxTokens");
674
- assertPositiveInteger(completions, "request.modelConfig.n");
675
- const maximumOutputTokens = maxTokens * completions;
676
- assertPositiveInteger(maximumOutputTokens, "maximum output tokens");
677
- if (containsUnboundedOrCacheableContent(request)) return void 0;
678
- let inputTokens;
679
- try {
680
- const pricedRequest = request.model === void 0 ? {
681
- ...request,
682
- model
683
- } : request;
684
- inputTokens = new TextEncoder().encode(JSON.stringify(pricedRequest)).byteLength * completions;
685
- } catch {
686
- return;
687
- }
688
- assertPositiveInteger(inputTokens, "maximum input tokens");
689
- return {
690
- model,
691
- inputTokens,
692
- outputTokens: maximumOutputTokens
693
- };
694
- }
695
- function boundOutputTokens(request, limit) {
696
- const requested = request.modelConfig?.maxTokens;
697
- if (requested !== void 0) assertPositiveInteger(requested, "request.modelConfig.maxTokens");
698
- const completions = request.modelConfig?.n ?? 1;
699
- assertPositiveInteger(completions, "request.modelConfig.n");
700
- const maxTokens = requested === void 0 ? limit : Math.min(requested, limit);
701
- return {
702
- ...request,
703
- ...request.functionCall === void 0 && !request.functions?.length ? { functionCall: "none" } : {},
704
- modelConfig: {
705
- ...request.modelConfig,
706
- maxTokens,
707
- n: completions
708
- }
709
- };
710
- }
711
- function costReceiptFromAxResponse(response, fallbackModel) {
712
- const usage = response.modelUsage;
713
- const tokens = usage?.tokens;
714
- const model = usage?.model || fallbackModel;
715
- if (!tokens || !validUsage(tokens.promptTokens) || !validUsage(tokens.completionTokens) || !validUsage(tokens.totalTokens) || tokens.totalTokens < tokens.promptTokens + tokens.completionTokens || !validOptionalUsage(tokens.cacheReadTokens) || !validOptionalUsage(tokens.cacheCreationTokens) || !validOptionalUsage(tokens.reasoningTokens) || !validOptionalUsage(tokens.thoughtsTokens)) return {
716
- model,
717
- inputTokens: 0,
718
- outputTokens: 0,
719
- usageUnknown: true
720
- };
721
- const cacheReadTokens = validUsage(tokens.cacheReadTokens) ? tokens.cacheReadTokens : 0;
722
- const cacheCreationTokens = validUsage(tokens.cacheCreationTokens) ? tokens.cacheCreationTokens : 0;
723
- const totalCacheTokens = cacheReadTokens + cacheCreationTokens;
724
- const thoughtsTokens = validUsage(tokens.thoughtsTokens) ? tokens.thoughtsTokens : 0;
725
- const reasoningTokens = Math.max(validUsage(tokens.reasoningTokens) ? tokens.reasoningTokens : 0, thoughtsTokens);
726
- const separateReasoningTokens = reasoningTokens > tokens.completionTokens ? reasoningTokens : 0;
727
- const additionalOutputTokens = Math.max(thoughtsTokens, separateReasoningTokens);
728
- const outputTokens = tokens.completionTokens + additionalOutputTokens;
729
- const directTotal = tokens.promptTokens + tokens.completionTokens;
730
- if (!(/* @__PURE__ */ new Set([
731
- directTotal,
732
- directTotal + totalCacheTokens,
733
- directTotal + additionalOutputTokens,
734
- directTotal + totalCacheTokens + additionalOutputTokens
735
- ])).has(tokens.totalTokens)) return {
736
- model,
737
- inputTokens: 0,
738
- outputTokens: 0,
739
- usageUnknown: true
740
- };
741
- return {
742
- model,
743
- inputTokens: tokens.promptTokens,
744
- outputTokens,
745
- ...reasoningTokens > 0 ? { reasoningTokens } : {},
746
- ...cacheReadTokens > 0 ? { cachedTokens: cacheReadTokens } : {},
747
- ...cacheCreationTokens > 0 ? { cacheWriteTokens: cacheCreationTokens } : {}
748
- };
749
- }
750
- function combineSignals(first, second) {
751
- if (!first) return {
752
- signal: second,
753
- dispose: () => {}
754
- };
755
- if (!second || first === second) return {
756
- signal: first,
757
- dispose: () => {}
758
- };
759
- if (typeof AbortSignal.any === "function") return {
760
- signal: AbortSignal.any([first, second]),
761
- dispose: () => {}
762
- };
763
- const controller = new AbortController();
764
- const dispose = () => {
765
- first.removeEventListener("abort", abortFromFirst);
766
- second.removeEventListener("abort", abortFromSecond);
767
- };
768
- const abortFrom = (source) => {
769
- if (!controller.signal.aborted) controller.abort(source.reason);
770
- dispose();
771
- };
772
- const abortFromFirst = () => abortFrom(first);
773
- const abortFromSecond = () => abortFrom(second);
774
- if (first.aborted) abortFrom(first);
775
- else if (second.aborted) abortFrom(second);
776
- else {
777
- first.addEventListener("abort", abortFromFirst, { once: true });
778
- second.addEventListener("abort", abortFromSecond, { once: true });
779
- }
780
- return {
781
- signal: controller.signal,
782
- dispose
783
- };
784
- }
785
- function containsUnboundedOrCacheableContent(request) {
786
- if (request.functions?.some((fn) => fn.cache === true)) return true;
787
- return request.chatPrompt.some((message) => {
788
- if (message.cache === true) return true;
789
- if (!("content" in message) || !Array.isArray(message.content)) return false;
790
- return message.content.some((part) => part.cache === true || part.type !== "text");
791
- });
792
- }
793
- function modelName(value) {
794
- return typeof value === "string" ? value : "";
795
- }
796
- function canDisableThinking(ai, model) {
797
- const namedAi = ai;
798
- const serviceName = typeof namedAi.getName === "function" ? namedAi.getName() : "";
799
- const modelId = model.slice(model.lastIndexOf("/") + 1);
800
- return serviceName !== "GoogleGeminiAI" || !/^gemini-3(?:[.-]|$)/i.test(modelId);
801
- }
802
- function validUsage(value) {
803
- return typeof value === "number" && Number.isSafeInteger(value) && value >= 0;
804
- }
805
- function validOptionalUsage(value) {
806
- return value === void 0 || validUsage(value);
807
- }
808
- function assertPositiveInteger(value, field) {
809
- if (!Number.isSafeInteger(value) || value <= 0) throw new RangeError(`meterAxChatService: ${field} must be a positive integer`);
810
- }
811
- //#endregion
812
- //#region src/ledger-core/deep-freeze.ts
813
- /** Freeze a detached canonical-JSON graph. Canonicalization has already ruled out cycles.
814
- *
815
- * Lives outside canonical.ts so the analyst-benchmark implementation digest,
816
- * which covers canonical.ts, stays bound to the published benchmark evidence. */
817
- function deepFreezeCanonicalJson(value) {
818
- if (value && typeof value === "object" && !Object.isFrozen(value)) {
819
- Object.freeze(value);
820
- for (const nested of Object.values(value)) deepFreezeCanonicalJson(nested);
821
- }
822
- return value;
823
- }
824
- //#endregion
825
- //#region src/analyst/exact-types.ts
826
- /** Canonical identity for any live component admitted to an exact run. */
827
- function snapshotExactExecutionComponentIdentity(value, context) {
828
- let detached;
829
- try {
830
- detached = JSON.parse(canonicalString(value));
831
- } catch (cause) {
832
- throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
833
- }
834
- const parsed = componentIdentitySchema.safeParse(detached);
835
- if (!parsed.success) throw new TypeError(`${context} requires non-empty id/version and object config`);
836
- return deepFreezeCanonicalJson({
837
- id: parsed.data.id,
838
- version: parsed.data.version,
839
- config_digest: hashCanonical(parsed.data.config)
840
- });
841
- }
842
- const nonEmptyString = z.string().min(1);
843
- const digest = z.string().regex(/^sha256:[a-f0-9]{64}$/);
844
- const finiteNonnegative$1 = z.number().finite().nonnegative();
845
- const nonnegativeSafeInteger$1 = z.number().int().min(0).max(Number.MAX_SAFE_INTEGER);
846
- const positiveTimeout = z.number().int().positive().max(2147483647);
847
- const componentSnapshotSchema = z.strictObject({
848
- id: nonEmptyString,
849
- version: nonEmptyString,
850
- config_digest: digest
851
- });
852
- const componentIdentitySchema = z.strictObject({
853
- id: nonEmptyString,
854
- version: nonEmptyString,
855
- config: z.record(z.string(), z.unknown())
856
- });
857
- const deterministicCostSchema = z.strictObject({
858
- kind: z.literal("deterministic"),
859
- est_usd_per_run: finiteNonnegative$1.optional(),
860
- models: z.array(nonEmptyString).optional()
861
- });
862
- const llmCostSchema = z.strictObject({
863
- kind: z.literal("llm"),
864
- est_usd_per_run: finiteNonnegative$1.optional(),
865
- models: z.array(nonEmptyString).optional(),
866
- settlement_timeout_ms: nonnegativeSafeInteger$1.optional()
867
- });
868
- const requirementsSchema = z.strictObject({
869
- min_shots: nonnegativeSafeInteger$1.optional(),
870
- capabilities: z.array(nonEmptyString).optional()
871
- }).nullable();
872
- const analystSnapshotSchema = z.strictObject({
873
- id: nonEmptyString,
874
- version: nonEmptyString,
875
- input_kind: z.enum([
876
- "trace-store",
877
- "artifact-dir",
878
- "run-record",
879
- "judge-input",
880
- "custom"
881
- ]),
882
- cost: z.discriminatedUnion("kind", [deterministicCostSchema, llmCostSchema]),
883
- requirements: requirementsSchema,
884
- execution_config_digest: digest
885
- });
886
- const allocationsSchema = z.record(nonEmptyString, z.union([finiteNonnegative$1, z.null()]));
887
- const weightsSchema = z.record(nonEmptyString, finiteNonnegative$1);
888
- const budgetSnapshotSchema = z.discriminatedUnion("kind", [
889
- z.strictObject({ kind: z.literal("none") }),
890
- z.strictObject({
891
- kind: z.literal("equal"),
892
- total_usd: finiteNonnegative$1,
893
- allocations_usd: allocationsSchema
894
- }),
895
- z.strictObject({
896
- kind: z.literal("weighted"),
897
- total_usd: finiteNonnegative$1,
898
- weights: weightsSchema,
899
- allocations_usd: allocationsSchema
900
- })
901
- ]);
902
- const priorFindingsSchema = z.discriminatedUnion("kind", [
903
- z.strictObject({ kind: z.literal("none") }),
904
- z.strictObject({
905
- kind: z.literal("ordered"),
906
- count: nonnegativeSafeInteger$1,
907
- digest
908
- }),
909
- z.strictObject({
910
- kind: z.literal("by_analyst"),
911
- keys: z.array(nonEmptyString),
912
- count: nonnegativeSafeInteger$1,
913
- digest
914
- })
915
- ]);
916
- const exactRunPolicySchema = z.strictObject({
917
- budget: budgetSnapshotSchema,
918
- total_timeout_ms: positiveTimeout.nullable(),
919
- signal_provided: z.boolean(),
920
- cost_ledger: componentSnapshotSchema.nullable(),
921
- cost_phase: nonEmptyString.nullable(),
922
- tags: z.record(z.string(), z.string()).nullable(),
923
- prior_findings: priorFindingsSchema,
924
- chain_findings: z.boolean(),
925
- missing_input_mode: z.enum(["skip", "abort"]),
926
- registry_hooks: componentSnapshotSchema.nullable(),
927
- registry_chat: componentSnapshotSchema.nullable()
928
- });
929
- const exactExecutionPlanSchema = z.strictObject({
930
- schema_version: z.literal("1.0.0"),
931
- analysts: z.array(analystSnapshotSchema).min(1),
932
- policy: exactRunPolicySchema,
933
- digest
934
- }).superRefine((plan, context) => {
935
- const issue = (path, message) => context.addIssue({
936
- code: "custom",
937
- path,
938
- message
939
- });
940
- const analystIds = plan.analysts.map((analyst) => analyst.id);
941
- if (new Set(analystIds).size !== analystIds.length) issue(["analysts"], "analyst ids must be unique");
942
- if (plan.policy.cost_ledger === null && plan.policy.cost_phase !== null) issue(["policy", "cost_phase"], "cost phase requires a cost ledger");
943
- if (plan.policy.prior_findings.kind === "by_analyst" && plan.policy.prior_findings.keys.some((key, index, keys) => index > 0 && key <= keys[index - 1])) issue([
944
- "policy",
945
- "prior_findings",
946
- "keys"
947
- ], "keys must be sorted and unique");
948
- const budget = plan.policy.budget;
949
- if (budget.kind === "none") return;
950
- const allocationIds = Object.keys(budget.allocations_usd).sort();
951
- const selectedIds = [...analystIds].sort();
952
- if (allocationIds.length !== selectedIds.length || allocationIds.some((id, index) => id !== selectedIds[index])) {
953
- issue([
954
- "policy",
955
- "budget",
956
- "allocations_usd"
957
- ], "allocations must name every analyst and no others");
958
- return;
959
- }
960
- const runnableIds = analystIds.filter((id) => budget.allocations_usd[id] !== null);
961
- const epsilon = Math.max(1, budget.total_usd) * Number.EPSILON * 8;
962
- if (runnableIds.length === 0) return;
963
- if (budget.kind === "weighted") {
964
- const weightIds = Object.keys(budget.weights).sort();
965
- if (weightIds.length !== selectedIds.length || weightIds.some((id, index) => id !== selectedIds[index])) {
966
- issue([
967
- "policy",
968
- "budget",
969
- "weights"
970
- ], "weights must name every analyst and no others");
971
- return;
972
- }
973
- const totalWeight = runnableIds.reduce((sum, id) => sum + (budget.weights[id] ?? 0), 0);
974
- if (totalWeight === 0) {
975
- issue([
976
- "policy",
977
- "budget",
978
- "weights"
979
- ], "runnable analysts must have positive total weight");
980
- return;
981
- }
982
- for (const id of runnableIds) {
983
- const expected = budget.total_usd * (budget.weights[id] ?? 0) / totalWeight;
984
- if (Math.abs((budget.allocations_usd[id] ?? 0) - expected) > epsilon) issue([
985
- "policy",
986
- "budget",
987
- "allocations_usd",
988
- id
989
- ], "allocation does not match the weighted policy");
990
- }
991
- return;
992
- }
993
- const expected = budget.total_usd / runnableIds.length;
994
- for (const id of runnableIds) if (Math.abs((budget.allocations_usd[id] ?? 0) - expected) > epsilon) issue([
995
- "policy",
996
- "budget",
997
- "allocations_usd",
998
- id
999
- ], "allocation does not match the equal policy");
1000
- });
1001
- /**
1002
- * Canonicalize and validate the one exact-plan representation shared by execution and archival.
1003
- * Unknown fields fail at every level; the returned graph is detached and deeply frozen.
1004
- */
1005
- function snapshotExactExecutionPlan(value, context = "exact analyst execution plan") {
1006
- let detached;
1007
- try {
1008
- detached = JSON.parse(canonicalString(value));
1009
- } catch (cause) {
1010
- throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
1011
- }
1012
- const parsed = exactExecutionPlanSchema.safeParse(detached);
1013
- if (!parsed.success) {
1014
- const issue = parsed.error.issues[0];
1015
- const path = issue?.path.length ? ` ${issue.path.join(".")}` : "";
1016
- throw new TypeError(`${context}${path}: ${issue?.message ?? "is invalid"}`);
1017
- }
1018
- const expectedDigest = hashCanonical({
1019
- schema_version: parsed.data.schema_version,
1020
- analysts: parsed.data.analysts,
1021
- policy: parsed.data.policy
1022
- });
1023
- if (parsed.data.digest !== expectedDigest) throw new TypeError(`${context} digest does not match its content`);
1024
- return deepFreezeCanonicalJson(parsed.data);
1025
- }
1026
- //#endregion
1027
- //#region src/analyst/finding-subject.ts
1028
- /**
1029
- * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
1030
- *
1031
- * Background: kind actor prompts have always documented a subject grammar
1032
- * (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
1033
- * LLM was unconstrained — it could emit `subject: "fix the prompt"`
1034
- * (prose) and downstream adapters routed on `startsWith(...)` would
1035
- * silently skip it. Every per-vertical `ImprovementAdapter` had a
1036
- * routing table that mostly caught nothing.
1037
- *
1038
- * This module fixes that:
1039
- * - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
1040
- * when `raw` matches the grammar, else `null`. Used at the
1041
- * `RawAnalystFindingSchema` boundary so malformed subjects are
1042
- * rejected loudly instead of silently lifted into the registry.
1043
- * - `FindingSubjectKind` — the union of valid locus categories. Each
1044
- * variant carries the typed components downstream adapters resolve
1045
- * against the agent's surface manifest (no string parsing in the
1046
- * adapter).
1047
- * - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
1048
- * grammar string embedded in kind actor prompts. Drift between
1049
- * prompt and parser is impossible if every kind imports this.
1050
- *
1051
- * The grammar is intentionally NARROW — only loci the substrate's
1052
- * default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
1053
- * finding with a subject outside this set fails the parser; the kind
1054
- * author either extends the grammar here (and adds adapter routing)
1055
- * or rephrases the prompt to map onto an existing variant.
1056
- *
1057
- * `failure-mode` is the one exception — its subjects are free-form
1058
- * cluster labels, not loci. The schema preserves them as
1059
- * `{ kind: 'cluster', label }` and the adapters skip them (cluster
1060
- * findings are evidence, not actionable mutations).
1061
- */
1062
- const FINDING_SUBJECT_KINDS = [
1063
- "knowledge.wiki",
1064
- "knowledge.claim",
1065
- "knowledge.raw",
1066
- "knowledge.stale",
1067
- "system-prompt",
1068
- "skill",
1069
- "tool-doc",
1070
- "new-tool",
1071
- "mcp",
1072
- "hook",
1073
- "subagent",
1074
- "workflow",
1075
- "rollout-policy",
1076
- "agent-profile",
1077
- "code",
1078
- "rag",
1079
- "memory",
1080
- "scaffolding",
1081
- "output-schema",
1082
- "websearch.outdated",
1083
- "prior-run-summary",
1084
- "cluster"
1085
- ];
1086
- /**
1087
- * Parse a raw subject string emitted by an analyst kind's actor.
1088
- *
1089
- * Returns the typed `FindingSubject` when `raw` matches the grammar,
1090
- * else `null`. Callers use the `null` return as a signal to either
1091
- * (a) reject the finding at parse time (kinds that emit typed loci —
1092
- * knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
1093
- * a cluster label (failure-mode).
1094
- *
1095
- * Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
1096
- * paths sane downstream. Topics / keys / sections allow any non-empty
1097
- * string (free-form for the LLM's voice) but get trimmed.
1098
- *
1099
- * Empty / whitespace-only inputs return `null`. `undefined` returns
1100
- * `null`. Both are surfaced by the caller as a rejected subject.
1101
- */
1102
- function parseFindingSubject(raw) {
1103
- if (raw === null || raw === void 0) return null;
1104
- const trimmed = raw.trim();
1105
- if (trimmed.length === 0) return null;
1106
- const wiki = trimmed.match(/^agent-knowledge:wiki:([a-z0-9][a-z0-9-]*)(?:#([a-z0-9][a-z0-9-]*))?$/);
1107
- if (wiki) return {
1108
- kind: "knowledge.wiki",
1109
- slug: wiki[1],
1110
- ...wiki[2] ? { heading: wiki[2] } : {}
1111
- };
1112
- const claim = trimmed.match(/^agent-knowledge:claim:(.+)$/);
1113
- if (claim && claim[1].trim().length > 0) return {
1114
- kind: "knowledge.claim",
1115
- topic: claim[1].trim()
1116
- };
1117
- const raw_ = trimmed.match(/^agent-knowledge:raw:(.+)$/);
1118
- if (raw_ && raw_[1].trim().length > 0) return {
1119
- kind: "knowledge.raw",
1120
- sourceId: raw_[1].trim()
1121
- };
1122
- const stale = trimmed.match(/^agent-knowledge:stale:([a-z0-9][a-z0-9-]*)$/);
1123
- if (stale) return {
1124
- kind: "knowledge.stale",
1125
- slug: stale[1]
1126
- };
1127
- const sp = trimmed.match(/^system-prompt:(.+)$/);
1128
- if (sp && sp[1].trim().length > 0) return {
1129
- kind: "system-prompt",
1130
- section: sp[1].trim()
1131
- };
1132
- const skill = trimmed.match(/^skill:([a-z0-9][a-z0-9_.-]*)$/);
1133
- if (skill) return {
1134
- kind: "skill",
1135
- name: skill[1]
1136
- };
1137
- const tdAspect = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*):(.+)$/);
1138
- if (tdAspect && tdAspect[2].trim().length > 0) return {
1139
- kind: "tool-doc",
1140
- tool: tdAspect[1],
1141
- aspect: tdAspect[2].trim()
1142
- };
1143
- const td = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*)$/);
1144
- if (td) return {
1145
- kind: "tool-doc",
1146
- tool: td[1]
1147
- };
1148
- const nt = trimmed.match(/^new-tool:([a-z0-9][a-z0-9_-]*)$/);
1149
- if (nt) return {
1150
- kind: "new-tool",
1151
- name: nt[1]
1152
- };
1153
- const mcp = trimmed.match(/^mcp:([a-z0-9][a-z0-9_.-]*)(?::([a-z0-9][a-z0-9_.-]*))?$/);
1154
- if (mcp) return {
1155
- kind: "mcp",
1156
- server: mcp[1],
1157
- ...mcp[2] ? { tool: mcp[2] } : {}
1158
- };
1159
- const hook = trimmed.match(/^hook:([a-z0-9][a-z0-9_.-]*)$/);
1160
- if (hook) return {
1161
- kind: "hook",
1162
- name: hook[1]
1163
- };
1164
- const subagent = trimmed.match(/^subagent:([a-z0-9][a-z0-9_.-]*)$/);
1165
- if (subagent) return {
1166
- kind: "subagent",
1167
- name: subagent[1]
1168
- };
1169
- const workflow = trimmed.match(/^workflow:([a-z0-9][a-z0-9_.-]*)$/);
1170
- if (workflow) return {
1171
- kind: "workflow",
1172
- name: workflow[1]
1173
- };
1174
- const rolloutPolicy = trimmed.match(/^rollout-policy:(.+)$/);
1175
- if (rolloutPolicy && rolloutPolicy[1].trim().length > 0) return {
1176
- kind: "rollout-policy",
1177
- field: rolloutPolicy[1].trim()
1178
- };
1179
- const agentProfile = trimmed.match(/^agent-profile:(.+)$/);
1180
- if (agentProfile && agentProfile[1].trim().length > 0) return {
1181
- kind: "agent-profile",
1182
- field: agentProfile[1].trim()
1183
- };
1184
- const code = trimmed.match(/^code:(.+)$/);
1185
- if (code && code[1].trim().length > 0) return {
1186
- kind: "code",
1187
- path: code[1].trim()
1188
- };
1189
- const rag = trimmed.match(/^rag:([a-z0-9][a-z0-9_-]*):(.+)$/);
1190
- if (rag && rag[2].trim().length > 0) return {
1191
- kind: "rag",
1192
- corpus: rag[1],
1193
- docId: rag[2].trim()
1194
- };
1195
- const mem = trimmed.match(/^memory:(.+)$/);
1196
- if (mem && mem[1].trim().length > 0) return {
1197
- kind: "memory",
1198
- key: mem[1].trim()
1199
- };
1200
- const sc = trimmed.match(/^scaffolding:(.+)$/);
1201
- if (sc && sc[1].trim().length > 0) return {
1202
- kind: "scaffolding",
1203
- concern: sc[1].trim()
1204
- };
1205
- const os = trimmed.match(/^output-schema:(.+)$/);
1206
- if (os && os[1].trim().length > 0) return {
1207
- kind: "output-schema",
1208
- field: os[1].trim()
1209
- };
1210
- const ws = trimmed.match(/^websearch:outdated:(.+)$/);
1211
- if (ws && ws[1].trim().length > 0) return {
1212
- kind: "websearch.outdated",
1213
- topic: ws[1].trim()
1214
- };
1215
- const prs = trimmed.match(/^prior-run-summary:(.+)$/);
1216
- if (prs && prs[1].trim().length > 0) return {
1217
- kind: "prior-run-summary",
1218
- topic: prs[1].trim()
1219
- };
1220
- if (/^[a-z0-9][a-z0-9._-]*$/.test(trimmed) && trimmed.length <= 80) return {
1221
- kind: "cluster",
1222
- label: trimmed
1223
- };
1224
- return null;
1225
- }
1226
- /**
1227
- * Render the parsed subject back to its canonical string form. Inverse
1228
- * of `parseFindingSubject`; useful when the substrate constructs new
1229
- * findings programmatically (e.g. for tests, replays, or
1230
- * `id_basis` carry-forward).
1231
- */
1232
- function renderFindingSubject(s) {
1233
- switch (s.kind) {
1234
- case "knowledge.wiki": return s.heading ? `agent-knowledge:wiki:${s.slug}#${s.heading}` : `agent-knowledge:wiki:${s.slug}`;
1235
- case "knowledge.claim": return `agent-knowledge:claim:${s.topic}`;
1236
- case "knowledge.raw": return `agent-knowledge:raw:${s.sourceId}`;
1237
- case "knowledge.stale": return `agent-knowledge:stale:${s.slug}`;
1238
- case "system-prompt": return `system-prompt:${s.section}`;
1239
- case "skill": return `skill:${s.name}`;
1240
- case "tool-doc": return s.aspect ? `tool-doc:${s.tool}:${s.aspect}` : `tool-doc:${s.tool}`;
1241
- case "new-tool": return `new-tool:${s.name}`;
1242
- case "mcp": return s.tool ? `mcp:${s.server}:${s.tool}` : `mcp:${s.server}`;
1243
- case "hook": return `hook:${s.name}`;
1244
- case "subagent": return `subagent:${s.name}`;
1245
- case "workflow": return `workflow:${s.name}`;
1246
- case "rollout-policy": return `rollout-policy:${s.field}`;
1247
- case "agent-profile": return `agent-profile:${s.field}`;
1248
- case "code": return `code:${s.path}`;
1249
- case "rag": return `rag:${s.corpus}:${s.docId}`;
1250
- case "memory": return `memory:${s.key}`;
1251
- case "scaffolding": return `scaffolding:${s.concern}`;
1252
- case "output-schema": return `output-schema:${s.field}`;
1253
- case "websearch.outdated": return `websearch:outdated:${s.topic}`;
1254
- case "prior-run-summary": return `prior-run-summary:${s.topic}`;
1255
- case "cluster": return s.label;
1256
- }
1257
- }
1258
- /**
1259
- * The grammar text embedded into kind actor prompts. Kinds opt into
1260
- * the subset of variants they emit (e.g. `improvement` excludes the
1261
- * cluster variant; `failure-mode` includes ONLY the cluster variant).
1262
- *
1263
- * Drift between prompt and parser is impossible: every kind imports
1264
- * this constant + the matching `expects` set, and the unit tests below
1265
- * lock the table to the parser.
1266
- */
1267
- const FINDING_SUBJECT_SYNTAX = {
1268
- "knowledge.wiki": "agent-knowledge:wiki:<slug>[#<heading>]",
1269
- "knowledge.claim": "agent-knowledge:claim:<topic>",
1270
- "knowledge.raw": "agent-knowledge:raw:<source-id>",
1271
- "knowledge.stale": "agent-knowledge:stale:<slug>",
1272
- "system-prompt": "system-prompt:<section>",
1273
- skill: "skill:<name>",
1274
- "tool-doc": "tool-doc:<tool>[:<aspect>]",
1275
- "new-tool": "new-tool:<name>",
1276
- mcp: "mcp:<server>[:<tool>]",
1277
- hook: "hook:<name>",
1278
- subagent: "subagent:<name>",
1279
- workflow: "workflow:<name>",
1280
- "rollout-policy": "rollout-policy:<field>",
1281
- "agent-profile": "agent-profile:<field>",
1282
- code: "code:<path>",
1283
- rag: "rag:<corpus>:<doc-id>",
1284
- memory: "memory:<key>",
1285
- scaffolding: "scaffolding:<concern>",
1286
- "output-schema": "output-schema:<field>",
1287
- "websearch.outdated": "websearch:outdated:<topic>",
1288
- "prior-run-summary": "prior-run-summary:<topic>",
1289
- cluster: "<lowercase-cluster-label>"
1290
- };
1291
- const FINDING_SUBJECT_PURPOSE = {
1292
- "knowledge.wiki": "create or update a wiki page",
1293
- "knowledge.claim": "draft a claim or relation",
1294
- "knowledge.raw": "curate a raw source",
1295
- "knowledge.stale": "mark a stale page",
1296
- "system-prompt": "revise a system-prompt section",
1297
- skill: "create or revise a skill",
1298
- "tool-doc": "revise a tool contract",
1299
- "new-tool": "propose a new tool",
1300
- mcp: "revise an MCP server or tool",
1301
- hook: "revise a lifecycle hook",
1302
- subagent: "revise a delegated agent",
1303
- workflow: "revise an orchestration workflow",
1304
- "rollout-policy": "revise budget, sampling, or stop policy",
1305
- "agent-profile": "revise another AgentProfile field",
1306
- code: "revise an implementation path",
1307
- rag: "ingest or correct a RAG document",
1308
- memory: "invalidate or set memory",
1309
- scaffolding: "revise preconditions, retries, or verification",
1310
- "output-schema": "constrain the output shape",
1311
- "websearch.outdated": "identify a stale web result",
1312
- "prior-run-summary": "identify a stale prior-run summary",
1313
- cluster: "name one failure cluster"
1314
- };
1315
- function renderFindingSubjectGrammar(kinds) {
1316
- return [
1317
- "Subjects MUST match one of these forms — anything else is rejected at parse time:",
1318
- ...kinds.map((kind) => ` ${FINDING_SUBJECT_SYNTAX[kind]} — ${FINDING_SUBJECT_PURPOSE[kind]}`),
1319
- "Runtime ids are lowercase [a-z0-9_.-]+. Topics, keys, paths, and sections are free-form and trimmed."
1320
- ].join("\n");
1321
- }
1322
- const FINDING_SUBJECT_GRAMMAR_PROMPT = renderFindingSubjectGrammar(FINDING_SUBJECT_KINDS);
1323
- /**
1324
- * The variants each kind is allowed to emit. Used at the kind factory
1325
- * boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
1326
- * subject (the improvement-analyst's job) and vice versa.
1327
- *
1328
- * `failure-mode` is restricted to `cluster` — the only kind that emits
1329
- * a non-locus subject.
1330
- */
1331
- const KIND_EXPECTED_SUBJECTS = {
1332
- "failure-mode": ["cluster"],
1333
- "knowledge-gap": [
1334
- "knowledge.wiki",
1335
- "knowledge.claim",
1336
- "knowledge.raw",
1337
- "knowledge.stale",
1338
- "tool-doc",
1339
- "system-prompt",
1340
- "skill",
1341
- "mcp",
1342
- "subagent",
1343
- "workflow",
1344
- "memory",
1345
- "websearch.outdated",
1346
- "prior-run-summary"
1347
- ],
1348
- "knowledge-poisoning": [
1349
- "knowledge.wiki",
1350
- "knowledge.claim",
1351
- "knowledge.raw",
1352
- "tool-doc",
1353
- "system-prompt",
1354
- "skill",
1355
- "mcp",
1356
- "hook",
1357
- "memory",
1358
- "websearch.outdated",
1359
- "prior-run-summary"
1360
- ],
1361
- improvement: [
1362
- "system-prompt",
1363
- "skill",
1364
- "tool-doc",
1365
- "new-tool",
1366
- "mcp",
1367
- "hook",
1368
- "subagent",
1369
- "workflow",
1370
- "rollout-policy",
1371
- "agent-profile",
1372
- "code",
1373
- "rag",
1374
- "memory",
1375
- "scaffolding",
1376
- "output-schema",
1377
- "knowledge.wiki",
1378
- "knowledge.claim"
1379
- ]
1380
- };
1381
- /** Render only the subject forms one analyst kind is permitted to emit. */
1382
- function findingSubjectGrammarPromptFor(kindId) {
1383
- const kinds = KIND_EXPECTED_SUBJECTS[kindId];
1384
- if (!kinds) throw new Error(`unknown analyst kind: ${kindId}`);
1385
- return renderFindingSubjectGrammar(kinds);
1386
- }
1387
- /**
1388
- * Zod schema that validates a raw subject string and returns the parsed
1389
- * `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
1390
- * `transform`, so `subject` arrives at the kind factory either as a
1391
- * typed locus or as a parse error attached to a single Zod issue.
1392
- *
1393
- * Optionality is preserved: subjects ARE optional on the wire (some
1394
- * findings are descriptive, not actionable). When present, they MUST
1395
- * parse — emitting a malformed subject is a contract violation, not a
1396
- * soft signal.
1397
- */
1398
- const FindingSubjectStringSchema = z.string().refine((s) => parseFindingSubject(s) !== null, { message: "subject does not match the finding-subject grammar" });
1399
- //#endregion
1400
- //#region src/analyst/parse-tolerant.ts
1401
- /**
1402
- * Forgiving pre-parse for analyst findings. Weak models routinely emit
1403
- * schema-correct content in an unusable wrapper — fenced ```json blocks, a
1404
- * single object where an array is expected, trailing commas. Measured: GPT-4o
1405
- * drops to 0% usable output purely from markdown-fence wrapping
1406
- * (arXiv:2605.02363). A five-line de-fence recovers most of it. This module is
1407
- * the de-fence/coerce step that runs BEFORE Zod, so a recoverable finding is
1408
- * repaired, not dropped.
1409
- *
1410
- * Pure + deterministic. No model, no network.
1411
- */
1412
- /** Strip a ```lang ... ``` (or bare ``` ... ```) code fence, if the string is one. */
1413
- function stripCodeFences(text) {
1414
- const t = text.trim();
1415
- const m = t.match(/^```[a-zA-Z0-9]*\s*\n?([\s\S]*?)\n?```$/);
1416
- return m ? m[1].trim() : t;
1417
- }
1418
- /** Remove trailing commas before } or ] — the most common near-JSON defect. */
1419
- function dropTrailingCommas(s) {
1420
- return s.replace(/,(\s*[}\]])/g, "$1");
1421
- }
1422
- /**
1423
- * Best-effort parse of a string into JSON. De-fences, drops trailing commas,
1424
- * then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
1425
- */
1426
- function coerceJson(text) {
1427
- const candidate = dropTrailingCommas(stripCodeFences(text));
1428
- try {
1429
- return JSON.parse(candidate);
1430
- } catch {
1431
- return;
1432
- }
1433
- }
1434
- /**
1435
- * Coerce arbitrary actor/structurer output into an array of candidate finding
1436
- * rows: a JSON string → parse; a single object → 1-element array; an array →
1437
- * as-is; anything else → []. Callers still run each row through Zod
1438
- * (`parseRawFinding`) — this only fixes the shape and never invents fields.
1439
- */
1440
- function coerceToFindingRows(raw) {
1441
- let value = raw;
1442
- if (typeof value === "string") {
1443
- const parsed = coerceJson(value);
1444
- if (parsed === void 0) return [];
1445
- value = parsed;
1446
- }
1447
- if (Array.isArray(value)) return value;
1448
- if (value && typeof value === "object") {
1449
- const inner = value.findings;
1450
- if (Array.isArray(inner)) return inner;
1451
- return [value];
1452
- }
1453
- return [];
1454
- }
1455
- //#endregion
1456
- //#region src/analyst/finding-signature.ts
1457
- /**
1458
- * Typed Ax output for analyst findings.
1459
- *
1460
- * Ax binds the field as `findings:json[]` so the provider emits native
1461
- * structured output. At the kind-factory boundary every row is validated
1462
- * before it becomes an `AnalystFinding`.
1463
- *
1464
- * Why not `f.object().array()` directly in the signature? The Ax
1465
- * signature string `question:string -> findings:json[]` already lets
1466
- * the provider emit JSON arrays. A Zod boundary is required either
1467
- * way (the provider can return any JSON), and Zod gives us a single
1468
- * validation surface independent of which Ax version is installed.
1469
- */
1470
- const ANALYST_SEVERITIES = [
1471
- "critical",
1472
- "high",
1473
- "medium",
1474
- "low",
1475
- "info"
1476
- ];
1477
- const RawAnalystEvidenceSchema = z.object({
1478
- uri: z.string().trim().min(1).max(2e3),
1479
- excerpt: z.string().max(2e3).optional()
1480
- }).strict();
1481
- const RawAnalystFindingBaseShape = {
1482
- severity: z.enum(ANALYST_SEVERITIES),
1483
- claim: z.string().min(1).max(2e3),
1484
- subject: z.string().max(400).refine((subject) => parseFindingSubject(subject) !== null, { message: "subject does not match the finding-subject grammar" }).optional(),
1485
- confidence: z.number().min(0).max(1),
1486
- rationale: z.string().max(4e3).optional(),
1487
- recommended_action: z.string().max(2e3).optional()
1488
- };
1489
- const RawAnalystFindingSchema = z.object({
1490
- ...RawAnalystFindingBaseShape,
1491
- evidence: z.array(RawAnalystEvidenceSchema).min(1)
1492
- }).strict();
1493
- /**
1494
- * Description embedded into the actor prompt so the LLM knows what
1495
- * shape to emit. Kept here so kinds share one source of truth rather
1496
- * than restating the schema in every prompt.
1497
- */
1498
- const RAW_FINDING_SCHEMA_PROMPT = `Each finding MUST be a strict JSON object with:
1499
- - severity: "critical" | "high" | "medium" | "low" | "info"
1500
- - claim: one-sentence statement (max 2000 chars)
1501
- - subject?: one exact subject form listed by this kind; omit rather than guess
1502
- - evidence: REQUIRED non-empty array of {"uri": string, "excerpt"?: string}. Use real identifiers with span://, event://, artifact://, metric://, or finding://. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.
1503
- - confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)
1504
- - rationale?: one or two reasoning sentences
1505
- - recommended_action?: concrete imperative change; omit for descriptive findings
1506
-
1507
- Unknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.`;
1508
- /** Convert raw citations into the public finding evidence envelope. */
1509
- function evidenceRefsFromRawFinding(finding) {
1510
- return finding.evidence.map(({ uri, excerpt }) => ({
1511
- kind: evidenceKindFromUri(uri),
1512
- uri,
1513
- excerpt
1514
- }));
1515
- }
1516
- function parseRawFinding(row, log) {
1517
- return parseFindingWithSchema(RawAnalystFindingSchema, row, log);
1518
- }
1519
- function parseFindingWithSchema(schema, row, log) {
1520
- const result = schema.safeParse(row);
1521
- if (result.success) return result.data;
1522
- if (typeof row === "string") {
1523
- const coerced = coerceJson(row);
1524
- if (coerced !== void 0) {
1525
- const retry = schema.safeParse(coerced);
1526
- if (retry.success) return retry.data;
1527
- }
1528
- }
1529
- log?.("finding rejected: schema failure", { issues: result.error.issues.map((i) => ({
1530
- path: i.path.join("."),
1531
- code: i.code,
1532
- message: i.message
1533
- })) });
1534
- return null;
1535
- }
1536
- function evidenceKindFromUri(uri) {
1537
- if (uri.startsWith("span://")) return "span";
1538
- if (/^trace:\/\/[^/]+\/span\/[^/]+$/.test(uri)) return "span";
1539
- if (uri.startsWith("event://")) return "event";
1540
- if (uri.startsWith("finding://")) return "finding";
1541
- if (uri.startsWith("metric://")) return "metric";
1542
- return "artifact";
1543
- }
1544
- //#endregion
1545
- //#region src/analyst/structure-findings.ts
1546
- /**
1547
- * `structureFindings` — the deferred structuring pass (DSPy TwoStepAdapter /
1548
- * HALO `synthesize_traces` analog). The agentic actor reasons FREE-FORM and
1549
- * emits a prose `report` (which any model does reliably); this separate, cheap
1550
- * call's ONLY job is to turn that report into `AnalystFinding[]`. Decoupling
1551
- * reasoning from structuring is what makes the SEMANTIC findings model-agnostic
1552
- * — the reasoning model never has to satisfy a strict typed-array contract
1553
- * while it diagnoses.
1554
- *
1555
- * Forgiving: the response runs through `coerceToFindingRows` (de-fence, lift
1556
- * single→array) before Zod, and on a zero-finding extraction from a substantive
1557
- * report it reasks ONCE with the schema restated. Returns a typed outcome so a
1558
- * legitimate "nothing to report" is distinguishable from a failed extraction
1559
- * (no silent empty).
1560
- */
1561
- const SYSTEM = [
1562
- "You convert a free-form trace-analysis report into a STRICT JSON array of findings.",
1563
- "Output ONLY the JSON array — no prose, no code fences.",
1564
- RAW_FINDING_SCHEMA_PROMPT,
1565
- "Omit subject when the report does not contain an exact valid locus.",
1566
- "If the report asserts NO problems, output exactly []."
1567
- ].join(" ");
1568
- function buildRows(raw, opts) {
1569
- const rows = coerceToFindingRows(raw);
1570
- const out = [];
1571
- for (const row of rows) {
1572
- const parsed = parseRawFinding(row);
1573
- if (!parsed) continue;
1574
- const callbackResult = opts.processRow ? opts.processRow(parsed) : parsed;
1575
- if (!callbackResult) continue;
1576
- const processed = parseRawFinding(callbackResult);
1577
- if (!processed) continue;
1578
- out.push(makeFinding({
1579
- analyst_id: opts.analystId,
1580
- area: opts.area,
1581
- subject: processed.subject,
1582
- claim: processed.claim,
1583
- rationale: processed.rationale,
1584
- severity: processed.severity,
1585
- confidence: processed.confidence,
1586
- evidence_refs: evidenceRefsFromRawFinding(processed),
1587
- recommended_action: processed.recommended_action,
1588
- ...opts.findingMetadata ? { metadata: { ...opts.findingMetadata } } : {}
1589
- }));
1590
- }
1591
- return out;
1592
- }
1593
- async function structureFindings(opts) {
1594
- const maxReasks = opts.maxReasks ?? 1;
1595
- if (!Number.isSafeInteger(maxReasks) || maxReasks < 0) throw new RangeError("structureFindings: maxReasks must be a non-negative safe integer");
1596
- const llm = {
1597
- baseUrl: opts.baseUrl,
1598
- apiKey: opts.apiKey,
1599
- fetch: opts.fetchImpl
1600
- };
1601
- const costLedger = opts.costLedger ?? new CostLedger();
1602
- let user = `TRACE-ANALYSIS REPORT:\n${opts.report}\n\nReturn the findings JSON array.`;
1603
- for (let attempt = 0; attempt <= maxReasks; attempt++) {
1604
- const request = {
1605
- model: opts.model,
1606
- messages: [{
1607
- role: "system",
1608
- content: SYSTEM
1609
- }, {
1610
- role: "user",
1611
- content: user
1612
- }],
1613
- maxTokens: opts.maxTokens ?? 2e3
1614
- };
1615
- const paid = await costLedger.runPaidCall({
1616
- channel: "analyst",
1617
- phase: opts.costPhase ?? "analyst.structure-findings",
1618
- actor: "structure-findings",
1619
- model: opts.model,
1620
- signal: opts.signal,
1621
- maximumCharge: maximumChargeForLlmRequest(request, llm),
1622
- tags: {
1623
- ...opts.costTags,
1624
- analystId: opts.analystId,
1625
- attempt: String(attempt)
1626
- },
1627
- execute: (signal, callId) => callLlm(request, {
1628
- ...llm,
1629
- signal,
1630
- idempotencyKey: callId
1631
- }),
1632
- receipt: costReceiptFromLlm,
1633
- receiptFromError: costReceiptFromLlmError
1634
- });
1635
- if (!paid.succeeded) throw paid.error;
1636
- const findings = buildRows(paid.value.content.trim(), opts);
1637
- if (findings.length > 0) return {
1638
- findings,
1639
- outcome: "ok"
1640
- };
1641
- if (opts.report.trim().length < 200) return {
1642
- findings: [],
1643
- outcome: "ok"
1644
- };
1645
- user = `${user}\n\nThat produced no valid findings. The report DOES describe issues — re-extract them as the strict JSON array described in the system prompt. Output ONLY the array.`;
1646
- }
1647
- return {
1648
- findings: [],
1649
- outcome: "extraction_failed"
1650
- };
1651
- }
1652
- //#endregion
1653
- //#region src/analyst/kind-factory.ts
1654
- /**
1655
- * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
1656
- *
1657
- * Lifts the Ax pipeline once at registration time so the registry
1658
- * gets a stateless analyst. The Ax agent is freshly constructed per
1659
- * `analyze()` call (the agent carries chat-log + usage state we don't
1660
- * want shared across analyst runs).
1661
- */
1662
- function createTraceAnalystKind(spec, opts) {
1663
- rejectRemovedKindOptions(spec);
1664
- const version = opts.versionSuffix ? `${spec.version}+${opts.versionSuffix}` : spec.version;
1665
- const model = resolveAnalystModel(opts.ai, opts.model);
1666
- const minimumEvidenceCitations = spec.minimumEvidenceCitations ?? 1;
1667
- if (!Number.isInteger(minimumEvidenceCitations) || minimumEvidenceCitations < 1) throw new TypeError("minimumEvidenceCitations must be a positive integer");
1668
- const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs);
1669
- const maxOutputTokens = spec.maxOutputTokens ?? 4096;
1670
- const aiIdentity = opts.aiIdentity === void 0 ? null : snapshotExactExecutionComponentIdentity(opts.aiIdentity, "createTraceAnalystKind aiIdentity");
1671
- return {
1672
- id: spec.id,
1673
- description: spec.description,
1674
- inputKind: "trace-store",
1675
- cost: {
1676
- ...spec.cost,
1677
- settlement_timeout_ms: settlementTimeoutMs
1678
- },
1679
- version,
1680
- executionConfig: {
1681
- kind: "trace-analyst",
1682
- model,
1683
- ai_identity: aiIdentity,
1684
- actor_description_digest: hashCanonical(spec.actorDescription.trim()),
1685
- max_subqueries: spec.subqueries?.maxCalls ?? 0,
1686
- max_parallel_subqueries: spec.subqueries?.maxParallel ?? 2,
1687
- max_turns: spec.maxTurns ?? 12,
1688
- max_runtime_chars: spec.maxRuntimeChars ?? 6e3,
1689
- max_output_tokens: maxOutputTokens,
1690
- minimum_evidence_citations: minimumEvidenceCitations,
1691
- require_structured_findings: spec.requireStructuredFindings ?? false,
1692
- prepare_context: spec.prepareContext === void 0 ? "disabled" : "version-bound",
1693
- post_process: spec.postProcess === void 0 ? "disabled" : "version-bound",
1694
- recovery: opts.recovery === void 0 ? null : {
1695
- base_url: opts.recovery.baseUrl,
1696
- model: opts.recovery.model ?? model,
1697
- api_key_provided: opts.recovery.apiKey !== void 0,
1698
- fetch_implementation: opts.recovery.fetchImpl === void 0 ? "global" : "version-bound"
1699
- },
1700
- settlement_timeout_ms: settlementTimeoutMs
1701
- },
1702
- async analyze(store, ctx) {
1703
- const costLedger = ctx.costLedger ?? new CostLedger(ctx.budgetUsd);
1704
- const costTags = {
1705
- ...ctx.tags ?? {},
1706
- analystId: spec.id,
1707
- ...ctx.correlationId ? { analystRunId: ctx.correlationId } : {}
1708
- };
1709
- const meteredAi = meterAxChatService(opts.ai, {
1710
- ledger: costLedger,
1711
- actor: spec.id,
1712
- maxOutputTokens,
1713
- defaultModel: model,
1714
- phase: ctx.costPhase,
1715
- signal: ctx.signal,
1716
- tags: costTags
1717
- });
1718
- try {
1719
- const preparedContext = await spec.prepareContext?.(store, ctx);
1720
- if (preparedContext !== void 0 && typeof preparedContext !== "string") throw new TypeError(`Trace analyst '${spec.id}' prepareContext must return a string`);
1721
- const tools = preparedContext === void 0 ? spec.buildTools(store) : [];
1722
- const analysisMode = preparedContext === void 0 ? "tool-loop" : "prepared-context";
1723
- const maxSubqueries = spec.subqueries?.maxCalls ?? 0;
1724
- const maxParallel = spec.subqueries?.maxParallel ?? 2;
1725
- const priorContext = renderPriorFindings(ctx.priorFindings);
1726
- const upstreamContext = renderUpstreamFindings(ctx.upstreamFindings);
1727
- const actorDescription = spec.actorDescription.trim() + priorContext + upstreamContext + "\n\n" + RAW_FINDING_SCHEMA_PROMPT + (minimumEvidenceCitations > 1 ? `\n\nThis kind requires at least ${minimumEvidenceCitations} evidence citations per finding; rows with fewer are rejected.` : "") + "\n\nFirst write `report`: a concise free-form prose diagnosis of what the traces show — what succeeded, what was suboptimal or failed — with concrete trace ids and numbers. THEN return the structured `findings` array (it MAY be empty when there is nothing to report).";
1728
- ctx.log?.(`analyst.kind ${spec.id} forward`, {
1729
- max_subqueries: maxSubqueries,
1730
- tool_count: tools.length,
1731
- analysis_mode: analysisMode,
1732
- prepared_context_chars: preparedContext?.length ?? 0,
1733
- tags: ctx.tags
1734
- });
1735
- const completed = await runTraceAnalysisLoop({
1736
- id: spec.id,
1737
- description: spec.description,
1738
- prompt: actorDescription,
1739
- question: deriveQuestion(ctx, spec),
1740
- ai: meteredAi,
1741
- model,
1742
- tools,
1743
- findingType: "object",
1744
- maxSubqueries,
1745
- maxParallelSubqueries: maxParallel,
1746
- maxTurns: spec.maxTurns ?? 12,
1747
- maxRuntimeChars: spec.maxRuntimeChars ?? 6e3,
1748
- ...preparedContext !== void 0 ? { context: preparedContext } : {},
1749
- ...ctx.signal ? { signal: ctx.signal } : {}
1750
- });
1751
- const { report, findings: submittedFindings } = completed;
1752
- const expectedSubjects = KIND_EXPECTED_SUBJECTS[spec.id];
1753
- const out = [];
1754
- const rawRows = submittedFindings;
1755
- let rejectedWrongKind = 0;
1756
- let rejectedInsufficientEvidence = 0;
1757
- const processRow = (parsed) => {
1758
- const callbackResult = spec.postProcess ? spec.postProcess(parsed, ctx) : parsed;
1759
- if (!callbackResult) return null;
1760
- const postProcessed = parseRawFinding(callbackResult, ctx.log);
1761
- if (!postProcessed) return null;
1762
- if (expectedSubjects && postProcessed.subject !== void 0) {
1763
- const parsedSubject = parseFindingSubject(postProcessed.subject);
1764
- if (parsedSubject === null) {
1765
- ctx.log?.("finding rejected: subject failed to parse", {
1766
- kind: spec.id,
1767
- subject: postProcessed.subject
1768
- });
1769
- rejectedWrongKind += 1;
1770
- return null;
1771
- }
1772
- if (!expectedSubjects.includes(parsedSubject.kind)) {
1773
- ctx.log?.("finding rejected: subject variant not allowed for this kind", {
1774
- kind: spec.id,
1775
- subject_kind: parsedSubject.kind,
1776
- subject: postProcessed.subject,
1777
- allowed: expectedSubjects
1778
- });
1779
- rejectedWrongKind += 1;
1780
- return null;
1781
- }
1782
- }
1783
- const distinctEvidenceCitations = new Set(postProcessed.evidence.map((citation) => citation.uri.trim())).size;
1784
- if (distinctEvidenceCitations < minimumEvidenceCitations) {
1785
- ctx.log?.("finding rejected: insufficient evidence citations", {
1786
- kind: spec.id,
1787
- required: minimumEvidenceCitations,
1788
- received: postProcessed.evidence.length,
1789
- distinct: distinctEvidenceCitations
1790
- });
1791
- rejectedInsufficientEvidence += 1;
1792
- return null;
1793
- }
1794
- return postProcessed;
1795
- };
1796
- for (const row of rawRows) {
1797
- const parsed = parseRawFinding(row, ctx.log);
1798
- if (!parsed) continue;
1799
- const postProcessed = processRow(parsed);
1800
- if (!postProcessed) continue;
1801
- out.push(toAnalystFinding(spec, version, postProcessed, {
1802
- analysis_mode: analysisMode,
1803
- analysis_turn_count: completed.turnCount
1804
- }));
1805
- }
1806
- ctx.log?.(`analyst.kind ${spec.id} done`, {
1807
- emitted: rawRows.length,
1808
- accepted: out.length,
1809
- rejected_wrong_subject: rejectedWrongKind,
1810
- rejected_insufficient_evidence: rejectedInsufficientEvidence
1811
- });
1812
- if (out.length === 0 && report.trim().length >= 200) {
1813
- if (opts.recovery) {
1814
- const wrongKindBefore = rejectedWrongKind;
1815
- const insufficientEvidenceBefore = rejectedInsufficientEvidence;
1816
- const recovered = await structureFindings({
1817
- report,
1818
- analystId: spec.id,
1819
- area: spec.area,
1820
- model: opts.recovery.model ?? model,
1821
- baseUrl: opts.recovery.baseUrl,
1822
- apiKey: opts.recovery.apiKey,
1823
- fetchImpl: opts.recovery.fetchImpl,
1824
- costLedger,
1825
- costPhase: ctx.costPhase,
1826
- costTags,
1827
- signal: ctx.signal,
1828
- maxTokens: Math.min(maxOutputTokens, 2e3),
1829
- processRow,
1830
- findingMetadata: { kind_version: version }
1831
- });
1832
- out.push(...recovered.findings);
1833
- ctx.log?.(`analyst.kind ${spec.id} recovery`, {
1834
- outcome: recovered.outcome,
1835
- recovered: recovered.findings.length,
1836
- rejected_wrong_subject: rejectedWrongKind - wrongKindBefore,
1837
- rejected_insufficient_evidence: rejectedInsufficientEvidence - insufficientEvidenceBefore
1838
- });
1839
- }
1840
- if (out.length === 0) {
1841
- if (spec.requireStructuredFindings) throw new Error(`Trace analyst '${spec.id}' produced no valid structured findings after ${completed.turnCount} turns: ${truncateForContext(report, 600)}`);
1842
- const fallback = processRow({
1843
- claim: "Analyst produced a diagnosis but no structured findings — see report.",
1844
- rationale: report.slice(0, 1500),
1845
- severity: "info",
1846
- confidence: .3,
1847
- evidence: [{
1848
- uri: "report://summary",
1849
- excerpt: report.slice(0, 2e3)
1850
- }]
1851
- });
1852
- if (fallback) out.push(toAnalystFinding(spec, version, fallback, {
1853
- analysis_mode: analysisMode,
1854
- analysis_turn_count: completed.turnCount,
1855
- outcome: "extraction_failed"
1856
- }));
1857
- else throw new Error(`Trace analyst '${spec.id}' produced a substantive report, but no finding satisfied its acceptance rules`);
1858
- }
1859
- }
1860
- return out;
1861
- } finally {
1862
- const usage = await settleUsageReceiptFromCostLedger(costLedger, {
1863
- tags: {
1864
- analystId: spec.id,
1865
- ...ctx.correlationId ? { analystRunId: ctx.correlationId } : {}
1866
- },
1867
- timeoutMs: settlementTimeoutMs
1868
- });
1869
- if (!usage.settled) ctx.log?.(`analyst.kind ${spec.id} provider settlement timed out`, {
1870
- pending_calls: usage.pendingCalls,
1871
- timeout_ms: settlementTimeoutMs
1872
- });
1873
- ctx.recordUsage?.(usage.receipt);
1874
- }
1875
- }
1876
- };
1877
- }
1878
- function rejectRemovedKindOptions(spec) {
1879
- const supplied = spec;
1880
- for (const [removed, replacement] of [
1881
- ["recursion", "subqueries"],
1882
- ["responderDescription", "actorDescription"],
1883
- ["maxDepth", "subqueries"],
1884
- ["maxParallelSubagents", "subqueries.maxParallel"],
1885
- ["subagentDescription", "actorDescription"]
1886
- ]) if (removed in supplied) throw new TypeError(`createTraceAnalystKind: '${removed}' is unsupported; use '${replacement}'`);
1887
- }
1888
- function deriveQuestion(ctx, spec) {
1889
- const focus = ctx.tags?.focus?.trim();
1890
- const task = `Analyze this trace dataset with the available tools and report ${spec.area} findings. ${spec.description}`;
1891
- return focus ? `${task} Focus: ${focus}.` : task;
1892
- }
1893
- function toAnalystFinding(spec, version, raw, metadata = {}) {
1894
- return makeFinding({
1895
- analyst_id: spec.id,
1896
- area: spec.area,
1897
- subject: raw.subject,
1898
- claim: raw.claim,
1899
- rationale: raw.rationale,
1900
- severity: raw.severity,
1901
- confidence: raw.confidence,
1902
- evidence_refs: evidenceRefsFromRawFinding(raw),
1903
- recommended_action: raw.recommended_action,
1904
- metadata: {
1905
- kind_version: version,
1906
- ...metadata
1907
- }
1908
- });
1909
- }
1910
- /**
1911
- * Render a compact prior-findings block the actor reads alongside its
1912
- * brief. Each row is one line so the actor can scan dozens cheaply.
1913
- * The kind's prompt instructs the actor to (a) check whether a new
1914
- * cluster matches a prior `finding_id` (carry the id forward via
1915
- * `id_basis` to keep diffs stable) and (b) raise severity / confidence
1916
- * when a prior finding has reappeared without remediation.
1917
- *
1918
- * Returns the empty string when there are no prior findings — most
1919
- * runs are "first-of-its-kind" and the prompt stays unchanged.
1920
- *
1921
- * Exported for tests + for consumers that build their own actor
1922
- * prompts (e.g. specialized analysts living outside the default kinds).
1923
- */
1924
- function renderPriorFindings(prior) {
1925
- if (!prior || prior.length === 0) return "";
1926
- const MAX_ROWS = 40;
1927
- const rows = prior.slice(0, MAX_ROWS).map((f) => {
1928
- const subject = f.subject ? ` [${f.subject}]` : "";
1929
- return ` - id=${f.finding_id} ${f.severity}${subject} ${truncateForContext(f.claim, 160)}`;
1930
- });
1931
- const overflow = prior.length > MAX_ROWS ? `\n ... +${prior.length - MAX_ROWS} more prior findings (older history truncated)` : "";
1932
- return [
1933
- "",
1934
- "",
1935
- "PRIOR FINDINGS (from a previous run on related data):",
1936
- "When the work you do now matches a row below, REUSE the `finding_id` (pass it as `id_basis`) so the cross-run diff stays stable.",
1937
- "A finding that reappears with no remediation evidence SHOULD raise its `confidence` and may justify a higher `severity`.",
1938
- ...rows,
1939
- overflow
1940
- ].filter(Boolean).join("\n");
1941
- }
1942
- /** Render findings produced earlier in this same registry run. */
1943
- function renderUpstreamFindings(upstream) {
1944
- if (!upstream || upstream.length === 0) return "";
1945
- const MAX_ROWS = 40;
1946
- const rows = upstream.slice(0, MAX_ROWS).map((finding) => {
1947
- const subject = finding.subject ? ` [${finding.subject}]` : "";
1948
- const action = finding.recommended_action ? ` action=${truncateForContext(finding.recommended_action, 120)}` : "";
1949
- const evidence = finding.evidence_refs[0] ? ` evidence=${truncateForContext(finding.evidence_refs[0].uri, 120)}` : "";
1950
- return ` - id=${finding.finding_id} source=${finding.analyst_id} ${finding.severity}${subject} claim=${truncateForContext(finding.claim, 160)}${action}${evidence}`;
1951
- });
1952
- const overflow = upstream.length > MAX_ROWS ? `\n ... +${upstream.length - MAX_ROWS} more upstream findings (truncated)` : "";
1953
- return [
1954
- "",
1955
- "",
1956
- "UPSTREAM FINDINGS (produced earlier in this same registry run):",
1957
- "Use these as intermediate evidence. Build on them instead of repeating the same diagnosis, and cite a dependency with `finding://<id>`.",
1958
- ...rows,
1959
- overflow
1960
- ].filter(Boolean).join("\n");
1961
- }
1962
- function truncateForContext(s, max) {
1963
- if (s.length <= max) return s;
1964
- return `${s.slice(0, max - 1).trimEnd()}…`;
1965
- }
1966
- //#endregion
1967
566
  //#region src/analyst/kinds/control-integrity.ts
1968
567
  const ANALYST_ID = "control-integrity";
1969
568
  function shown(value) {
@@ -2023,95 +622,39 @@ var ControlIntegrityAnalyst = class {
2023
622
  }
2024
623
  };
2025
624
  const CONTROL_INTEGRITY_ANALYST = new ControlIntegrityAnalyst();
2026
- //#endregion
2027
- //#region src/analyst/tool-groups.ts
2028
- const TOOL_NAMES_BY_GROUP = {
2029
- all: /* @__PURE__ */ new Set(),
2030
- discovery: /* @__PURE__ */ new Set([
2031
- "getDatasetOverview",
2032
- "queryTraces",
2033
- "countTraces"
2034
- ]),
2035
- discoveryAndRead: /* @__PURE__ */ new Set([
2036
- "getDatasetOverview",
2037
- "queryTraces",
2038
- "countTraces",
2039
- "viewTrace",
2040
- "viewSpans"
2041
- ]),
2042
- discoveryAndSearch: /* @__PURE__ */ new Set([
2043
- "getDatasetOverview",
2044
- "queryTraces",
2045
- "countTraces",
2046
- "searchTrace",
2047
- "searchSpan"
2048
- ]),
2049
- targeted: /* @__PURE__ */ new Set([
2050
- "getDatasetOverview",
2051
- "queryTraces",
2052
- "viewSpans",
2053
- "searchSpan"
2054
- ]),
2055
- singleTrace: /* @__PURE__ */ new Set([
2056
- "getDatasetOverview",
2057
- "viewTrace",
2058
- "viewSpans",
2059
- "searchTrace",
2060
- "searchSpan"
2061
- ])
2062
- };
2063
- /**
2064
- * Build the tool set for a named group bound to a specific trace store.
2065
- *
2066
- * `all` returns every tool. Other groups filter `buildTraceAnalystTools`
2067
- * by name to the documented subset. An unrecognised group name throws —
2068
- * silently returning all tools would defeat the cost-control point.
2069
- */
2070
- function buildTraceToolsForGroup(group, store) {
2071
- const all = buildTraceAnalystTools({ store });
2072
- if (group === "all") return all;
2073
- const allow = TOOL_NAMES_BY_GROUP[group];
2074
- if (!allow) throw new Error(`unknown trace tool group: ${group}`);
2075
- return all.filter((tool) => allow.has(tool.name));
2076
- }
2077
625
  const FAILURE_MODE_KIND_SPEC = {
2078
626
  id: "failure-mode",
2079
627
  description: "Clusters trace-dataset failures into distinct failure modes with cited evidence and a short recommended action.",
2080
628
  area: "failure-mode",
2081
629
  version: "1.2.0",
2082
- actorDescription: `You are a failure-mode classifier for an OTLP trace dataset. Your job is to identify the **distinct ways agents failed** in this dataset, not to grade individual runs.
630
+ instructions: `You are a failure-mode classifier for an OTLP trace dataset. Your job is to identify the **distinct ways agents failed** in this dataset, not to grade individual runs.
2083
631
 
2084
632
  ${findingSubjectGrammarPromptFor("failure-mode")}
2085
633
 
2086
634
  DISCOVERY → CLUSTER → CITE protocol:
2087
635
 
2088
- 1. Call \`traces.getDatasetOverview({})\` first. Use \`has_errors\`, \`models\`, \`agent_names\`, \`tools\`, and \`sample_trace_ids\` to size the failure surface.
2089
- 2. Use \`traces.queryTraces({ filters: { has_errors: true }, limit })\` to pull error-bearing traces. Combine with \`traces.countTraces\` to see what fraction of the dataset failed.
2090
- 3. For each candidate failure cluster, use \`traces.searchTrace\` with regex like \`STATUS_CODE_ERROR\`, \`MaxTurnsExceeded\`, \`assertion\`, \`unauthorized\`, \`timeout\`, \`429\`, \`5\\d\\d\`, the agent's specific error strings, or the names of its tools. Pull one or two representative traces per cluster, **not all** of them.
636
+ 1. Call \`getDatasetOverview({})\` first. Use \`has_errors\`, \`models\`, \`agent_names\`, \`tools\`, and \`sample_trace_ids\` to size the failure surface.
637
+ 2. Use \`queryTraces(filters={"has_errors": true}, limit=...)\` to pull error-bearing traces. Combine with \`countTraces\` to see what fraction of the dataset failed.
638
+ 3. For each candidate failure cluster, use \`searchTrace\` with regex like \`STATUS_CODE_ERROR\`, \`MaxTurnsExceeded\`, \`assertion\`, \`unauthorized\`, \`timeout\`, \`429\`, \`5\\d\\d\`, the agent's specific error strings, or the names of its tools. Pull one or two representative traces per cluster, **not all** of them.
2091
639
  4. **Cluster, do not enumerate.** Two failures with the same root cause should be ONE finding citing both traces, not two findings. The point of this analyst is to compress N runs into K modes.
2092
640
  5. For each defensible cluster, emit ONE finding. Use a lowercase cluster label matching the subject grammar ("tool-call-loop", "auth-revoked-mid-run", ...). Rate it critical when it blocks the run, high when the run finishes degraded, and medium when it slows convergence. Cite representative spans and include exact error, payload, or contradictory-output quotes. Use confidence 0.85+ when multiple traces show the same shape, 0.6-0.8 for a single-trace inference, and <0.5 for speculation. Keep the imperative fix idea short; the improvement analyst expands it.
2093
641
 
2094
642
  If the dataset has no failures, return an empty findings array — do NOT pad with low-confidence speculation.
2095
643
 
2096
- **Use subqueries over loaded evidence.** After the first scan, load representative span excerpts for each candidate cluster. Then send one bounded \`llmQuery\` per cluster in one batch, including the exact excerpts and asking it to classify the root cause. Subqueries cannot call trace tools. Merge or split clusters yourself from their classifications and the cited source evidence.
2097
-
2098
- OBSERVABILITY rules:
2099
- - Each non-final turn must emit at least one \`console.log\` for evidence.
2100
- - Reuse runtime variables across turns; don't recompute.`,
2101
- buildTools: (store) => buildTraceToolsForGroup("all", store),
2102
- subqueries: {
2103
- maxCalls: 8,
2104
- maxParallel: 4
2105
- },
2106
- maxTurns: 24,
2107
- cost: { kind: "llm" }
644
+ **Use subqueries over loaded evidence.** After the first scan, load representative span excerpts for each candidate cluster. Then send one bounded \`llm_query\` per cluster in one batch, including the exact excerpts and asking it to classify the root cause. Subqueries cannot call trace tools. Merge or split clusters yourself from their classifications and the cited source evidence.`,
645
+ toolGroup: "all",
646
+ limits: {
647
+ maxLlmCalls: 8,
648
+ maxIterations: 24,
649
+ maxToolCalls: 64
650
+ }
2108
651
  };
2109
652
  const IMPROVEMENT_KIND_SPEC = {
2110
653
  id: "improvement",
2111
654
  description: "Converts upstream failure / gap / poisoning findings into concrete locus-named edits (prompt, tool-doc, RAG, scaffolding) with leverage grades.",
2112
655
  area: "improvement",
2113
656
  version: "1.2.0",
2114
- actorDescription: `You are a self-improvement analyst. Your job is to propose **concrete, locus-named edits** the agent's runtime should adopt to fix the failure modes, knowledge gaps, and poisonings present in this dataset.
657
+ instructions: `You are a self-improvement analyst. Your job is to propose **concrete, locus-named edits** the agent's runtime should adopt to fix the failure modes, knowledge gaps, and poisonings present in this dataset.
2115
658
 
2116
659
  Upstream analysts have already classified the problems. Your job is to convert each problem into a *change to make* and grade its expected leverage. Each finding is one proposed edit.
2117
660
 
@@ -2119,7 +662,7 @@ ${findingSubjectGrammarPromptFor("improvement")}
2119
662
 
2120
663
  DISCOVERY → CANDIDATE-FIXES → COMPETE → CITE protocol:
2121
664
 
2122
- 1. \`traces.getDatasetOverview({})\` first. Note the agents, tools, and any system-prompt fingerprints (look for the prompt text echoed in early spans).
665
+ 1. \`getDatasetOverview({})\` first. Note the agents, tools, and any system-prompt fingerprints (look for the prompt text echoed in early spans).
2123
666
  2. For each high-severity failure pattern, generate 2-3 candidate fixes. Real candidate axes:
2124
667
  - **System-prompt edit** — add an instruction, remove a misleading one, restructure precedence
2125
668
  - **Tool description edit** — rewrite a tool's description so the agent picks it correctly / passes valid args
@@ -2131,33 +674,29 @@ DISCOVERY → CANDIDATE-FIXES → COMPETE → CITE protocol:
2131
674
  - **Skill / MCP / hook / subagent** — change the reusable profile component responsible for the behavior
2132
675
  - **Workflow / rollout policy** — change orchestration, budget, sampling, or stopping behavior
2133
676
  - **Code** — change an implementation path when profile edits cannot repair the behavior
2134
- 3. **Compare candidate fixes with bounded subqueries.** Load the representative failure excerpts, then send one \`llmQuery\` per candidate-fix axis the same evidence. Ask for likely effect, side effects, and implementation scope. Subqueries cannot call trace tools; trace ids alone are insufficient context.
677
+ 3. **Compare candidate fixes with bounded subqueries.** Load the representative failure excerpts, then send one \`llm_query\` per candidate-fix axis the same evidence. Ask for likely effect, side effects, and implementation scope. Subqueries cannot call trace tools; trace ids alone are insufficient context.
2135
678
  4. After the comparisons return, **pick the winning candidate per cluster** based on expected effect and risk, then emit ONE finding. Keep the alternatives and rejection reasons in the rationale so the recommendation is auditable.
2136
679
  5. **Cross-reference upstream findings.** Cite prior failure-mode or knowledge-gap findings as \`finding://<prior-finding-id>\`. This builds the dependency graph that lets the dashboard show "fix #X resolves failure modes A, B, C."
2137
680
 
2138
681
  For each winning recommendation, emit ONE finding. Use one exact locus from the subject grammar and state the edit in one sentence. Match leverage to the source failure's severity; use medium for quality-of-life changes and info for cleanup with no behavioral effect. Cite the targeted \`finding://<id>\` when available and the most representative span when useful. Quote the problem being fixed. Use confidence 0.85+ for a mechanical fix to a well-evidenced failure, 0.6-0.8 when judgment is required, and <0.5 for speculation. Explain in at most two sentences why this candidate beat its alternatives. The recommended action must be the literal diff, quoted replacement, tool description, or setting change.
2139
682
 
2140
- If no upstream failure findings exist in this run, derive your own from the trace dataset using the failure-mode protocol inline (\`searchTrace\` for STATUS_CODE_ERROR / MaxTurnsExceeded / etc.). But prefer to consume upstream findings when present the kinds are designed to chain.
2141
-
2142
- Do NOT propose a fix you cannot defend with evidence. "Tighten the prompt" is not a finding; "Add 'When the user asks for X, always Y' to the system prompt section "request-classification"" is.
683
+ If no upstream failure findings exist in this run, derive your own from the trace dataset using the failure-mode protocol inline (\`searchTrace\` for STATUS_CODE_ERROR / MaxTurnsExceeded / etc.). Prefer upstream findings when present because the analysts are designed to chain.
2143
684
 
2144
- OBSERVABILITY rules:
2145
- - Each non-final turn must emit at least one \`console.log\` for evidence.`,
2146
- buildTools: (store) => buildTraceToolsForGroup("all", store),
2147
- subqueries: {
2148
- maxCalls: 8,
2149
- maxParallel: 4
2150
- },
2151
- maxTurns: 30,
2152
- maxRuntimeChars: 12e3,
2153
- cost: { kind: "llm" }
685
+ Do NOT propose a fix you cannot defend with evidence. "Tighten the prompt" is not a finding; "Add 'When the user asks for X, always Y' to the system prompt section "request-classification"" is.`,
686
+ toolGroup: "all",
687
+ limits: {
688
+ maxLlmCalls: 8,
689
+ maxIterations: 30,
690
+ maxToolCalls: 80,
691
+ maxOutputChars: 12e3
692
+ }
2154
693
  };
2155
694
  const KNOWLEDGE_GAP_KIND_SPEC = {
2156
695
  id: "knowledge-gap",
2157
696
  description: "Identifies missing or stale pieces of knowledge — primarily against the agent-knowledge wiki — and attributes each to the runtime layer (wiki page, claim, raw source, websearch, tool-doc, system-prompt, memory) that should have held it.",
2158
697
  area: "knowledge-gap",
2159
698
  version: "1.2.0",
2160
- actorDescription: `You are a knowledge-gap analyst for an OTLP trace dataset. Your job is to identify the **specific pieces of information the agent lacked, or that were stale**, that caused poor decisions.
699
+ instructions: `You are a knowledge-gap analyst for an OTLP trace dataset. Your job is to identify the **specific pieces of information the agent lacked, or that were stale**, that caused poor decisions.
2161
700
 
2162
701
  The agent under analysis maintains a curated knowledge base via \`@tangle-network/agent-knowledge\` — a wiki of \`KnowledgePage\`s with raw source anchors, claims, and relations. The primary expected store of agent-knowable facts IS that wiki. A "knowledge gap" is anything the agent had to discover or guess at run-time that the wiki should have held — or an outdated/contradictory fact the agent picked up from a non-wiki source.
2163
702
 
@@ -2165,7 +704,7 @@ ${findingSubjectGrammarPromptFor("knowledge-gap")}
2165
704
 
2166
705
  DISCOVERY → ATTRIBUTE-TO-LAYER → CITE protocol:
2167
706
 
2168
- 1. \`traces.getDatasetOverview({})\` first. Note which agents, tools, and models appear.
707
+ 1. \`getDatasetOverview({})\` first. Note which agents, tools, and models appear.
2169
708
  2. Pull traces where the agent shows gap signals. The strongest signals are:
2170
709
  - Self-correction turns ("I assumed X but…", "let me re-check", "actually,")
2171
710
  - Clarifying-question turns where the agent asked the user something the runtime should have surfaced
@@ -2174,36 +713,32 @@ DISCOVERY → ATTRIBUTE-TO-LAYER → CITE protocol:
2174
713
  - Web-search calls returning pages dated before a known cutoff for content that changes (versioned APIs, schemas, policies)
2175
714
  - Agent quoting a tool's docs / system prompt incorrectly because the actual text was insufficient
2176
715
  - Fabricated identifiers that don't appear in dataset \`sample_trace_ids\`
2177
- Use \`traces.searchTrace\` with patterns like \`I (don.?t|do not) know\`, \`assumed\`, \`unclear\`, \`could you (clarify|tell me|provide)\`, \`not found\`, \`undefined\`, \`unknown\`, \`null\`, dates older than the analysis window, or the agent's specific clarification phrases.
716
+ Use \`searchTrace\` with patterns like \`I (don.?t|do not) know\`, \`assumed\`, \`unclear\`, \`could you (clarify|tell me|provide)\`, \`not found\`, \`undefined\`, \`unknown\`, \`null\`, dates older than the analysis window, or the agent's specific clarification phrases.
2178
717
  3. For each gap, identify the **layer of the runtime that should have prevented it** and use its exact locus from the subject grammar above.
2179
718
  4. For each defensible gap, emit ONE finding. Use an exact locus from the subject grammar and name the missing or stale knowledge (for example, "wiki has no page on invoice line-item shape; agent re-derived it from raw spans"). Rate it high when it caused failure or a clarifying question, medium for unnecessary turns, and low for minor inefficiency. Cite the span where the question, correction, retrieval miss, or stale result surfaced and quote it exactly. Use confidence 0.85+ when the agent articulated the gap and 0.6-0.8 when inferred. Recommend a concrete wiki edit for an agent-knowledge locus or a prompt/tool-description edit otherwise.
2180
719
 
2181
- **Compare layers over loaded evidence.** After the first scan, load the exact excerpts behind candidates across \`agent-knowledge:*\`, \`websearch:outdated\`, \`tool-doc:*\`, \`system-prompt:*\`, and \`memory:*\`. Use one bounded \`llmQuery\` per layer to classify those excerpts. Subqueries cannot call trace tools. Merge their classifications into the final finding set only when the source excerpts support them.
720
+ **Compare layers over loaded evidence.** After the first scan, load the exact excerpts behind candidates across \`agent-knowledge:*\`, \`websearch:outdated\`, \`tool-doc:*\`, \`system-prompt:*\`, and \`memory:*\`. Use one bounded \`llm_query\` per layer to classify those excerpts. Subqueries cannot call trace tools. Merge their classifications into the final finding set only when the source excerpts support them.
2182
721
 
2183
- Do NOT report a gap that the agent later recovered from cleanly within the same turn that's resilience, not a gap. Cite the *non-recovery* version when both exist.
2184
-
2185
- OBSERVABILITY rules:
2186
- - Each non-final turn must emit at least one \`console.log\` for evidence.`,
2187
- buildTools: (store) => buildTraceToolsForGroup("discoveryAndSearch", store),
2188
- subqueries: {
2189
- maxCalls: 5,
2190
- maxParallel: 4
2191
- },
2192
- maxTurns: 18,
2193
- cost: { kind: "llm" }
722
+ Do NOT report a gap that the agent later recovered from cleanly within the same turn. That is resilience, not a gap. Cite the non-recovery version when both exist.`,
723
+ toolGroup: "discoveryAndSearch",
724
+ limits: {
725
+ maxLlmCalls: 5,
726
+ maxIterations: 18,
727
+ maxToolCalls: 48
728
+ }
2194
729
  };
2195
730
  const KNOWLEDGE_POISONING_KIND_SPEC = {
2196
731
  id: "knowledge-poisoning",
2197
732
  description: "Identifies confident-but-wrong actions caused by stale memory, contradicting RAG, deprecated tool docs, or outdated system-prompt instructions.",
2198
733
  area: "knowledge-poisoning",
2199
734
  version: "1.2.0",
2200
- actorDescription: `You are a knowledge-poisoning analyst for an OTLP trace dataset. Your job is to identify cases where the agent **confidently used wrong information** — not where it lacked information (that's the knowledge-gap analyst).
735
+ instructions: `You are a knowledge-poisoning analyst for an OTLP trace dataset. Your job is to identify cases where the agent **confidently used wrong information** — not where it lacked information (that's the knowledge-gap analyst).
2201
736
 
2202
737
  ${findingSubjectGrammarPromptFor("knowledge-poisoning")}
2203
738
 
2204
739
  DISCOVERY → DUAL-VERIFY → CITE protocol:
2205
740
 
2206
- 1. \`traces.getDatasetOverview({})\` first. Identify the agents, models, and tools.
741
+ 1. \`getDatasetOverview({})\` first. Identify the agents, models, and tools.
2207
742
  2. Pull traces where the agent's confident action was later contradicted. Strongest signals:
2208
743
  - Agent stated a fact in one span; a later span surfaced contradictory evidence; the agent then proceeded anyway or fabricated reconciliation.
2209
744
  - Tool call with stale arguments (an id that no longer exists, an API shape that changed).
@@ -2211,28 +746,24 @@ DISCOVERY → DUAL-VERIFY → CITE protocol:
2211
746
  - Web-search result the agent cited that returned an outdated page; agent treated it as canonical.
2212
747
  - System-prompt instruction the agent followed that ground-truth evidence in the trace contradicts (e.g. prompt says "use endpoint A"; tool reply says "endpoint A deprecated, use B").
2213
748
  - Repeated wrong-shape parsing despite the tool's actual output proving the shape.
2214
- 3. Use \`traces.searchTrace\` with regex on phrases like \`actually\`, \`turns out\`, \`previously assumed\`, \`old version\`, \`deprecated\`, \`updated to\`, \`now uses\`, or specific entity names you suspect have changed.
749
+ 3. Use \`searchTrace\` with regex on phrases like \`actually\`, \`turns out\`, \`previously assumed\`, \`old version\`, \`deprecated\`, \`updated to\`, \`now uses\`, or specific entity names you suspect have changed.
2215
750
  4. For each candidate poisoning, **DUAL-VERIFY**:
2216
751
  - Confirm the agent actually acted on the false belief (cite the span where it did)
2217
752
  - Confirm the belief is actually false in this trace's own evidence (cite the span that contradicts it)
2218
- Only emit a finding when both halves are nailed down. If you can only nail one, drop it single-evidence poisoning findings are too speculative to be useful.
753
+ Only emit a finding when both halves are supported. If you can only support one, drop it because single-evidence poisoning findings are too speculative to be useful.
2219
754
 
2220
- **Independently assess both halves.** Load the action excerpt and contradicting excerpt yourself, then send bounded \`llmQuery\` calls the exact evidence for "did the agent act?" and "does the trace contradict the belief?" Subqueries cannot call trace tools. Accept a poisoning only when both assessments and the source excerpts support it.
755
+ **Independently assess both halves.** Load the action excerpt and contradicting excerpt yourself, then send bounded \`llm_query\` calls the exact evidence for "did the agent act?" and "does the trace contradict the belief?" Subqueries cannot call trace tools. Accept a poisoning only when both assessments and the source excerpts support it.
2221
756
 
2222
757
  For each confirmed poisoning, emit ONE finding. Use the source of the false belief as the exact subject. State "agent believed X (from source S); trace evidence shows X is false." Rate it critical for a wrong user-visible action, high when caught internally after significant waste, and medium for inefficiency. Cite BOTH the action span and the contradicting span with exact quotes. Use confidence 0.85+ when both halves have exact quotes and 0.6-0.8 when one half is inferred. Recommend the literal source correction: update the wiki claim, invalidate and re-curate the raw source, or replace the stale prompt/tool instruction.
2223
758
 
2224
- Do NOT report a finding if the agent caught and corrected the false belief in the same turn — that's the system working. Reserve poisoning for cases where the false belief shaped downstream action.
2225
-
2226
- OBSERVABILITY rules:
2227
- - Each non-final turn must emit at least one \`console.log\` for evidence.`,
2228
- buildTools: (store) => buildTraceToolsForGroup("all", store),
2229
- subqueries: {
2230
- maxCalls: 8,
2231
- maxParallel: 4
759
+ Do NOT report a finding if the agent caught and corrected the false belief in the same turn. Reserve poisoning for cases where the false belief shaped downstream action.`,
760
+ toolGroup: "all",
761
+ limits: {
762
+ maxLlmCalls: 8,
763
+ maxIterations: 20,
764
+ maxToolCalls: 64
2232
765
  },
2233
- maxTurns: 20,
2234
- minimumEvidenceCitations: 2,
2235
- cost: { kind: "llm" }
766
+ minimumEvidenceCitations: 2
2236
767
  };
2237
768
  //#endregion
2238
769
  //#region src/analyst/kinds/index.ts
@@ -3820,20 +2351,14 @@ function selectPriorFindings(source, analystId) {
3820
2351
  }
3821
2352
  //#endregion
3822
2353
  //#region src/analyst/default-registry.ts
3823
- function buildDefaultAnalystRegistry(opts = {}) {
3824
- const registry = new AnalystRegistry(opts.registry);
3825
- if (opts.includeBehavioral !== false) registry.register(behavioralAnalyst(opts.behavioral));
3826
- if (opts.ai) {
3827
- const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
3828
- for (const spec of kinds) registry.register(createTraceAnalystKind(spec, {
3829
- ai: opts.ai,
3830
- model: opts.model,
3831
- aiIdentity: opts.aiIdentity
3832
- }));
3833
- }
2354
+ function buildDefaultAnalystRegistry(options = {}) {
2355
+ if (options.definitions && !options.engine) throw new TypeError("buildDefaultAnalystRegistry: definitions require an engine — a definition cannot run without one");
2356
+ const registry = new AnalystRegistry(options.registry);
2357
+ if (options.includeBehavioral !== false) registry.register(behavioralAnalyst(options.behavioral));
2358
+ if (options.engine) for (const definition of options.definitions ?? DEFAULT_TRACE_ANALYST_KINDS) registry.register(createTraceAnalyst(definition, { engine: options.engine }));
3834
2359
  return registry;
3835
2360
  }
3836
2361
  //#endregion
3837
- export { parseRawFinding as A, parseFindingSubject as B, renderUpstreamFindings as C, RawAnalystEvidenceSchema as D, RAW_FINDING_SCHEMA_PROMPT as E, FINDING_SUBJECT_KINDS as F, createChatClient as G, behavioralAnalyst as H, FINDING_SUBJECT_SYNTAX as I, createAnalystAi as K, FindingSubjectStringSchema as L, coerceToFindingRows as M, stripCodeFences as N, RawAnalystFindingSchema as O, FINDING_SUBJECT_GRAMMAR_PROMPT as P, KIND_EXPECTED_SUBJECTS as R, renderPriorFindings as S, ANALYST_SEVERITIES as T, deriveEfficiencyFindings as U, renderFindingSubject as V, computeTraceMetrics as W, buildTraceToolsForGroup as _, analystFindingDigest as a, emitControlIntegrityFindings as b, completedAnalystReviewQuality as c, validateAnalystReviewDecisions as d, DEFAULT_TRACE_ANALYST_KINDS as f, FAILURE_MODE_KIND_SPEC as g, IMPROVEMENT_KIND_SPEC as h, assertExactRegistryRunOpts as i, coerceJson as j, evidenceRefsFromRawFinding as k, readAnalystReview as l, KNOWLEDGE_GAP_KIND_SPEC as m, AnalystRegistry as n, analystRunDigest as o, KNOWLEDGE_POISONING_KIND_SPEC as p, ExactAnalystRunExecutionError as r, assertUniqueFindingIds as s, buildDefaultAnalystRegistry as t, snapshotAnalystRun as u, CONTROL_INTEGRITY_ANALYST as v, structureFindings as w, createTraceAnalystKind as x, ControlIntegrityAnalyst as y, findingSubjectGrammarPromptFor as z };
2362
+ export { createChatClient as C, computeTraceMetrics as S, CONTROL_INTEGRITY_ANALYST as _, analystFindingDigest as a, behavioralAnalyst as b, completedAnalystReviewQuality as c, validateAnalystReviewDecisions as d, DEFAULT_TRACE_ANALYST_KINDS as f, FAILURE_MODE_KIND_SPEC as g, IMPROVEMENT_KIND_SPEC as h, assertExactRegistryRunOpts as i, readAnalystReview as l, KNOWLEDGE_GAP_KIND_SPEC as m, AnalystRegistry as n, analystRunDigest as o, KNOWLEDGE_POISONING_KIND_SPEC as p, ExactAnalystRunExecutionError as r, assertUniqueFindingIds as s, buildDefaultAnalystRegistry as t, snapshotAnalystRun as u, ControlIntegrityAnalyst as v, deriveEfficiencyFindings as x, emitControlIntegrityFindings as y };
3838
2363
 
3839
- //# sourceMappingURL=default-registry-lp5R0lve.js.map
2364
+ //# sourceMappingURL=default-registry-BgJJItGr.js.map