@tangle-network/agent-eval 0.138.0 → 0.139.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -1
- package/dist/analyst/index.d.ts +41 -94
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +9 -24
- package/dist/analyst/index.js.map +1 -1
- package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
- package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
- package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
- package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
- package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
- package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
- package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
- package/dist/campaign-DrS6_hLd.js.map +1 -0
- package/dist/canonical-D011XM8r.js +86 -0
- package/dist/canonical-D011XM8r.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
- package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
- package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
- package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
- package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
- package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
- package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
- package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
- package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
- package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
- package/dist/default-registry-BgJJItGr.js.map +1 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
- package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
- package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
- package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
- package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
- package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
- package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
- package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
- package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
- package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
- package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
- package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
- package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
- package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
- package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
- package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
- package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
- package/dist/index.d.ts +35 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +51 -176
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
- package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
- package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
- package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
- package/dist/kind-factory-CFxA0JQX.js +2133 -0
- package/dist/kind-factory-CFxA0JQX.js.map +1 -0
- package/dist/ledger-core/index.js +2 -1
- package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
- package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
- package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
- package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
- package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
- package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
- package/dist/replay-DjG4IG60.d.ts.map +1 -0
- package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
- package/dist/replay-SA4OB7O7.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
- package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
- package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
- package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
- package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
- package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
- package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
- package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
- package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
- package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
- package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
- package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
- package/dist/single-run-lock-BTTtPZ9N.js +989 -0
- package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
- package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
- package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
- package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
- package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
- package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
- package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
- package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
- package/dist/store-otlp-DX4fGIcf.js +757 -0
- package/dist/store-otlp-DX4fGIcf.js.map +1 -0
- package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
- package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
- package/dist/tool-groups-CdYq22lX.d.ts +258 -0
- package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
- package/dist/traces.d.ts +7 -6
- package/dist/traces.js +5 -5
- package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
- package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
- package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
- package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
- package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
- package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
- package/dist/usage-receipt-CgxMEBZq.js +134 -0
- package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +170 -484
- package/package.json +1 -2
- package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
- package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
- package/dist/benchmark-command-CMqVqReF.js.map +0 -1
- package/dist/campaign-BIBS-NHV.js.map +0 -1
- package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
- package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
- package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
- package/dist/default-registry-PUhIVRWz.d.ts +0 -215
- package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
- package/dist/default-registry-lp5R0lve.js.map +0 -1
- package/dist/ledger-core-C0Yx1I14.js.map +0 -1
- package/dist/registry-C4yJTza7.d.ts +0 -178
- package/dist/registry-C4yJTza7.d.ts.map +0 -1
- package/dist/replay-Cb-4Vf0k.js.map +0 -1
- package/dist/replay-DbIYwso6.d.ts.map +0 -1
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
- package/dist/single-run-lock-D_bS5xhj.js +0 -318
- package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
- package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
- package/dist/store-otlp-BenKynPE.js +0 -1688
- package/dist/store-otlp-BenKynPE.js.map +0 -1
- package/dist/tools-DZGdROtG.js +0 -255
- package/dist/tools-DZGdROtG.js.map +0 -1
|
@@ -0,0 +1,2133 @@
|
|
|
1
|
+
import { a as LimitExceededError, c as ValidationError, n as CaptureIntegrityError, o as NotFoundError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
2
|
+
import { i as CostLedger } from "./cost-ledger-CZ9diLxY.js";
|
|
3
|
+
import { INPUT_VALUE, LLM_MODEL_ATTR_KEYS, OUTPUT_VALUE, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS } from "./trace-attributes.js";
|
|
4
|
+
import { i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding } from "./usage-receipt-CgxMEBZq.js";
|
|
5
|
+
import { i as hashCanonical, r as canonicalString } from "./canonical-D011XM8r.js";
|
|
6
|
+
import { z } from "zod";
|
|
7
|
+
import { RE2JS } from "re2js";
|
|
8
|
+
//#region src/trace/otlp-attributes.ts
|
|
9
|
+
/** Canonical OpenInference-over-OTLP attribute vocabulary used at the trace boundary. */
|
|
10
|
+
const TOOL_SPAN_ATTRIBUTE_KEYS = [
|
|
11
|
+
TOOL_NAME,
|
|
12
|
+
TOOL_ARGS_CAPTURED,
|
|
13
|
+
TOOL_LATENCY_MS,
|
|
14
|
+
INPUT_VALUE,
|
|
15
|
+
OUTPUT_VALUE
|
|
16
|
+
];
|
|
17
|
+
const EXPLICIT_SPAN_ROLES = /* @__PURE__ */ new Set([
|
|
18
|
+
"AGENT",
|
|
19
|
+
"CHAIN",
|
|
20
|
+
"EVALUATOR",
|
|
21
|
+
"GUARDRAIL",
|
|
22
|
+
"LLM",
|
|
23
|
+
"SPAN",
|
|
24
|
+
"TOOL"
|
|
25
|
+
]);
|
|
26
|
+
/**
|
|
27
|
+
* Classify a span once for both measurement and error accounting.
|
|
28
|
+
* An explicit OpenInference kind wins; untyped spans use the same tool and
|
|
29
|
+
* model signals in online and offline intake.
|
|
30
|
+
*/
|
|
31
|
+
function classifyOtlpSpanRole(input) {
|
|
32
|
+
const explicitKind = input.kind ?? firstStringAttribute(input.attributes, SPAN_KIND_ATTR_KEYS);
|
|
33
|
+
if (explicitKind) {
|
|
34
|
+
const normalized = explicitKind.toUpperCase();
|
|
35
|
+
if (EXPLICIT_SPAN_ROLES.has(normalized)) return normalized;
|
|
36
|
+
}
|
|
37
|
+
if (firstStringAttribute(input.attributes, TOOL_NAME_ATTR_KEYS) !== void 0 || /^(?:function|tool)[.:/]/i.test(input.name)) return "TOOL";
|
|
38
|
+
const spanType = input.attributes["span.type"];
|
|
39
|
+
if (typeof spanType === "string" && spanType.toLowerCase() === "llm_request" || /(?:^|[.:/_-])(?:chat[._-]?completions?|llm)(?:$|[.:/_-])/i.test(input.name) || firstStringAttribute(input.attributes, LLM_MODEL_ATTR_KEYS) !== void 0 || typeof input.attributes["gen_ai.operation.name"] === "string") return "LLM";
|
|
40
|
+
return "UNKNOWN";
|
|
41
|
+
}
|
|
42
|
+
function isOtlpModelCall(input) {
|
|
43
|
+
return classifyOtlpSpanRole(input) === "LLM";
|
|
44
|
+
}
|
|
45
|
+
function toolSpanOtlpAttributes(span) {
|
|
46
|
+
const argsCaptured = span.argsCaptured !== false;
|
|
47
|
+
const attributes = {
|
|
48
|
+
[TOOL_NAME]: span.toolName,
|
|
49
|
+
[TOOL_ARGS_CAPTURED]: argsCaptured
|
|
50
|
+
};
|
|
51
|
+
if (span.latencyMs !== void 0) attributes[TOOL_LATENCY_MS] = span.latencyMs;
|
|
52
|
+
if (argsCaptured) attributes[INPUT_VALUE] = stringifyTraceValue(span.args);
|
|
53
|
+
if (span.result !== void 0) attributes[OUTPUT_VALUE] = stringifyTraceValue(span.result);
|
|
54
|
+
return attributes;
|
|
55
|
+
}
|
|
56
|
+
function applyToolSpanOtlpAttributes(attributes, span) {
|
|
57
|
+
for (const key of TOOL_SPAN_ATTRIBUTE_KEYS) delete attributes[key];
|
|
58
|
+
Object.assign(attributes, toolSpanOtlpAttributes(span));
|
|
59
|
+
}
|
|
60
|
+
function traceSpanKindToOpenInferenceKind(kind) {
|
|
61
|
+
switch (kind) {
|
|
62
|
+
case "llm": return "LLM";
|
|
63
|
+
case "tool": return "TOOL";
|
|
64
|
+
case "retrieval": return "CHAIN";
|
|
65
|
+
case "judge": return "EVALUATOR";
|
|
66
|
+
case "sandbox": return "CHAIN";
|
|
67
|
+
case "agent": return "AGENT";
|
|
68
|
+
default: return "SPAN";
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
function firstStringAttribute(attributes, keys) {
|
|
72
|
+
for (const key of keys) {
|
|
73
|
+
const value = attributes[key];
|
|
74
|
+
if (typeof value === "string" && value.length > 0) return value;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
function stringifyTraceValue(value) {
|
|
78
|
+
if (value === void 0) return "null";
|
|
79
|
+
if (typeof value === "string") return value;
|
|
80
|
+
try {
|
|
81
|
+
return JSON.stringify(value) ?? String(value);
|
|
82
|
+
} catch {
|
|
83
|
+
return String(value);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
//#endregion
|
|
87
|
+
//#region src/trace-analyst/otlp-span.ts
|
|
88
|
+
/**
|
|
89
|
+
* Canonical OTLP-flat-line readers shared by every consumer of the
|
|
90
|
+
* OTLP-JSONL wire shape (one OTLP span per line; the form
|
|
91
|
+
* `flattenOtlpExportToNdjson` produces and the form AppWorld / HALO
|
|
92
|
+
* emit via their OpenInference OTLP exporter).
|
|
93
|
+
*
|
|
94
|
+
* `OtlpFileTraceStore` indexes spans with these; `otlpToRunRecords`
|
|
95
|
+
* aggregates spans into `RunRecord`s with the same readers. One parser,
|
|
96
|
+
* one vocabulary — a divergence between the analyst's view of a trace and
|
|
97
|
+
* the RunRecord projected from it is a class of bug this consolidation
|
|
98
|
+
* removes by construction.
|
|
99
|
+
*
|
|
100
|
+
* Vocabulary. The readers understand BOTH dialects that appear in the
|
|
101
|
+
* wild:
|
|
102
|
+
* - the substrate's own `llm.*` / `tool.*` / `span.kind` attributes
|
|
103
|
+
* (`flattenSpanAttributes` in `trace/otel.ts`), and
|
|
104
|
+
* - the OpenInference / inference-export attributes AppWorld / HALO
|
|
105
|
+
* emit (`openinference.span.kind`, `inference.observation_kind`,
|
|
106
|
+
* `inference.llm.input_tokens`, `llm.token_count.prompt`, …).
|
|
107
|
+
*
|
|
108
|
+
* Pure, no I/O.
|
|
109
|
+
*/
|
|
110
|
+
/**
|
|
111
|
+
* Project one parsed OTLP-JSONL object to `ProjectedOtlpSpan`, or `null`
|
|
112
|
+
* when the line is missing the mandatory `trace_id` + `span_id`.
|
|
113
|
+
*/
|
|
114
|
+
function projectOtlpFlatLine(raw) {
|
|
115
|
+
const trace_id = stringField(raw, "trace_id") ?? stringField(raw, "traceId");
|
|
116
|
+
const span_id = stringField(raw, "span_id") ?? stringField(raw, "spanId");
|
|
117
|
+
if (!trace_id || !span_id) return null;
|
|
118
|
+
const parent_id = normalizeParentSpanId(trace_id, span_id, stringField(raw, "parent_span_id") ?? stringField(raw, "parentSpanId") ?? null);
|
|
119
|
+
const name = stringField(raw, "name") ?? "unknown";
|
|
120
|
+
const start_time = stringField(raw, "start_time") ?? stringField(raw, "startTime") ?? "";
|
|
121
|
+
const end_time = stringField(raw, "end_time") ?? stringField(raw, "endTime") ?? start_time;
|
|
122
|
+
const status = readOtlpStatus(raw);
|
|
123
|
+
const attributes = extractOtlpAttributes(raw);
|
|
124
|
+
const service_name = asString(attributes["service.name"]) ?? asString(attributes["resource.attributes.service.name"]) ?? null;
|
|
125
|
+
const agent_name = asString(attributes["agent.name"]) ?? asString(attributes["inference.agent.name"]) ?? asString(attributes["inference.agent_name"]) ?? null;
|
|
126
|
+
const model_name = firstStringAttr(attributes, LLM_MODEL_ATTR_KEYS);
|
|
127
|
+
const tool_name = firstStringAttr(attributes, TOOL_NAME_ATTR_KEYS);
|
|
128
|
+
const kind = inferOtlpKind(attributes);
|
|
129
|
+
let duration_ms = 0;
|
|
130
|
+
if (start_time && end_time) {
|
|
131
|
+
const a = spanEpochMillis(start_time);
|
|
132
|
+
const b = spanEpochMillis(end_time);
|
|
133
|
+
if (a !== null && b !== null) duration_ms = Math.max(0, b - a);
|
|
134
|
+
}
|
|
135
|
+
return {
|
|
136
|
+
trace_id,
|
|
137
|
+
span_id,
|
|
138
|
+
parent_span_id: parent_id && parent_id.length > 0 ? parent_id : null,
|
|
139
|
+
name,
|
|
140
|
+
kind,
|
|
141
|
+
start_time,
|
|
142
|
+
end_time,
|
|
143
|
+
duration_ms,
|
|
144
|
+
status: status.code,
|
|
145
|
+
status_message: status.message,
|
|
146
|
+
service_name,
|
|
147
|
+
agent_name,
|
|
148
|
+
model_name,
|
|
149
|
+
tool_name,
|
|
150
|
+
attributes
|
|
151
|
+
};
|
|
152
|
+
}
|
|
153
|
+
function normalizeParentSpanId(traceId, spanId, parentId) {
|
|
154
|
+
if (!parentId) return null;
|
|
155
|
+
const prefix = `${traceId}:`;
|
|
156
|
+
return spanId.startsWith(prefix) && !parentId.startsWith(prefix) ? `${prefix}${parentId}` : parentId;
|
|
157
|
+
}
|
|
158
|
+
function readOtlpStatus(raw) {
|
|
159
|
+
const status = raw.status;
|
|
160
|
+
if (status && typeof status === "object" && !Array.isArray(status)) {
|
|
161
|
+
const codeRaw = status.code;
|
|
162
|
+
const code = codeRaw === "STATUS_CODE_OK" || codeRaw === "OK" ? "OK" : codeRaw === "STATUS_CODE_ERROR" || codeRaw === "ERROR" ? "ERROR" : "UNSET";
|
|
163
|
+
const messageRaw = status.message;
|
|
164
|
+
return {
|
|
165
|
+
code,
|
|
166
|
+
message: typeof messageRaw === "string" && messageRaw.length > 0 ? messageRaw : void 0
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
return {
|
|
170
|
+
code: "UNSET",
|
|
171
|
+
message: void 0
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
function inferOtlpKind(attrs) {
|
|
175
|
+
const opik = firstStringAttr(attrs, SPAN_KIND_ATTR_KEYS);
|
|
176
|
+
if (opik) {
|
|
177
|
+
const upper = opik.toUpperCase();
|
|
178
|
+
if (upper === "AGENT" || upper === "LLM" || upper === "TOOL" || upper === "CHAIN" || upper === "EVALUATOR" || upper === "GUARDRAIL" || upper === "SPAN") return upper;
|
|
179
|
+
}
|
|
180
|
+
return "UNKNOWN";
|
|
181
|
+
}
|
|
182
|
+
/**
|
|
183
|
+
* Flatten OTLP `attributes` + `resource.attributes` into a single
|
|
184
|
+
* dotted-key map. Span attributes override resource attributes when keys
|
|
185
|
+
* overlap. Nested objects/arrays are preserved as-is.
|
|
186
|
+
*/
|
|
187
|
+
function extractOtlpAttributes(raw) {
|
|
188
|
+
const out = {};
|
|
189
|
+
const resource = raw.resource;
|
|
190
|
+
if (resource && typeof resource === "object" && !Array.isArray(resource)) {
|
|
191
|
+
const ra = resource.attributes;
|
|
192
|
+
if (ra && typeof ra === "object" && !Array.isArray(ra)) for (const [k, v] of Object.entries(ra)) out[k] = v;
|
|
193
|
+
}
|
|
194
|
+
const spanAttrs = raw.attributes;
|
|
195
|
+
if (spanAttrs && typeof spanAttrs === "object" && !Array.isArray(spanAttrs)) for (const [k, v] of Object.entries(spanAttrs)) out[k] = v;
|
|
196
|
+
return out;
|
|
197
|
+
}
|
|
198
|
+
function stringField(raw, key) {
|
|
199
|
+
const v = raw[key];
|
|
200
|
+
return typeof v === "string" ? v : void 0;
|
|
201
|
+
}
|
|
202
|
+
function asString(v) {
|
|
203
|
+
return typeof v === "string" && v.length > 0 ? v : null;
|
|
204
|
+
}
|
|
205
|
+
/** First non-empty string value across a list of candidate attribute keys. */
|
|
206
|
+
function firstStringAttr(attrs, keys) {
|
|
207
|
+
for (const k of keys) {
|
|
208
|
+
const s = asString(attrs[k]);
|
|
209
|
+
if (s !== null) return s;
|
|
210
|
+
}
|
|
211
|
+
return null;
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* Parse a span timestamp to epoch millis, or null when empty/unparseable. The
|
|
215
|
+
* OTLP readers accept BOTH ISO-8601 and epoch-millis-string dialects, so raw
|
|
216
|
+
* string comparison (`<`, `localeCompare`) mis-orders across dialects and
|
|
217
|
+
* `Date.parse` returns NaN for a bare epoch-millis string.
|
|
218
|
+
*/
|
|
219
|
+
function spanEpochMillis(ts) {
|
|
220
|
+
if (!ts) return null;
|
|
221
|
+
if (/^\d+$/.test(ts)) return Number(ts);
|
|
222
|
+
const n = Date.parse(ts);
|
|
223
|
+
return Number.isNaN(n) ? null : n;
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* Order comparator for span timestamps across mixed ISO/epoch dialects.
|
|
227
|
+
* Unparseable timestamps sort as epoch 0 (earliest), never NaN (which would
|
|
228
|
+
* make the sort non-deterministic).
|
|
229
|
+
*/
|
|
230
|
+
function compareSpanTime(a, b) {
|
|
231
|
+
return (spanEpochMillis(a) ?? 0) - (spanEpochMillis(b) ?? 0);
|
|
232
|
+
}
|
|
233
|
+
//#endregion
|
|
234
|
+
//#region src/analyst/engine.ts
|
|
235
|
+
const DEFAULT_TRACE_ANALYST_LIMITS = {
|
|
236
|
+
maxIterations: 12,
|
|
237
|
+
maxLlmCalls: 8,
|
|
238
|
+
maxToolCalls: 48,
|
|
239
|
+
maxOutputChars: 1e4
|
|
240
|
+
};
|
|
241
|
+
function resolveTraceAnalystLimits(limits) {
|
|
242
|
+
const resolved = {
|
|
243
|
+
...DEFAULT_TRACE_ANALYST_LIMITS,
|
|
244
|
+
...limits
|
|
245
|
+
};
|
|
246
|
+
for (const [name, value] of Object.entries(resolved)) if (!Number.isSafeInteger(value) || value <= 0) throw new TypeError(`trace analyst ${name} must be a positive safe integer`);
|
|
247
|
+
return resolved;
|
|
248
|
+
}
|
|
249
|
+
//#endregion
|
|
250
|
+
//#region src/ledger-core/deep-freeze.ts
|
|
251
|
+
/** Freeze a detached canonical-JSON graph. Canonicalization has already ruled out cycles.
|
|
252
|
+
*
|
|
253
|
+
* Lives outside canonical.ts so the analyst-benchmark implementation digest,
|
|
254
|
+
* which covers canonical.ts, stays bound to the published benchmark evidence. */
|
|
255
|
+
function deepFreezeCanonicalJson(value) {
|
|
256
|
+
if (value && typeof value === "object" && !Object.isFrozen(value)) {
|
|
257
|
+
Object.freeze(value);
|
|
258
|
+
for (const nested of Object.values(value)) deepFreezeCanonicalJson(nested);
|
|
259
|
+
}
|
|
260
|
+
return value;
|
|
261
|
+
}
|
|
262
|
+
//#endregion
|
|
263
|
+
//#region src/analyst/exact-types.ts
|
|
264
|
+
/** Canonical identity for any live component admitted to an exact run. */
|
|
265
|
+
function snapshotExactExecutionComponentIdentity(value, context) {
|
|
266
|
+
let detached;
|
|
267
|
+
try {
|
|
268
|
+
detached = JSON.parse(canonicalString(value));
|
|
269
|
+
} catch (cause) {
|
|
270
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
271
|
+
}
|
|
272
|
+
const parsed = componentIdentitySchema.safeParse(detached);
|
|
273
|
+
if (!parsed.success) throw new TypeError(`${context} requires non-empty id/version and object config`);
|
|
274
|
+
return deepFreezeCanonicalJson({
|
|
275
|
+
id: parsed.data.id,
|
|
276
|
+
version: parsed.data.version,
|
|
277
|
+
config_digest: hashCanonical(parsed.data.config)
|
|
278
|
+
});
|
|
279
|
+
}
|
|
280
|
+
const nonEmptyString = z.string().min(1);
|
|
281
|
+
const digest = z.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
282
|
+
const finiteNonnegative = z.number().finite().nonnegative();
|
|
283
|
+
const nonnegativeSafeInteger = z.number().int().min(0).max(Number.MAX_SAFE_INTEGER);
|
|
284
|
+
const positiveTimeout = z.number().int().positive().max(2147483647);
|
|
285
|
+
const componentSnapshotSchema = z.strictObject({
|
|
286
|
+
id: nonEmptyString,
|
|
287
|
+
version: nonEmptyString,
|
|
288
|
+
config_digest: digest
|
|
289
|
+
});
|
|
290
|
+
const componentIdentitySchema = z.strictObject({
|
|
291
|
+
id: nonEmptyString,
|
|
292
|
+
version: nonEmptyString,
|
|
293
|
+
config: z.record(z.string(), z.unknown())
|
|
294
|
+
});
|
|
295
|
+
const deterministicCostSchema = z.strictObject({
|
|
296
|
+
kind: z.literal("deterministic"),
|
|
297
|
+
est_usd_per_run: finiteNonnegative.optional(),
|
|
298
|
+
models: z.array(nonEmptyString).optional()
|
|
299
|
+
});
|
|
300
|
+
const llmCostSchema = z.strictObject({
|
|
301
|
+
kind: z.literal("llm"),
|
|
302
|
+
est_usd_per_run: finiteNonnegative.optional(),
|
|
303
|
+
models: z.array(nonEmptyString).optional(),
|
|
304
|
+
settlement_timeout_ms: nonnegativeSafeInteger.optional()
|
|
305
|
+
});
|
|
306
|
+
const requirementsSchema = z.strictObject({
|
|
307
|
+
min_shots: nonnegativeSafeInteger.optional(),
|
|
308
|
+
capabilities: z.array(nonEmptyString).optional()
|
|
309
|
+
}).nullable();
|
|
310
|
+
const analystSnapshotSchema = z.strictObject({
|
|
311
|
+
id: nonEmptyString,
|
|
312
|
+
version: nonEmptyString,
|
|
313
|
+
input_kind: z.enum([
|
|
314
|
+
"trace-store",
|
|
315
|
+
"artifact-dir",
|
|
316
|
+
"run-record",
|
|
317
|
+
"judge-input",
|
|
318
|
+
"custom"
|
|
319
|
+
]),
|
|
320
|
+
cost: z.discriminatedUnion("kind", [deterministicCostSchema, llmCostSchema]),
|
|
321
|
+
requirements: requirementsSchema,
|
|
322
|
+
execution_config_digest: digest
|
|
323
|
+
});
|
|
324
|
+
const allocationsSchema = z.record(nonEmptyString, z.union([finiteNonnegative, z.null()]));
|
|
325
|
+
const weightsSchema = z.record(nonEmptyString, finiteNonnegative);
|
|
326
|
+
const budgetSnapshotSchema = z.discriminatedUnion("kind", [
|
|
327
|
+
z.strictObject({ kind: z.literal("none") }),
|
|
328
|
+
z.strictObject({
|
|
329
|
+
kind: z.literal("equal"),
|
|
330
|
+
total_usd: finiteNonnegative,
|
|
331
|
+
allocations_usd: allocationsSchema
|
|
332
|
+
}),
|
|
333
|
+
z.strictObject({
|
|
334
|
+
kind: z.literal("weighted"),
|
|
335
|
+
total_usd: finiteNonnegative,
|
|
336
|
+
weights: weightsSchema,
|
|
337
|
+
allocations_usd: allocationsSchema
|
|
338
|
+
})
|
|
339
|
+
]);
|
|
340
|
+
const priorFindingsSchema = z.discriminatedUnion("kind", [
|
|
341
|
+
z.strictObject({ kind: z.literal("none") }),
|
|
342
|
+
z.strictObject({
|
|
343
|
+
kind: z.literal("ordered"),
|
|
344
|
+
count: nonnegativeSafeInteger,
|
|
345
|
+
digest
|
|
346
|
+
}),
|
|
347
|
+
z.strictObject({
|
|
348
|
+
kind: z.literal("by_analyst"),
|
|
349
|
+
keys: z.array(nonEmptyString),
|
|
350
|
+
count: nonnegativeSafeInteger,
|
|
351
|
+
digest
|
|
352
|
+
})
|
|
353
|
+
]);
|
|
354
|
+
const exactRunPolicySchema = z.strictObject({
|
|
355
|
+
budget: budgetSnapshotSchema,
|
|
356
|
+
total_timeout_ms: positiveTimeout.nullable(),
|
|
357
|
+
signal_provided: z.boolean(),
|
|
358
|
+
cost_ledger: componentSnapshotSchema.nullable(),
|
|
359
|
+
cost_phase: nonEmptyString.nullable(),
|
|
360
|
+
tags: z.record(z.string(), z.string()).nullable(),
|
|
361
|
+
prior_findings: priorFindingsSchema,
|
|
362
|
+
chain_findings: z.boolean(),
|
|
363
|
+
missing_input_mode: z.enum(["skip", "abort"]),
|
|
364
|
+
registry_hooks: componentSnapshotSchema.nullable(),
|
|
365
|
+
registry_chat: componentSnapshotSchema.nullable()
|
|
366
|
+
});
|
|
367
|
+
const exactExecutionPlanSchema = z.strictObject({
|
|
368
|
+
schema_version: z.literal("1.0.0"),
|
|
369
|
+
analysts: z.array(analystSnapshotSchema).min(1),
|
|
370
|
+
policy: exactRunPolicySchema,
|
|
371
|
+
digest
|
|
372
|
+
}).superRefine((plan, context) => {
|
|
373
|
+
const issue = (path, message) => context.addIssue({
|
|
374
|
+
code: "custom",
|
|
375
|
+
path,
|
|
376
|
+
message
|
|
377
|
+
});
|
|
378
|
+
const analystIds = plan.analysts.map((analyst) => analyst.id);
|
|
379
|
+
if (new Set(analystIds).size !== analystIds.length) issue(["analysts"], "analyst ids must be unique");
|
|
380
|
+
if (plan.policy.cost_ledger === null && plan.policy.cost_phase !== null) issue(["policy", "cost_phase"], "cost phase requires a cost ledger");
|
|
381
|
+
if (plan.policy.prior_findings.kind === "by_analyst" && plan.policy.prior_findings.keys.some((key, index, keys) => index > 0 && key <= keys[index - 1])) issue([
|
|
382
|
+
"policy",
|
|
383
|
+
"prior_findings",
|
|
384
|
+
"keys"
|
|
385
|
+
], "keys must be sorted and unique");
|
|
386
|
+
const budget = plan.policy.budget;
|
|
387
|
+
if (budget.kind === "none") return;
|
|
388
|
+
const allocationIds = Object.keys(budget.allocations_usd).sort();
|
|
389
|
+
const selectedIds = [...analystIds].sort();
|
|
390
|
+
if (allocationIds.length !== selectedIds.length || allocationIds.some((id, index) => id !== selectedIds[index])) {
|
|
391
|
+
issue([
|
|
392
|
+
"policy",
|
|
393
|
+
"budget",
|
|
394
|
+
"allocations_usd"
|
|
395
|
+
], "allocations must name every analyst and no others");
|
|
396
|
+
return;
|
|
397
|
+
}
|
|
398
|
+
const runnableIds = analystIds.filter((id) => budget.allocations_usd[id] !== null);
|
|
399
|
+
const epsilon = Math.max(1, budget.total_usd) * Number.EPSILON * 8;
|
|
400
|
+
if (runnableIds.length === 0) return;
|
|
401
|
+
if (budget.kind === "weighted") {
|
|
402
|
+
const weightIds = Object.keys(budget.weights).sort();
|
|
403
|
+
if (weightIds.length !== selectedIds.length || weightIds.some((id, index) => id !== selectedIds[index])) {
|
|
404
|
+
issue([
|
|
405
|
+
"policy",
|
|
406
|
+
"budget",
|
|
407
|
+
"weights"
|
|
408
|
+
], "weights must name every analyst and no others");
|
|
409
|
+
return;
|
|
410
|
+
}
|
|
411
|
+
const totalWeight = runnableIds.reduce((sum, id) => sum + (budget.weights[id] ?? 0), 0);
|
|
412
|
+
if (totalWeight === 0) {
|
|
413
|
+
issue([
|
|
414
|
+
"policy",
|
|
415
|
+
"budget",
|
|
416
|
+
"weights"
|
|
417
|
+
], "runnable analysts must have positive total weight");
|
|
418
|
+
return;
|
|
419
|
+
}
|
|
420
|
+
for (const id of runnableIds) {
|
|
421
|
+
const expected = budget.total_usd * (budget.weights[id] ?? 0) / totalWeight;
|
|
422
|
+
if (Math.abs((budget.allocations_usd[id] ?? 0) - expected) > epsilon) issue([
|
|
423
|
+
"policy",
|
|
424
|
+
"budget",
|
|
425
|
+
"allocations_usd",
|
|
426
|
+
id
|
|
427
|
+
], "allocation does not match the weighted policy");
|
|
428
|
+
}
|
|
429
|
+
return;
|
|
430
|
+
}
|
|
431
|
+
const expected = budget.total_usd / runnableIds.length;
|
|
432
|
+
for (const id of runnableIds) if (Math.abs((budget.allocations_usd[id] ?? 0) - expected) > epsilon) issue([
|
|
433
|
+
"policy",
|
|
434
|
+
"budget",
|
|
435
|
+
"allocations_usd",
|
|
436
|
+
id
|
|
437
|
+
], "allocation does not match the equal policy");
|
|
438
|
+
});
|
|
439
|
+
/**
|
|
440
|
+
* Canonicalize and validate the one exact-plan representation shared by execution and archival.
|
|
441
|
+
* Unknown fields fail at every level; the returned graph is detached and deeply frozen.
|
|
442
|
+
*/
|
|
443
|
+
function snapshotExactExecutionPlan(value, context = "exact analyst execution plan") {
|
|
444
|
+
let detached;
|
|
445
|
+
try {
|
|
446
|
+
detached = JSON.parse(canonicalString(value));
|
|
447
|
+
} catch (cause) {
|
|
448
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
449
|
+
}
|
|
450
|
+
const parsed = exactExecutionPlanSchema.safeParse(detached);
|
|
451
|
+
if (!parsed.success) {
|
|
452
|
+
const issue = parsed.error.issues[0];
|
|
453
|
+
const path = issue?.path.length ? ` ${issue.path.join(".")}` : "";
|
|
454
|
+
throw new TypeError(`${context}${path}: ${issue?.message ?? "is invalid"}`);
|
|
455
|
+
}
|
|
456
|
+
const expectedDigest = hashCanonical({
|
|
457
|
+
schema_version: parsed.data.schema_version,
|
|
458
|
+
analysts: parsed.data.analysts,
|
|
459
|
+
policy: parsed.data.policy
|
|
460
|
+
});
|
|
461
|
+
if (parsed.data.digest !== expectedDigest) throw new TypeError(`${context} digest does not match its content`);
|
|
462
|
+
return deepFreezeCanonicalJson(parsed.data);
|
|
463
|
+
}
|
|
464
|
+
//#endregion
|
|
465
|
+
//#region src/analyst/finding-subject.ts
|
|
466
|
+
/**
|
|
467
|
+
* Typed `FindingSubject` — the canonical grammar every analyst kind emits.
|
|
468
|
+
*
|
|
469
|
+
* Background: kind actor prompts have always documented a subject grammar
|
|
470
|
+
* (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`) but the
|
|
471
|
+
* LLM was unconstrained — it could emit `subject: "fix the prompt"`
|
|
472
|
+
* (prose) and downstream adapters routed on `startsWith(...)` would
|
|
473
|
+
* silently skip it. Every per-vertical `ImprovementAdapter` had a
|
|
474
|
+
* routing table that mostly caught nothing.
|
|
475
|
+
*
|
|
476
|
+
* This module fixes that:
|
|
477
|
+
* - `parseFindingSubject(raw)` — returns the typed `FindingSubject`
|
|
478
|
+
* when `raw` matches the grammar, else `null`. Used at the
|
|
479
|
+
* `RawAnalystFindingSchema` boundary so malformed subjects are
|
|
480
|
+
* rejected loudly instead of silently lifted into the registry.
|
|
481
|
+
* - `FindingSubjectKind` — the union of valid locus categories. Each
|
|
482
|
+
* variant carries the typed components downstream adapters resolve
|
|
483
|
+
* against the agent's surface manifest (no string parsing in the
|
|
484
|
+
* adapter).
|
|
485
|
+
* - `FINDING_SUBJECT_GRAMMAR_PROMPT` — single source of truth for the
|
|
486
|
+
* grammar string embedded in kind actor prompts. Drift between
|
|
487
|
+
* prompt and parser is impossible if every kind imports this.
|
|
488
|
+
*
|
|
489
|
+
* The grammar is intentionally NARROW — only loci the substrate's
|
|
490
|
+
* default `ImprovementAdapter` / `KnowledgeAdapter` can act on. A
|
|
491
|
+
* finding with a subject outside this set fails the parser; the kind
|
|
492
|
+
* author either extends the grammar here (and adds adapter routing)
|
|
493
|
+
* or rephrases the prompt to map onto an existing variant.
|
|
494
|
+
*
|
|
495
|
+
* `failure-mode` is the one exception — its subjects are free-form
|
|
496
|
+
* cluster labels, not loci. The schema preserves them as
|
|
497
|
+
* `{ kind: 'cluster', label }` and the adapters skip them (cluster
|
|
498
|
+
* findings are evidence, not actionable mutations).
|
|
499
|
+
*/
|
|
500
|
+
const FINDING_SUBJECT_KINDS = [
|
|
501
|
+
"knowledge.wiki",
|
|
502
|
+
"knowledge.claim",
|
|
503
|
+
"knowledge.raw",
|
|
504
|
+
"knowledge.stale",
|
|
505
|
+
"system-prompt",
|
|
506
|
+
"skill",
|
|
507
|
+
"tool-doc",
|
|
508
|
+
"new-tool",
|
|
509
|
+
"mcp",
|
|
510
|
+
"hook",
|
|
511
|
+
"subagent",
|
|
512
|
+
"workflow",
|
|
513
|
+
"rollout-policy",
|
|
514
|
+
"agent-profile",
|
|
515
|
+
"code",
|
|
516
|
+
"rag",
|
|
517
|
+
"memory",
|
|
518
|
+
"scaffolding",
|
|
519
|
+
"output-schema",
|
|
520
|
+
"websearch.outdated",
|
|
521
|
+
"prior-run-summary",
|
|
522
|
+
"cluster"
|
|
523
|
+
];
|
|
524
|
+
/**
|
|
525
|
+
* Parse a raw subject string emitted by an analyst kind's actor.
|
|
526
|
+
*
|
|
527
|
+
* Returns the typed `FindingSubject` when `raw` matches the grammar,
|
|
528
|
+
* else `null`. Callers use the `null` return as a signal to either
|
|
529
|
+
* (a) reject the finding at parse time (kinds that emit typed loci —
|
|
530
|
+
* knowledge-gap, improvement, knowledge-poisoning) or (b) lift it as
|
|
531
|
+
* a cluster label (failure-mode).
|
|
532
|
+
*
|
|
533
|
+
* Slugs are constrained to `[a-z0-9-]+` (lowercase kebab) to keep file
|
|
534
|
+
* paths sane downstream. Topics / keys / sections allow any non-empty
|
|
535
|
+
* string (free-form for the LLM's voice) but get trimmed.
|
|
536
|
+
*
|
|
537
|
+
* Empty / whitespace-only inputs return `null`. `undefined` returns
|
|
538
|
+
* `null`. Both are surfaced by the caller as a rejected subject.
|
|
539
|
+
*/
|
|
540
|
+
function parseFindingSubject(raw) {
|
|
541
|
+
if (raw === null || raw === void 0) return null;
|
|
542
|
+
const trimmed = raw.trim();
|
|
543
|
+
if (trimmed.length === 0) return null;
|
|
544
|
+
const wiki = trimmed.match(/^agent-knowledge:wiki:([a-z0-9][a-z0-9-]*)(?:#([a-z0-9][a-z0-9-]*))?$/);
|
|
545
|
+
if (wiki) return {
|
|
546
|
+
kind: "knowledge.wiki",
|
|
547
|
+
slug: wiki[1],
|
|
548
|
+
...wiki[2] ? { heading: wiki[2] } : {}
|
|
549
|
+
};
|
|
550
|
+
const claim = trimmed.match(/^agent-knowledge:claim:(.+)$/);
|
|
551
|
+
if (claim && claim[1].trim().length > 0) return {
|
|
552
|
+
kind: "knowledge.claim",
|
|
553
|
+
topic: claim[1].trim()
|
|
554
|
+
};
|
|
555
|
+
const raw_ = trimmed.match(/^agent-knowledge:raw:(.+)$/);
|
|
556
|
+
if (raw_ && raw_[1].trim().length > 0) return {
|
|
557
|
+
kind: "knowledge.raw",
|
|
558
|
+
sourceId: raw_[1].trim()
|
|
559
|
+
};
|
|
560
|
+
const stale = trimmed.match(/^agent-knowledge:stale:([a-z0-9][a-z0-9-]*)$/);
|
|
561
|
+
if (stale) return {
|
|
562
|
+
kind: "knowledge.stale",
|
|
563
|
+
slug: stale[1]
|
|
564
|
+
};
|
|
565
|
+
const sp = trimmed.match(/^system-prompt:(.+)$/);
|
|
566
|
+
if (sp && sp[1].trim().length > 0) return {
|
|
567
|
+
kind: "system-prompt",
|
|
568
|
+
section: sp[1].trim()
|
|
569
|
+
};
|
|
570
|
+
const skill = trimmed.match(/^skill:([a-z0-9][a-z0-9_.-]*)$/);
|
|
571
|
+
if (skill) return {
|
|
572
|
+
kind: "skill",
|
|
573
|
+
name: skill[1]
|
|
574
|
+
};
|
|
575
|
+
const tdAspect = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*):(.+)$/);
|
|
576
|
+
if (tdAspect && tdAspect[2].trim().length > 0) return {
|
|
577
|
+
kind: "tool-doc",
|
|
578
|
+
tool: tdAspect[1],
|
|
579
|
+
aspect: tdAspect[2].trim()
|
|
580
|
+
};
|
|
581
|
+
const td = trimmed.match(/^tool-doc:([a-z0-9][a-z0-9_-]*)$/);
|
|
582
|
+
if (td) return {
|
|
583
|
+
kind: "tool-doc",
|
|
584
|
+
tool: td[1]
|
|
585
|
+
};
|
|
586
|
+
const nt = trimmed.match(/^new-tool:([a-z0-9][a-z0-9_-]*)$/);
|
|
587
|
+
if (nt) return {
|
|
588
|
+
kind: "new-tool",
|
|
589
|
+
name: nt[1]
|
|
590
|
+
};
|
|
591
|
+
const mcp = trimmed.match(/^mcp:([a-z0-9][a-z0-9_.-]*)(?::([a-z0-9][a-z0-9_.-]*))?$/);
|
|
592
|
+
if (mcp) return {
|
|
593
|
+
kind: "mcp",
|
|
594
|
+
server: mcp[1],
|
|
595
|
+
...mcp[2] ? { tool: mcp[2] } : {}
|
|
596
|
+
};
|
|
597
|
+
const hook = trimmed.match(/^hook:([a-z0-9][a-z0-9_.-]*)$/);
|
|
598
|
+
if (hook) return {
|
|
599
|
+
kind: "hook",
|
|
600
|
+
name: hook[1]
|
|
601
|
+
};
|
|
602
|
+
const subagent = trimmed.match(/^subagent:([a-z0-9][a-z0-9_.-]*)$/);
|
|
603
|
+
if (subagent) return {
|
|
604
|
+
kind: "subagent",
|
|
605
|
+
name: subagent[1]
|
|
606
|
+
};
|
|
607
|
+
const workflow = trimmed.match(/^workflow:([a-z0-9][a-z0-9_.-]*)$/);
|
|
608
|
+
if (workflow) return {
|
|
609
|
+
kind: "workflow",
|
|
610
|
+
name: workflow[1]
|
|
611
|
+
};
|
|
612
|
+
const rolloutPolicy = trimmed.match(/^rollout-policy:(.+)$/);
|
|
613
|
+
if (rolloutPolicy && rolloutPolicy[1].trim().length > 0) return {
|
|
614
|
+
kind: "rollout-policy",
|
|
615
|
+
field: rolloutPolicy[1].trim()
|
|
616
|
+
};
|
|
617
|
+
const agentProfile = trimmed.match(/^agent-profile:(.+)$/);
|
|
618
|
+
if (agentProfile && agentProfile[1].trim().length > 0) return {
|
|
619
|
+
kind: "agent-profile",
|
|
620
|
+
field: agentProfile[1].trim()
|
|
621
|
+
};
|
|
622
|
+
const code = trimmed.match(/^code:(.+)$/);
|
|
623
|
+
if (code && code[1].trim().length > 0) return {
|
|
624
|
+
kind: "code",
|
|
625
|
+
path: code[1].trim()
|
|
626
|
+
};
|
|
627
|
+
const rag = trimmed.match(/^rag:([a-z0-9][a-z0-9_-]*):(.+)$/);
|
|
628
|
+
if (rag && rag[2].trim().length > 0) return {
|
|
629
|
+
kind: "rag",
|
|
630
|
+
corpus: rag[1],
|
|
631
|
+
docId: rag[2].trim()
|
|
632
|
+
};
|
|
633
|
+
const mem = trimmed.match(/^memory:(.+)$/);
|
|
634
|
+
if (mem && mem[1].trim().length > 0) return {
|
|
635
|
+
kind: "memory",
|
|
636
|
+
key: mem[1].trim()
|
|
637
|
+
};
|
|
638
|
+
const sc = trimmed.match(/^scaffolding:(.+)$/);
|
|
639
|
+
if (sc && sc[1].trim().length > 0) return {
|
|
640
|
+
kind: "scaffolding",
|
|
641
|
+
concern: sc[1].trim()
|
|
642
|
+
};
|
|
643
|
+
const os = trimmed.match(/^output-schema:(.+)$/);
|
|
644
|
+
if (os && os[1].trim().length > 0) return {
|
|
645
|
+
kind: "output-schema",
|
|
646
|
+
field: os[1].trim()
|
|
647
|
+
};
|
|
648
|
+
const ws = trimmed.match(/^websearch:outdated:(.+)$/);
|
|
649
|
+
if (ws && ws[1].trim().length > 0) return {
|
|
650
|
+
kind: "websearch.outdated",
|
|
651
|
+
topic: ws[1].trim()
|
|
652
|
+
};
|
|
653
|
+
const prs = trimmed.match(/^prior-run-summary:(.+)$/);
|
|
654
|
+
if (prs && prs[1].trim().length > 0) return {
|
|
655
|
+
kind: "prior-run-summary",
|
|
656
|
+
topic: prs[1].trim()
|
|
657
|
+
};
|
|
658
|
+
if (/^[a-z0-9][a-z0-9._-]*$/.test(trimmed) && trimmed.length <= 80) return {
|
|
659
|
+
kind: "cluster",
|
|
660
|
+
label: trimmed
|
|
661
|
+
};
|
|
662
|
+
return null;
|
|
663
|
+
}
|
|
664
|
+
/**
|
|
665
|
+
* Render the parsed subject back to its canonical string form. Inverse
|
|
666
|
+
* of `parseFindingSubject`; useful when the substrate constructs new
|
|
667
|
+
* findings programmatically (e.g. for tests, replays, or
|
|
668
|
+
* `id_basis` carry-forward).
|
|
669
|
+
*/
|
|
670
|
+
function renderFindingSubject(s) {
|
|
671
|
+
switch (s.kind) {
|
|
672
|
+
case "knowledge.wiki": return s.heading ? `agent-knowledge:wiki:${s.slug}#${s.heading}` : `agent-knowledge:wiki:${s.slug}`;
|
|
673
|
+
case "knowledge.claim": return `agent-knowledge:claim:${s.topic}`;
|
|
674
|
+
case "knowledge.raw": return `agent-knowledge:raw:${s.sourceId}`;
|
|
675
|
+
case "knowledge.stale": return `agent-knowledge:stale:${s.slug}`;
|
|
676
|
+
case "system-prompt": return `system-prompt:${s.section}`;
|
|
677
|
+
case "skill": return `skill:${s.name}`;
|
|
678
|
+
case "tool-doc": return s.aspect ? `tool-doc:${s.tool}:${s.aspect}` : `tool-doc:${s.tool}`;
|
|
679
|
+
case "new-tool": return `new-tool:${s.name}`;
|
|
680
|
+
case "mcp": return s.tool ? `mcp:${s.server}:${s.tool}` : `mcp:${s.server}`;
|
|
681
|
+
case "hook": return `hook:${s.name}`;
|
|
682
|
+
case "subagent": return `subagent:${s.name}`;
|
|
683
|
+
case "workflow": return `workflow:${s.name}`;
|
|
684
|
+
case "rollout-policy": return `rollout-policy:${s.field}`;
|
|
685
|
+
case "agent-profile": return `agent-profile:${s.field}`;
|
|
686
|
+
case "code": return `code:${s.path}`;
|
|
687
|
+
case "rag": return `rag:${s.corpus}:${s.docId}`;
|
|
688
|
+
case "memory": return `memory:${s.key}`;
|
|
689
|
+
case "scaffolding": return `scaffolding:${s.concern}`;
|
|
690
|
+
case "output-schema": return `output-schema:${s.field}`;
|
|
691
|
+
case "websearch.outdated": return `websearch:outdated:${s.topic}`;
|
|
692
|
+
case "prior-run-summary": return `prior-run-summary:${s.topic}`;
|
|
693
|
+
case "cluster": return s.label;
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
/**
|
|
697
|
+
* The grammar text embedded into kind actor prompts. Kinds opt into
|
|
698
|
+
* the subset of variants they emit (e.g. `improvement` excludes the
|
|
699
|
+
* cluster variant; `failure-mode` includes ONLY the cluster variant).
|
|
700
|
+
*
|
|
701
|
+
* Drift between prompt and parser is impossible: every kind imports
|
|
702
|
+
* this constant + the matching `expects` set, and the unit tests below
|
|
703
|
+
* lock the table to the parser.
|
|
704
|
+
*/
|
|
705
|
+
const FINDING_SUBJECT_SYNTAX = {
|
|
706
|
+
"knowledge.wiki": "agent-knowledge:wiki:<slug>[#<heading>]",
|
|
707
|
+
"knowledge.claim": "agent-knowledge:claim:<topic>",
|
|
708
|
+
"knowledge.raw": "agent-knowledge:raw:<source-id>",
|
|
709
|
+
"knowledge.stale": "agent-knowledge:stale:<slug>",
|
|
710
|
+
"system-prompt": "system-prompt:<section>",
|
|
711
|
+
skill: "skill:<name>",
|
|
712
|
+
"tool-doc": "tool-doc:<tool>[:<aspect>]",
|
|
713
|
+
"new-tool": "new-tool:<name>",
|
|
714
|
+
mcp: "mcp:<server>[:<tool>]",
|
|
715
|
+
hook: "hook:<name>",
|
|
716
|
+
subagent: "subagent:<name>",
|
|
717
|
+
workflow: "workflow:<name>",
|
|
718
|
+
"rollout-policy": "rollout-policy:<field>",
|
|
719
|
+
"agent-profile": "agent-profile:<field>",
|
|
720
|
+
code: "code:<path>",
|
|
721
|
+
rag: "rag:<corpus>:<doc-id>",
|
|
722
|
+
memory: "memory:<key>",
|
|
723
|
+
scaffolding: "scaffolding:<concern>",
|
|
724
|
+
"output-schema": "output-schema:<field>",
|
|
725
|
+
"websearch.outdated": "websearch:outdated:<topic>",
|
|
726
|
+
"prior-run-summary": "prior-run-summary:<topic>",
|
|
727
|
+
cluster: "<lowercase-cluster-label>"
|
|
728
|
+
};
|
|
729
|
+
const FINDING_SUBJECT_PURPOSE = {
|
|
730
|
+
"knowledge.wiki": "create or update a wiki page",
|
|
731
|
+
"knowledge.claim": "draft a claim or relation",
|
|
732
|
+
"knowledge.raw": "curate a raw source",
|
|
733
|
+
"knowledge.stale": "mark a stale page",
|
|
734
|
+
"system-prompt": "revise a system-prompt section",
|
|
735
|
+
skill: "create or revise a skill",
|
|
736
|
+
"tool-doc": "revise a tool contract",
|
|
737
|
+
"new-tool": "propose a new tool",
|
|
738
|
+
mcp: "revise an MCP server or tool",
|
|
739
|
+
hook: "revise a lifecycle hook",
|
|
740
|
+
subagent: "revise a delegated agent",
|
|
741
|
+
workflow: "revise an orchestration workflow",
|
|
742
|
+
"rollout-policy": "revise budget, sampling, or stop policy",
|
|
743
|
+
"agent-profile": "revise another AgentProfile field",
|
|
744
|
+
code: "revise an implementation path",
|
|
745
|
+
rag: "ingest or correct a RAG document",
|
|
746
|
+
memory: "invalidate or set memory",
|
|
747
|
+
scaffolding: "revise preconditions, retries, or verification",
|
|
748
|
+
"output-schema": "constrain the output shape",
|
|
749
|
+
"websearch.outdated": "identify a stale web result",
|
|
750
|
+
"prior-run-summary": "identify a stale prior-run summary",
|
|
751
|
+
cluster: "name one failure cluster"
|
|
752
|
+
};
|
|
753
|
+
function renderFindingSubjectGrammar(kinds) {
|
|
754
|
+
return [
|
|
755
|
+
"Subjects MUST match one of these forms — anything else is rejected at parse time:",
|
|
756
|
+
...kinds.map((kind) => ` ${FINDING_SUBJECT_SYNTAX[kind]} — ${FINDING_SUBJECT_PURPOSE[kind]}`),
|
|
757
|
+
"Runtime ids are lowercase [a-z0-9_.-]+. Topics, keys, paths, and sections are free-form and trimmed."
|
|
758
|
+
].join("\n");
|
|
759
|
+
}
|
|
760
|
+
const FINDING_SUBJECT_GRAMMAR_PROMPT = renderFindingSubjectGrammar(FINDING_SUBJECT_KINDS);
|
|
761
|
+
/**
|
|
762
|
+
* The variants each kind is allowed to emit. Used at the kind factory
|
|
763
|
+
* boundary so a knowledge-gap finding can't sneak in a `system-prompt:*`
|
|
764
|
+
* subject (the improvement-analyst's job) and vice versa.
|
|
765
|
+
*
|
|
766
|
+
* `failure-mode` is restricted to `cluster` — the only kind that emits
|
|
767
|
+
* a non-locus subject.
|
|
768
|
+
*/
|
|
769
|
+
const KIND_EXPECTED_SUBJECTS = {
|
|
770
|
+
"failure-mode": ["cluster"],
|
|
771
|
+
"knowledge-gap": [
|
|
772
|
+
"knowledge.wiki",
|
|
773
|
+
"knowledge.claim",
|
|
774
|
+
"knowledge.raw",
|
|
775
|
+
"knowledge.stale",
|
|
776
|
+
"tool-doc",
|
|
777
|
+
"system-prompt",
|
|
778
|
+
"skill",
|
|
779
|
+
"mcp",
|
|
780
|
+
"subagent",
|
|
781
|
+
"workflow",
|
|
782
|
+
"memory",
|
|
783
|
+
"websearch.outdated",
|
|
784
|
+
"prior-run-summary"
|
|
785
|
+
],
|
|
786
|
+
"knowledge-poisoning": [
|
|
787
|
+
"knowledge.wiki",
|
|
788
|
+
"knowledge.claim",
|
|
789
|
+
"knowledge.raw",
|
|
790
|
+
"tool-doc",
|
|
791
|
+
"system-prompt",
|
|
792
|
+
"skill",
|
|
793
|
+
"mcp",
|
|
794
|
+
"hook",
|
|
795
|
+
"memory",
|
|
796
|
+
"websearch.outdated",
|
|
797
|
+
"prior-run-summary"
|
|
798
|
+
],
|
|
799
|
+
improvement: [
|
|
800
|
+
"system-prompt",
|
|
801
|
+
"skill",
|
|
802
|
+
"tool-doc",
|
|
803
|
+
"new-tool",
|
|
804
|
+
"mcp",
|
|
805
|
+
"hook",
|
|
806
|
+
"subagent",
|
|
807
|
+
"workflow",
|
|
808
|
+
"rollout-policy",
|
|
809
|
+
"agent-profile",
|
|
810
|
+
"code",
|
|
811
|
+
"rag",
|
|
812
|
+
"memory",
|
|
813
|
+
"scaffolding",
|
|
814
|
+
"output-schema",
|
|
815
|
+
"knowledge.wiki",
|
|
816
|
+
"knowledge.claim"
|
|
817
|
+
]
|
|
818
|
+
};
|
|
819
|
+
/** Render only the subject forms one analyst kind is permitted to emit. */
|
|
820
|
+
function findingSubjectGrammarPromptFor(kindId) {
|
|
821
|
+
const kinds = KIND_EXPECTED_SUBJECTS[kindId];
|
|
822
|
+
if (!kinds) throw new Error(`unknown analyst kind: ${kindId}`);
|
|
823
|
+
return renderFindingSubjectGrammar(kinds);
|
|
824
|
+
}
|
|
825
|
+
/**
|
|
826
|
+
* Zod schema that validates a raw subject string and returns the parsed
|
|
827
|
+
* `FindingSubject`. Embedded in `RawAnalystFindingSchema` via
|
|
828
|
+
* `transform`, so `subject` arrives at the kind factory either as a
|
|
829
|
+
* typed locus or as a parse error attached to a single Zod issue.
|
|
830
|
+
*
|
|
831
|
+
* Optionality is preserved: subjects ARE optional on the wire (some
|
|
832
|
+
* findings are descriptive, not actionable). When present, they MUST
|
|
833
|
+
* parse — emitting a malformed subject is a contract violation, not a
|
|
834
|
+
* soft signal.
|
|
835
|
+
*/
|
|
836
|
+
const FindingSubjectStringSchema = z.string().refine((s) => parseFindingSubject(s) !== null, { message: "subject does not match the finding-subject grammar" });
|
|
837
|
+
//#endregion
|
|
838
|
+
//#region src/analyst/parse-tolerant.ts
|
|
839
|
+
/**
|
|
840
|
+
* Forgiving pre-parse for analyst findings. Weak models routinely emit
|
|
841
|
+
* schema-correct content in an unusable wrapper — fenced ```json blocks, a
|
|
842
|
+
* single object where an array is expected, trailing commas. Measured: GPT-4o
|
|
843
|
+
* drops to 0% usable output purely from markdown-fence wrapping
|
|
844
|
+
* (arXiv:2605.02363). A five-line de-fence recovers most of it. This module is
|
|
845
|
+
* the de-fence/coerce step that runs BEFORE Zod, so a recoverable finding is
|
|
846
|
+
* repaired, not dropped.
|
|
847
|
+
*
|
|
848
|
+
* Pure + deterministic. No model, no network.
|
|
849
|
+
*/
|
|
850
|
+
/** Strip a ```lang ... ``` (or bare ``` ... ```) code fence, if the string is one. */
|
|
851
|
+
function stripCodeFences(text) {
|
|
852
|
+
const t = text.trim();
|
|
853
|
+
const m = t.match(/^```[a-zA-Z0-9]*\s*\n?([\s\S]*?)\n?```$/);
|
|
854
|
+
return m ? m[1].trim() : t;
|
|
855
|
+
}
|
|
856
|
+
/** Remove trailing commas before } or ] — the most common near-JSON defect. */
|
|
857
|
+
function dropTrailingCommas(s) {
|
|
858
|
+
return s.replace(/,(\s*[}\]])/g, "$1");
|
|
859
|
+
}
|
|
860
|
+
/**
|
|
861
|
+
* Best-effort parse of a string into JSON. De-fences, drops trailing commas,
|
|
862
|
+
* then `JSON.parse`. Returns `undefined` (never throws) when unrecoverable.
|
|
863
|
+
*/
|
|
864
|
+
function coerceJson(text) {
|
|
865
|
+
const candidate = dropTrailingCommas(stripCodeFences(text));
|
|
866
|
+
try {
|
|
867
|
+
return JSON.parse(candidate);
|
|
868
|
+
} catch {
|
|
869
|
+
return;
|
|
870
|
+
}
|
|
871
|
+
}
|
|
872
|
+
/**
|
|
873
|
+
* Coerce arbitrary actor/structurer output into an array of candidate finding
|
|
874
|
+
* rows: a JSON string → parse; a single object → 1-element array; an array →
|
|
875
|
+
* as-is; anything else → []. Callers still run each row through Zod
|
|
876
|
+
* (`parseRawFinding`) — this only fixes the shape and never invents fields.
|
|
877
|
+
*/
|
|
878
|
+
function coerceToFindingRows(raw) {
|
|
879
|
+
let value = raw;
|
|
880
|
+
if (typeof value === "string") {
|
|
881
|
+
const parsed = coerceJson(value);
|
|
882
|
+
if (parsed === void 0) return [];
|
|
883
|
+
value = parsed;
|
|
884
|
+
}
|
|
885
|
+
if (Array.isArray(value)) return value;
|
|
886
|
+
if (value && typeof value === "object") {
|
|
887
|
+
const inner = value.findings;
|
|
888
|
+
if (Array.isArray(inner)) return inner;
|
|
889
|
+
return [value];
|
|
890
|
+
}
|
|
891
|
+
return [];
|
|
892
|
+
}
|
|
893
|
+
//#endregion
|
|
894
|
+
//#region src/analyst/finding-signature.ts
|
|
895
|
+
/**
|
|
896
|
+
* Engine-neutral structured output for trace-analyst findings.
|
|
897
|
+
*
|
|
898
|
+
* Every recursive engine returns this shape. The TypeScript boundary validates
|
|
899
|
+
* it before a finding can enter a registry or benchmark.
|
|
900
|
+
*/
|
|
901
|
+
const ANALYST_SEVERITIES = [
|
|
902
|
+
"critical",
|
|
903
|
+
"high",
|
|
904
|
+
"medium",
|
|
905
|
+
"low",
|
|
906
|
+
"info"
|
|
907
|
+
];
|
|
908
|
+
const RawAnalystEvidenceSchema = z.object({
|
|
909
|
+
uri: z.string().trim().min(1).max(2e3),
|
|
910
|
+
excerpt: z.string().max(2e3).optional()
|
|
911
|
+
}).strict();
|
|
912
|
+
const RawAnalystFindingBaseShape = {
|
|
913
|
+
severity: z.enum(ANALYST_SEVERITIES),
|
|
914
|
+
claim: z.string().min(1).max(2e3),
|
|
915
|
+
subject: z.string().max(400).refine((subject) => parseFindingSubject(subject) !== null, { message: "subject does not match the finding-subject grammar" }).optional(),
|
|
916
|
+
confidence: z.number().min(0).max(1),
|
|
917
|
+
rationale: z.string().max(4e3).optional(),
|
|
918
|
+
recommended_action: z.string().max(2e3).optional()
|
|
919
|
+
};
|
|
920
|
+
const RawAnalystFindingSchema = z.object({
|
|
921
|
+
...RawAnalystFindingBaseShape,
|
|
922
|
+
evidence: z.array(RawAnalystEvidenceSchema).min(1)
|
|
923
|
+
}).strict();
|
|
924
|
+
/**
|
|
925
|
+
* Description embedded into the actor prompt so the LLM knows what
|
|
926
|
+
* shape to emit. Kept here so kinds share one source of truth rather
|
|
927
|
+
* than restating the schema in every prompt.
|
|
928
|
+
*/
|
|
929
|
+
const RAW_FINDING_SCHEMA_PROMPT = `Each finding MUST be a strict JSON object with:
|
|
930
|
+
- severity: "critical" | "high" | "medium" | "low" | "info"
|
|
931
|
+
- claim: one-sentence statement (max 2000 chars)
|
|
932
|
+
- subject?: one exact subject form listed by this kind; omit rather than guess
|
|
933
|
+
- evidence: REQUIRED non-empty array of {"uri": string, "excerpt"?: string}. Use trace://<URL-encoded-trace-id>/span/<URL-encoded-span-id> for trace evidence or finding://<finding-id> for supplied prior findings. URL encoding means percent encoding, never base64. Include a short exact quote in excerpt when available. If nothing is citable, do not emit the finding.
|
|
934
|
+
- confidence: number 0..1 (0.9+ exact evidence; 0.6-0.8 inferred pattern; <0.5 speculative)
|
|
935
|
+
- rationale?: one or two reasoning sentences
|
|
936
|
+
- recommended_action?: concrete imperative change; omit for descriptive findings
|
|
937
|
+
|
|
938
|
+
Unknown fields are rejected. Do not emit area; the factory assigns it. Emit [] when there are no findings. Never fabricate evidence.`;
|
|
939
|
+
/** Convert raw citations into the public finding evidence envelope. */
|
|
940
|
+
function evidenceRefsFromRawFinding(finding) {
|
|
941
|
+
return finding.evidence.map(({ uri, excerpt }) => ({
|
|
942
|
+
kind: evidenceKindFromUri(uri),
|
|
943
|
+
uri,
|
|
944
|
+
excerpt
|
|
945
|
+
}));
|
|
946
|
+
}
|
|
947
|
+
function parseRawFinding(row, log) {
|
|
948
|
+
return parseFindingWithSchema(RawAnalystFindingSchema, row, log);
|
|
949
|
+
}
|
|
950
|
+
function parseFindingWithSchema(schema, row, log) {
|
|
951
|
+
const result = schema.safeParse(row);
|
|
952
|
+
if (result.success) return result.data;
|
|
953
|
+
if (typeof row === "string") {
|
|
954
|
+
const coerced = coerceJson(row);
|
|
955
|
+
if (coerced !== void 0) {
|
|
956
|
+
const retry = schema.safeParse(coerced);
|
|
957
|
+
if (retry.success) return retry.data;
|
|
958
|
+
}
|
|
959
|
+
}
|
|
960
|
+
log?.("finding rejected: schema failure", { issues: result.error.issues.map((i) => ({
|
|
961
|
+
path: i.path.join("."),
|
|
962
|
+
code: i.code,
|
|
963
|
+
message: i.message
|
|
964
|
+
})) });
|
|
965
|
+
return null;
|
|
966
|
+
}
|
|
967
|
+
function evidenceKindFromUri(uri) {
|
|
968
|
+
if (parseTraceSpanEvidenceUri(uri)) return "span";
|
|
969
|
+
if (uri.startsWith("finding://")) return "finding";
|
|
970
|
+
return "artifact";
|
|
971
|
+
}
|
|
972
|
+
function parseTraceSpanEvidenceUri(uri) {
|
|
973
|
+
const match = /^trace:\/\/([^/]+)\/span\/([^/]+)$/.exec(uri);
|
|
974
|
+
if (!match) return null;
|
|
975
|
+
try {
|
|
976
|
+
const traceId = decodeURIComponent(match[1]);
|
|
977
|
+
const spanId = decodeURIComponent(match[2]);
|
|
978
|
+
return traceId && spanId ? {
|
|
979
|
+
traceId,
|
|
980
|
+
spanId
|
|
981
|
+
} : null;
|
|
982
|
+
} catch {
|
|
983
|
+
return null;
|
|
984
|
+
}
|
|
985
|
+
}
|
|
986
|
+
//#endregion
|
|
987
|
+
//#region src/trace-analyst/errors.ts
|
|
988
|
+
/** Invalid trace-tool arguments, including malformed filters and regexes. */
|
|
989
|
+
var TraceAnalysisValidationError = class extends ValidationError {};
|
|
990
|
+
/** A trace read cannot fit within a documented count or byte limit. */
|
|
991
|
+
var TraceAnalysisLimitError = class extends LimitExceededError {
|
|
992
|
+
operation;
|
|
993
|
+
actual;
|
|
994
|
+
limit;
|
|
995
|
+
constructor(operation, actual, limit, message) {
|
|
996
|
+
super(message ?? `${operation} produced ${actual}, over the limit of ${limit}`);
|
|
997
|
+
this.operation = operation;
|
|
998
|
+
this.actual = actual;
|
|
999
|
+
this.limit = limit;
|
|
1000
|
+
}
|
|
1001
|
+
};
|
|
1002
|
+
/** A supplied store returned a shape that violates the trace read contract. */
|
|
1003
|
+
var TraceAnalysisStoreContractError = class extends AgentEvalError {
|
|
1004
|
+
operation;
|
|
1005
|
+
constructor(operation, message, options) {
|
|
1006
|
+
super("backend_integrity", `${operation}: ${message}`, options);
|
|
1007
|
+
this.operation = operation;
|
|
1008
|
+
}
|
|
1009
|
+
};
|
|
1010
|
+
var TraceFileMissingError = class extends NotFoundError {
|
|
1011
|
+
path;
|
|
1012
|
+
constructor(path) {
|
|
1013
|
+
super(`trace file not found: ${path}`);
|
|
1014
|
+
this.path = path;
|
|
1015
|
+
}
|
|
1016
|
+
};
|
|
1017
|
+
var TraceFileTooLargeError = class extends TraceAnalysisLimitError {
|
|
1018
|
+
path;
|
|
1019
|
+
size_bytes;
|
|
1020
|
+
max_bytes;
|
|
1021
|
+
constructor(path, size_bytes, max_bytes) {
|
|
1022
|
+
super("OtlpFileTraceStore.readBuffer", size_bytes, max_bytes, `trace file ${path} is ${size_bytes} bytes, over the ${max_bytes}-byte limit; raise OtlpFileTraceStoreOptions.maxFileBytes or pre-split the file`);
|
|
1023
|
+
this.path = path;
|
|
1024
|
+
this.size_bytes = size_bytes;
|
|
1025
|
+
this.max_bytes = max_bytes;
|
|
1026
|
+
}
|
|
1027
|
+
};
|
|
1028
|
+
var TraceFileMalformedError = class extends CaptureIntegrityError {
|
|
1029
|
+
path;
|
|
1030
|
+
line_number;
|
|
1031
|
+
byte_offset;
|
|
1032
|
+
constructor(path, line_number, byte_offset, cause) {
|
|
1033
|
+
super(`malformed trace row in ${path} at line ${line_number}, byte ${byte_offset}`, { cause });
|
|
1034
|
+
this.path = path;
|
|
1035
|
+
this.line_number = line_number;
|
|
1036
|
+
this.byte_offset = byte_offset;
|
|
1037
|
+
}
|
|
1038
|
+
};
|
|
1039
|
+
var TraceNotFoundError = class extends NotFoundError {
|
|
1040
|
+
trace_id;
|
|
1041
|
+
constructor(trace_id) {
|
|
1042
|
+
super(`trace not found: ${trace_id}`);
|
|
1043
|
+
this.trace_id = trace_id;
|
|
1044
|
+
}
|
|
1045
|
+
};
|
|
1046
|
+
var SpanNotFoundError = class extends NotFoundError {
|
|
1047
|
+
trace_id;
|
|
1048
|
+
span_id;
|
|
1049
|
+
constructor(trace_id, span_id) {
|
|
1050
|
+
super(`span ${span_id} not found in trace ${trace_id}`);
|
|
1051
|
+
this.trace_id = trace_id;
|
|
1052
|
+
this.span_id = span_id;
|
|
1053
|
+
}
|
|
1054
|
+
};
|
|
1055
|
+
//#endregion
|
|
1056
|
+
//#region src/trace-analyst/store-contract.ts
|
|
1057
|
+
const TRACE_ANALYSIS_LIMITS = {
|
|
1058
|
+
sampleTraceIds: 20,
|
|
1059
|
+
queryTraces: 200,
|
|
1060
|
+
viewSpans: 100,
|
|
1061
|
+
searchMatches: 500,
|
|
1062
|
+
filterValues: 100,
|
|
1063
|
+
identifierCharacters: 256,
|
|
1064
|
+
regexCharacters: 4096,
|
|
1065
|
+
minimumTextBudget: 64
|
|
1066
|
+
};
|
|
1067
|
+
//#endregion
|
|
1068
|
+
//#region src/trace-analyst/types.ts
|
|
1069
|
+
const DEFAULT_TRACE_ANALYST_BUDGETS = {
|
|
1070
|
+
perCallByteCeiling: 15e4,
|
|
1071
|
+
perAttributeViewBudget: 4096,
|
|
1072
|
+
perAttributeSpanBudget: 16384,
|
|
1073
|
+
perMatchTextBudget: 1024
|
|
1074
|
+
};
|
|
1075
|
+
/** Marker substituted in place of truncated string payloads. Callers
|
|
1076
|
+
* parsing tool output can detect it deterministically. */
|
|
1077
|
+
const TRACE_ANALYST_TRUNCATION_MARKER_PREFIX = "[trace-analyst truncated:";
|
|
1078
|
+
//#endregion
|
|
1079
|
+
//#region src/trace-analyst/store-bounds.ts
|
|
1080
|
+
function resolveTraceBudgets(overrides) {
|
|
1081
|
+
const budgets = {
|
|
1082
|
+
...DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
1083
|
+
...overrides
|
|
1084
|
+
};
|
|
1085
|
+
validateInteger(budgets.perCallByteCeiling, "perCallByteCeiling", 1);
|
|
1086
|
+
for (const name of [
|
|
1087
|
+
"perAttributeViewBudget",
|
|
1088
|
+
"perAttributeSpanBudget",
|
|
1089
|
+
"perMatchTextBudget"
|
|
1090
|
+
]) validateInteger(budgets[name], name, TRACE_ANALYSIS_LIMITS.minimumTextBudget);
|
|
1091
|
+
return budgets;
|
|
1092
|
+
}
|
|
1093
|
+
function boundOverview(result, byteCeiling) {
|
|
1094
|
+
if (result.sample_trace_ids.length > result.total_traces) throw contractError("getOverview", "sample_trace_ids contains more entries than total_traces");
|
|
1095
|
+
if (result.errors.trace_count > result.total_traces) throw contractError("getOverview", "errors.trace_count exceeds total_traces");
|
|
1096
|
+
return withinByteCeiling("getOverview", result, byteCeiling);
|
|
1097
|
+
}
|
|
1098
|
+
function boundTracePage(result, input, byteCeiling) {
|
|
1099
|
+
if (result.traces.length > input.limit) throw contractError("queryTraces", `store returned ${result.traces.length} traces for limit ${input.limit}`);
|
|
1100
|
+
if (result.total < input.offset + result.traces.length) throw contractError("queryTraces", `total ${result.total} is smaller than the returned page ending at ${input.offset + result.traces.length}`);
|
|
1101
|
+
if (result.has_more !== input.offset + result.traces.length < result.total) throw contractError("queryTraces", "has_more does not match the returned page position and total");
|
|
1102
|
+
if (result.has_more && result.traces.length === 0) throw contractError("queryTraces", "store reported more traces without returning any progress");
|
|
1103
|
+
const traces = [];
|
|
1104
|
+
for (const trace of result.traces) {
|
|
1105
|
+
if (encodedBytes("queryTraces", {
|
|
1106
|
+
traces: [...traces, trace],
|
|
1107
|
+
total: result.total,
|
|
1108
|
+
has_more: true
|
|
1109
|
+
}) > byteCeiling) break;
|
|
1110
|
+
traces.push(trace);
|
|
1111
|
+
}
|
|
1112
|
+
if (result.traces.length > 0 && traces.length === 0) throw responseItemTooLarge("queryTraces", {
|
|
1113
|
+
traces: [result.traces[0]],
|
|
1114
|
+
total: result.total,
|
|
1115
|
+
has_more: true
|
|
1116
|
+
}, byteCeiling);
|
|
1117
|
+
return withinByteCeiling("queryTraces", {
|
|
1118
|
+
traces,
|
|
1119
|
+
total: result.total,
|
|
1120
|
+
has_more: result.has_more || traces.length < result.traces.length
|
|
1121
|
+
}, byteCeiling);
|
|
1122
|
+
}
|
|
1123
|
+
function boundTraceView(result, expectedTraceId, perAttributeCap, byteCeiling) {
|
|
1124
|
+
requireEqual("viewTrace", "trace_id", result.trace_id, expectedTraceId);
|
|
1125
|
+
if (result.oversized) return withinByteCeiling("viewTrace", result, byteCeiling);
|
|
1126
|
+
const spans = result.spans.map((span) => {
|
|
1127
|
+
requireEqual("viewTrace", "span.trace_id", span.trace_id, expectedTraceId);
|
|
1128
|
+
return truncateSpanAttributes("viewTrace", span, perAttributeCap).span;
|
|
1129
|
+
});
|
|
1130
|
+
const full = {
|
|
1131
|
+
trace_id: expectedTraceId,
|
|
1132
|
+
spans
|
|
1133
|
+
};
|
|
1134
|
+
if (encodedBytes("viewTrace", full) <= byteCeiling) return full;
|
|
1135
|
+
return withinByteCeiling("viewTrace", {
|
|
1136
|
+
trace_id: expectedTraceId,
|
|
1137
|
+
oversized: oversizedFromSpans(spans)
|
|
1138
|
+
}, byteCeiling);
|
|
1139
|
+
}
|
|
1140
|
+
function boundSpansView(result, expectedTraceId, requested, existingSpanIds, perAttributeCap, byteCeiling) {
|
|
1141
|
+
requireEqual("viewSpans", "trace_id", result.trace_id, expectedTraceId);
|
|
1142
|
+
const requestedSet = new Set(requested);
|
|
1143
|
+
const missing = checkedAccountingIds("missing_span_ids", result.missing_span_ids, requestedSet);
|
|
1144
|
+
const omitted = checkedAccountingIds("omitted_span_ids", result.omitted_span_ids, requestedSet);
|
|
1145
|
+
const missingSet = new Set(missing);
|
|
1146
|
+
const omittedSet = new Set(omitted);
|
|
1147
|
+
const expectedMissing = requested.filter((id) => !existingSpanIds.has(id));
|
|
1148
|
+
if (expectedMissing.length !== missing.length || expectedMissing.some((id) => !missingSet.has(id))) throw contractError("viewSpans", "missing_span_ids does not match the store existence checks");
|
|
1149
|
+
if (result.has_more !== omitted.length > 0) throw contractError("viewSpans", "has_more must be true exactly when omitted_span_ids is non-empty");
|
|
1150
|
+
if (result.spans.length === 0 && requested.some((id) => existingSpanIds.has(id) && omittedSet.has(id))) throw contractError("viewSpans", "store omitted every existing requested span instead of returning progress or throwing a size error");
|
|
1151
|
+
const projected = /* @__PURE__ */ new Map();
|
|
1152
|
+
for (const span of result.spans) {
|
|
1153
|
+
requireEqual("viewSpans", "span.trace_id", span.trace_id, expectedTraceId);
|
|
1154
|
+
if (!requestedSet.has(span.span_id)) throw contractError("viewSpans", `store returned unrequested span ${JSON.stringify(span.span_id)}`);
|
|
1155
|
+
if (missingSet.has(span.span_id) || omittedSet.has(span.span_id)) throw contractError("viewSpans", `span ${JSON.stringify(span.span_id)} is both returned and unavailable`);
|
|
1156
|
+
if (projected.has(span.span_id)) throw contractError("viewSpans", `store returned duplicate span ${JSON.stringify(span.span_id)}`);
|
|
1157
|
+
projected.set(span.span_id, truncateSpanAttributes("viewSpans", span, perAttributeCap));
|
|
1158
|
+
}
|
|
1159
|
+
for (const id of requested) if (!missingSet.has(id) && !omittedSet.has(id) && !projected.has(id)) throw contractError("viewSpans", `store did not account for requested span ${JSON.stringify(id)}`);
|
|
1160
|
+
const spans = [];
|
|
1161
|
+
let addedTruncations = 0;
|
|
1162
|
+
const build = () => ({
|
|
1163
|
+
trace_id: expectedTraceId,
|
|
1164
|
+
spans,
|
|
1165
|
+
missing_span_ids: requested.filter((id) => missingSet.has(id)),
|
|
1166
|
+
omitted_span_ids: requested.filter((id) => omittedSet.has(id)),
|
|
1167
|
+
has_more: omittedSet.size > 0,
|
|
1168
|
+
truncated_attribute_count: result.truncated_attribute_count + addedTruncations
|
|
1169
|
+
});
|
|
1170
|
+
withinByteCeiling("viewSpans", build(), byteCeiling);
|
|
1171
|
+
for (const id of requested) {
|
|
1172
|
+
const item = projected.get(id);
|
|
1173
|
+
if (!item) continue;
|
|
1174
|
+
spans.push(item.span);
|
|
1175
|
+
omittedSet.delete(id);
|
|
1176
|
+
addedTruncations += item.truncations;
|
|
1177
|
+
if (encodedBytes("viewSpans", build()) <= byteCeiling) continue;
|
|
1178
|
+
spans.pop();
|
|
1179
|
+
omittedSet.add(id);
|
|
1180
|
+
addedTruncations -= item.truncations;
|
|
1181
|
+
}
|
|
1182
|
+
if (projected.size > 0 && spans.length === 0) {
|
|
1183
|
+
const first = projected.values().next().value;
|
|
1184
|
+
if (first) throw responseItemTooLarge("viewSpans", {
|
|
1185
|
+
trace_id: expectedTraceId,
|
|
1186
|
+
spans: [first.span],
|
|
1187
|
+
missing_span_ids: [],
|
|
1188
|
+
omitted_span_ids: [],
|
|
1189
|
+
has_more: false,
|
|
1190
|
+
truncated_attribute_count: result.truncated_attribute_count + first.truncations
|
|
1191
|
+
}, byteCeiling);
|
|
1192
|
+
}
|
|
1193
|
+
return withinByteCeiling("viewSpans", build(), byteCeiling);
|
|
1194
|
+
}
|
|
1195
|
+
function boundTraceSearch(result, expectedTraceId, maxMatches, budgets) {
|
|
1196
|
+
requireEqual("searchTrace", "trace_id", result.trace_id, expectedTraceId);
|
|
1197
|
+
return {
|
|
1198
|
+
trace_id: expectedTraceId,
|
|
1199
|
+
...boundSearchHits("searchTrace", result, maxMatches, budgets, (hit) => {
|
|
1200
|
+
requireEqual("searchTrace", "hit.trace_id", hit.trace_id, expectedTraceId);
|
|
1201
|
+
})
|
|
1202
|
+
};
|
|
1203
|
+
}
|
|
1204
|
+
function boundSpanSearch(result, expectedTraceId, expectedSpanId, maxMatches, budgets) {
|
|
1205
|
+
requireEqual("searchSpan", "trace_id", result.trace_id, expectedTraceId);
|
|
1206
|
+
requireEqual("searchSpan", "span_id", result.span_id, expectedSpanId);
|
|
1207
|
+
return {
|
|
1208
|
+
trace_id: expectedTraceId,
|
|
1209
|
+
span_id: expectedSpanId,
|
|
1210
|
+
...boundSearchHits("searchSpan", result, maxMatches, budgets, (hit) => {
|
|
1211
|
+
requireEqual("searchSpan", "hit.trace_id", hit.trace_id, expectedTraceId);
|
|
1212
|
+
requireEqual("searchSpan", "hit.span_id", hit.span_id, expectedSpanId);
|
|
1213
|
+
})
|
|
1214
|
+
};
|
|
1215
|
+
}
|
|
1216
|
+
function boundSearchHits(operation, result, maxMatches, budgets, validateHit) {
|
|
1217
|
+
if (result.hits.length > maxMatches) throw contractError(operation, `store returned ${result.hits.length} hits for max_matches ${maxMatches}`);
|
|
1218
|
+
if (result.has_more && result.hits.length === 0) throw contractError(operation, "store reported more matches without returning any progress");
|
|
1219
|
+
const hits = [];
|
|
1220
|
+
for (const raw of result.hits) {
|
|
1221
|
+
validateHit(raw);
|
|
1222
|
+
const hit = truncateMatchRecord(raw, budgets.perMatchTextBudget);
|
|
1223
|
+
if (encodedBytes(operation, {
|
|
1224
|
+
hits: [...hits, hit],
|
|
1225
|
+
has_more: true
|
|
1226
|
+
}) > budgets.perCallByteCeiling) break;
|
|
1227
|
+
hits.push(hit);
|
|
1228
|
+
}
|
|
1229
|
+
if (result.hits.length > 0 && hits.length === 0) throw responseItemTooLarge(operation, {
|
|
1230
|
+
hits: [truncateMatchRecord(result.hits[0], budgets.perMatchTextBudget)],
|
|
1231
|
+
has_more: true
|
|
1232
|
+
}, budgets.perCallByteCeiling);
|
|
1233
|
+
return withinByteCeiling(operation, {
|
|
1234
|
+
hits,
|
|
1235
|
+
has_more: result.has_more || hits.length < result.hits.length
|
|
1236
|
+
}, budgets.perCallByteCeiling);
|
|
1237
|
+
}
|
|
1238
|
+
function checkedAccountingIds(label, ids, requested) {
|
|
1239
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1240
|
+
for (const id of ids) {
|
|
1241
|
+
if (!requested.has(id)) throw contractError("viewSpans", `${label} contains unrequested id ${JSON.stringify(id)}`);
|
|
1242
|
+
if (seen.has(id)) throw contractError("viewSpans", `${label} contains duplicate id ${JSON.stringify(id)}`);
|
|
1243
|
+
seen.add(id);
|
|
1244
|
+
}
|
|
1245
|
+
return [...ids];
|
|
1246
|
+
}
|
|
1247
|
+
function truncateMatchRecord(record, cap) {
|
|
1248
|
+
return {
|
|
1249
|
+
...record,
|
|
1250
|
+
span_name: truncateForBudget(record.span_name, cap),
|
|
1251
|
+
attribute_path: truncateForBudget(record.attribute_path, cap),
|
|
1252
|
+
matched_text: truncateForBudget(record.matched_text, cap),
|
|
1253
|
+
context_before: truncateForBudget(record.context_before, cap),
|
|
1254
|
+
context_after: truncateForBudget(record.context_after, cap)
|
|
1255
|
+
};
|
|
1256
|
+
}
|
|
1257
|
+
function truncateSpanAttributes(operation, span, cap) {
|
|
1258
|
+
const attributes = {};
|
|
1259
|
+
let truncations = 0;
|
|
1260
|
+
for (const [key, value] of Object.entries(span.attributes)) {
|
|
1261
|
+
if (typeof value === "string") {
|
|
1262
|
+
const truncated = truncateForBudget(value, cap);
|
|
1263
|
+
if (truncated !== value) truncations += 1;
|
|
1264
|
+
attributes[key] = truncated;
|
|
1265
|
+
continue;
|
|
1266
|
+
}
|
|
1267
|
+
if (value !== null && typeof value === "object") {
|
|
1268
|
+
let json;
|
|
1269
|
+
try {
|
|
1270
|
+
json = JSON.stringify(value);
|
|
1271
|
+
} catch (cause) {
|
|
1272
|
+
throw new TraceAnalysisStoreContractError(operation, `span attribute ${JSON.stringify(key)} is not JSON-serializable`, { cause });
|
|
1273
|
+
}
|
|
1274
|
+
const truncated = truncateForBudget(json, cap);
|
|
1275
|
+
if (truncated !== json) {
|
|
1276
|
+
truncations += 1;
|
|
1277
|
+
attributes[key] = truncated;
|
|
1278
|
+
} else attributes[key] = value;
|
|
1279
|
+
continue;
|
|
1280
|
+
}
|
|
1281
|
+
attributes[key] = value;
|
|
1282
|
+
}
|
|
1283
|
+
return {
|
|
1284
|
+
span: {
|
|
1285
|
+
...span,
|
|
1286
|
+
attributes
|
|
1287
|
+
},
|
|
1288
|
+
truncations
|
|
1289
|
+
};
|
|
1290
|
+
}
|
|
1291
|
+
function oversizedFromSpans(spans) {
|
|
1292
|
+
const names = /* @__PURE__ */ new Map();
|
|
1293
|
+
let maxBytes = 0;
|
|
1294
|
+
let errors = 0;
|
|
1295
|
+
for (const span of spans) {
|
|
1296
|
+
names.set(span.name, (names.get(span.name) ?? 0) + 1);
|
|
1297
|
+
maxBytes = Math.max(maxBytes, encodedBytes("viewTrace", span));
|
|
1298
|
+
if (span.status === "ERROR") errors += 1;
|
|
1299
|
+
}
|
|
1300
|
+
return {
|
|
1301
|
+
span_count: spans.length,
|
|
1302
|
+
top_span_names: [...names.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 20),
|
|
1303
|
+
span_response_bytes_max: maxBytes,
|
|
1304
|
+
error_span_count: errors
|
|
1305
|
+
};
|
|
1306
|
+
}
|
|
1307
|
+
function compileSearchRegex(pattern) {
|
|
1308
|
+
if (typeof pattern !== "string" || pattern.length === 0) throw new TraceAnalysisValidationError("regex_pattern must be a non-empty string");
|
|
1309
|
+
let source = pattern;
|
|
1310
|
+
let flags = RE2JS.MULTILINE;
|
|
1311
|
+
if (source.startsWith("(?i)")) {
|
|
1312
|
+
source = source.slice(4);
|
|
1313
|
+
flags |= RE2JS.CASE_INSENSITIVE;
|
|
1314
|
+
}
|
|
1315
|
+
try {
|
|
1316
|
+
return RE2JS.compile(source, flags);
|
|
1317
|
+
} catch (cause) {
|
|
1318
|
+
throw new TraceAnalysisValidationError(`regex_pattern is invalid: ${cause instanceof Error ? cause.message : String(cause)}`, { cause });
|
|
1319
|
+
}
|
|
1320
|
+
}
|
|
1321
|
+
function truncateForBudget(value, byteCap) {
|
|
1322
|
+
validateInteger(byteCap, "byteCap", TRACE_ANALYSIS_LIMITS.minimumTextBudget);
|
|
1323
|
+
const original = Buffer.byteLength(value, "utf8");
|
|
1324
|
+
if (original <= byteCap) return value;
|
|
1325
|
+
const marker = `\n[trace-analyst truncated: original ${original} bytes]`;
|
|
1326
|
+
const contentCap = byteCap - Buffer.byteLength(marker, "utf8");
|
|
1327
|
+
let cut = Math.max(0, Math.floor(value.length * contentCap / original));
|
|
1328
|
+
while (cut > 0 && Buffer.byteLength(value.slice(0, cut), "utf8") > contentCap) cut -= 1;
|
|
1329
|
+
return `${value.slice(0, cut)}${marker}`;
|
|
1330
|
+
}
|
|
1331
|
+
function validateInteger(value, label, minimum, maximum = Number.MAX_SAFE_INTEGER) {
|
|
1332
|
+
if (typeof value !== "number" || !Number.isSafeInteger(value) || value < minimum || value > maximum) throw new TraceAnalysisValidationError(`${label} must be an integer ${minimum}..${maximum}, got ${String(value)}`);
|
|
1333
|
+
return value;
|
|
1334
|
+
}
|
|
1335
|
+
function requireEqual(operation, label, actual, expected) {
|
|
1336
|
+
if (actual !== expected) throw contractError(operation, `${label} ${JSON.stringify(actual)} does not match ${JSON.stringify(expected)}`);
|
|
1337
|
+
}
|
|
1338
|
+
function encodedBytes(operation, value) {
|
|
1339
|
+
try {
|
|
1340
|
+
const encoded = JSON.stringify(value);
|
|
1341
|
+
if (encoded === void 0) throw new TypeError("JSON.stringify returned undefined");
|
|
1342
|
+
return Buffer.byteLength(encoded, "utf8");
|
|
1343
|
+
} catch (cause) {
|
|
1344
|
+
if (cause instanceof TraceAnalysisStoreContractError) throw cause;
|
|
1345
|
+
throw new TraceAnalysisStoreContractError(operation, "store result is not JSON-serializable", { cause });
|
|
1346
|
+
}
|
|
1347
|
+
}
|
|
1348
|
+
function withinByteCeiling(operation, value, ceiling) {
|
|
1349
|
+
const actual = encodedBytes(operation, value);
|
|
1350
|
+
if (actual > ceiling) throw new TraceAnalysisLimitError(operation, actual, ceiling, `${operation} metadata requires ${actual} bytes, over the ${ceiling}-byte response limit`);
|
|
1351
|
+
return value;
|
|
1352
|
+
}
|
|
1353
|
+
function responseItemTooLarge(operation, value, ceiling) {
|
|
1354
|
+
return new TraceAnalysisLimitError(operation, encodedBytes(operation, value), ceiling, `${operation} cannot fit one result item in the ${ceiling}-byte response limit`);
|
|
1355
|
+
}
|
|
1356
|
+
function contractError(operation, message) {
|
|
1357
|
+
return new TraceAnalysisStoreContractError(operation, message);
|
|
1358
|
+
}
|
|
1359
|
+
//#endregion
|
|
1360
|
+
//#region src/trace-analyst/store-schemas.ts
|
|
1361
|
+
const identifier = z.string().min(1).max(TRACE_ANALYSIS_LIMITS.identifierCharacters);
|
|
1362
|
+
const timestamp = z.iso.datetime({ offset: true });
|
|
1363
|
+
const nonNegativeInteger = z.number().int().nonnegative();
|
|
1364
|
+
const nonNegativeNumber = z.number().nonnegative();
|
|
1365
|
+
const nullableString = z.string().nullable();
|
|
1366
|
+
const nullableIdentifier = identifier.nullable();
|
|
1367
|
+
const stringArray = z.array(z.string());
|
|
1368
|
+
const filterValues = z.array(identifier).max(TRACE_ANALYSIS_LIMITS.filterValues);
|
|
1369
|
+
const traceFiltersSchema = z.object({
|
|
1370
|
+
has_errors: z.boolean().optional(),
|
|
1371
|
+
service_names: filterValues.optional(),
|
|
1372
|
+
agent_names: filterValues.optional(),
|
|
1373
|
+
model_names: filterValues.optional(),
|
|
1374
|
+
tool_names: filterValues.optional(),
|
|
1375
|
+
start_time_after: timestamp.optional(),
|
|
1376
|
+
start_time_before: timestamp.optional(),
|
|
1377
|
+
regex_pattern: z.string().min(1).max(TRACE_ANALYSIS_LIMITS.regexCharacters).optional()
|
|
1378
|
+
}).strict();
|
|
1379
|
+
const byteCap = z.number().int().min(TRACE_ANALYSIS_LIMITS.minimumTextBudget);
|
|
1380
|
+
const searchPattern = z.string().min(1).max(TRACE_ANALYSIS_LIMITS.regexCharacters);
|
|
1381
|
+
const traceStoreInputSchemas = {
|
|
1382
|
+
hasTrace: z.object({ trace_id: identifier }).strict(),
|
|
1383
|
+
hasSpans: z.object({
|
|
1384
|
+
trace_id: identifier,
|
|
1385
|
+
span_ids: z.array(identifier).min(1).max(TRACE_ANALYSIS_LIMITS.viewSpans)
|
|
1386
|
+
}).strict(),
|
|
1387
|
+
getOverview: z.object({ filters: traceFiltersSchema.optional() }).strict(),
|
|
1388
|
+
queryTraces: z.object({
|
|
1389
|
+
filters: traceFiltersSchema.optional(),
|
|
1390
|
+
limit: z.number().int().min(1).max(TRACE_ANALYSIS_LIMITS.queryTraces),
|
|
1391
|
+
offset: z.number().int().nonnegative().optional()
|
|
1392
|
+
}).strict(),
|
|
1393
|
+
countTraces: z.object({ filters: traceFiltersSchema.optional() }).strict(),
|
|
1394
|
+
viewTrace: z.object({
|
|
1395
|
+
trace_id: identifier,
|
|
1396
|
+
per_attribute_byte_cap: byteCap.optional()
|
|
1397
|
+
}).strict(),
|
|
1398
|
+
viewSpans: z.object({
|
|
1399
|
+
trace_id: identifier,
|
|
1400
|
+
span_ids: z.array(identifier).min(1).max(TRACE_ANALYSIS_LIMITS.viewSpans),
|
|
1401
|
+
per_attribute_byte_cap: byteCap.optional()
|
|
1402
|
+
}).strict(),
|
|
1403
|
+
searchTrace: z.object({
|
|
1404
|
+
trace_id: identifier,
|
|
1405
|
+
regex_pattern: searchPattern,
|
|
1406
|
+
max_matches: z.number().int().min(1).max(TRACE_ANALYSIS_LIMITS.searchMatches).default(50)
|
|
1407
|
+
}).strict(),
|
|
1408
|
+
searchSpan: z.object({
|
|
1409
|
+
trace_id: identifier,
|
|
1410
|
+
span_id: identifier,
|
|
1411
|
+
regex_pattern: searchPattern,
|
|
1412
|
+
max_matches: z.number().int().min(1).max(TRACE_ANALYSIS_LIMITS.searchMatches).default(50)
|
|
1413
|
+
}).strict()
|
|
1414
|
+
};
|
|
1415
|
+
const spanKind = z.enum([
|
|
1416
|
+
"AGENT",
|
|
1417
|
+
"LLM",
|
|
1418
|
+
"TOOL",
|
|
1419
|
+
"CHAIN",
|
|
1420
|
+
"EVALUATOR",
|
|
1421
|
+
"GUARDRAIL",
|
|
1422
|
+
"SPAN",
|
|
1423
|
+
"UNKNOWN"
|
|
1424
|
+
]);
|
|
1425
|
+
const spanStatus = z.enum([
|
|
1426
|
+
"OK",
|
|
1427
|
+
"ERROR",
|
|
1428
|
+
"UNSET"
|
|
1429
|
+
]);
|
|
1430
|
+
const traceSpan = z.object({
|
|
1431
|
+
trace_id: identifier,
|
|
1432
|
+
span_id: identifier,
|
|
1433
|
+
parent_span_id: nullableIdentifier,
|
|
1434
|
+
name: z.string(),
|
|
1435
|
+
kind: spanKind,
|
|
1436
|
+
start_time: timestamp,
|
|
1437
|
+
end_time: timestamp,
|
|
1438
|
+
duration_ms: nonNegativeNumber,
|
|
1439
|
+
status: spanStatus,
|
|
1440
|
+
status_message: z.string().optional(),
|
|
1441
|
+
service_name: nullableString,
|
|
1442
|
+
agent_name: nullableString,
|
|
1443
|
+
model_name: nullableString,
|
|
1444
|
+
tool_name: nullableString,
|
|
1445
|
+
attributes: z.record(z.string(), z.json())
|
|
1446
|
+
}).strict();
|
|
1447
|
+
const traceSummary = z.object({
|
|
1448
|
+
trace_id: identifier,
|
|
1449
|
+
service_name: nullableString,
|
|
1450
|
+
agent_name: nullableString,
|
|
1451
|
+
span_count: nonNegativeInteger,
|
|
1452
|
+
has_errors: z.boolean(),
|
|
1453
|
+
start_time: timestamp,
|
|
1454
|
+
end_time: timestamp,
|
|
1455
|
+
duration_ms: nonNegativeNumber,
|
|
1456
|
+
raw_jsonl_bytes: nonNegativeInteger,
|
|
1457
|
+
models: stringArray,
|
|
1458
|
+
tools: stringArray
|
|
1459
|
+
}).strict();
|
|
1460
|
+
const errorCluster = z.object({
|
|
1461
|
+
signature: z.string(),
|
|
1462
|
+
status_message_sample: z.string(),
|
|
1463
|
+
span_name: nullableString,
|
|
1464
|
+
tool_name: nullableString,
|
|
1465
|
+
trace_count: nonNegativeInteger,
|
|
1466
|
+
span_count: nonNegativeInteger,
|
|
1467
|
+
prevalence: z.number().min(0).max(1),
|
|
1468
|
+
exemplar_trace_ids: z.array(identifier),
|
|
1469
|
+
exemplar_span_ids: z.array(identifier)
|
|
1470
|
+
}).strict();
|
|
1471
|
+
const overview = z.object({
|
|
1472
|
+
total_traces: nonNegativeInteger,
|
|
1473
|
+
raw_jsonl_bytes: nonNegativeInteger,
|
|
1474
|
+
services: stringArray,
|
|
1475
|
+
agents: stringArray,
|
|
1476
|
+
models: stringArray,
|
|
1477
|
+
tool_names: stringArray,
|
|
1478
|
+
sample_trace_ids: z.array(identifier).max(TRACE_ANALYSIS_LIMITS.sampleTraceIds),
|
|
1479
|
+
errors: z.object({
|
|
1480
|
+
trace_count: nonNegativeInteger,
|
|
1481
|
+
span_count: nonNegativeInteger
|
|
1482
|
+
}).strict(),
|
|
1483
|
+
error_clusters: z.array(errorCluster),
|
|
1484
|
+
time_range: z.object({
|
|
1485
|
+
earliest: timestamp,
|
|
1486
|
+
latest: timestamp
|
|
1487
|
+
}).strict().nullable()
|
|
1488
|
+
}).strict();
|
|
1489
|
+
const tracePage = z.object({
|
|
1490
|
+
traces: z.array(traceSummary),
|
|
1491
|
+
total: nonNegativeInteger,
|
|
1492
|
+
has_more: z.boolean()
|
|
1493
|
+
}).strict();
|
|
1494
|
+
const oversizedTrace = z.object({
|
|
1495
|
+
span_count: nonNegativeInteger,
|
|
1496
|
+
top_span_names: z.array(z.tuple([z.string(), nonNegativeInteger])).max(20),
|
|
1497
|
+
span_response_bytes_max: nonNegativeInteger,
|
|
1498
|
+
error_span_count: nonNegativeInteger
|
|
1499
|
+
}).strict();
|
|
1500
|
+
const traceView = z.object({
|
|
1501
|
+
trace_id: identifier,
|
|
1502
|
+
spans: z.array(traceSpan).optional(),
|
|
1503
|
+
oversized: oversizedTrace.optional()
|
|
1504
|
+
}).strict().refine((value) => value.spans === void 0 !== (value.oversized === void 0), { message: "exactly one of spans or oversized is required" });
|
|
1505
|
+
const spansView = z.object({
|
|
1506
|
+
trace_id: identifier,
|
|
1507
|
+
spans: z.array(traceSpan),
|
|
1508
|
+
missing_span_ids: z.array(identifier),
|
|
1509
|
+
omitted_span_ids: z.array(identifier),
|
|
1510
|
+
has_more: z.boolean(),
|
|
1511
|
+
truncated_attribute_count: nonNegativeInteger
|
|
1512
|
+
}).strict();
|
|
1513
|
+
const matchRecord = z.object({
|
|
1514
|
+
trace_id: identifier,
|
|
1515
|
+
span_id: identifier,
|
|
1516
|
+
span_name: z.string(),
|
|
1517
|
+
span_kind: spanKind,
|
|
1518
|
+
attribute_path: z.string(),
|
|
1519
|
+
matched_text: z.string(),
|
|
1520
|
+
context_before: z.string(),
|
|
1521
|
+
context_after: z.string(),
|
|
1522
|
+
match_offset: nonNegativeInteger
|
|
1523
|
+
}).strict();
|
|
1524
|
+
const traceSearch = z.object({
|
|
1525
|
+
trace_id: identifier,
|
|
1526
|
+
hits: z.array(matchRecord),
|
|
1527
|
+
has_more: z.boolean()
|
|
1528
|
+
}).strict();
|
|
1529
|
+
const spanSearch = traceSearch.extend({ span_id: identifier }).strict();
|
|
1530
|
+
const traceStoreOutputSchemas = {
|
|
1531
|
+
hasTrace: z.boolean(),
|
|
1532
|
+
hasSpans: z.array(identifier).max(TRACE_ANALYSIS_LIMITS.viewSpans),
|
|
1533
|
+
getOverview: overview,
|
|
1534
|
+
queryTraces: tracePage,
|
|
1535
|
+
countTraces: nonNegativeInteger,
|
|
1536
|
+
viewTrace: traceView,
|
|
1537
|
+
viewSpans: spansView,
|
|
1538
|
+
searchTrace: traceSearch,
|
|
1539
|
+
searchSpan: spanSearch
|
|
1540
|
+
};
|
|
1541
|
+
function parseTraceInput(operation, schema, value) {
|
|
1542
|
+
try {
|
|
1543
|
+
return schema.parse(value);
|
|
1544
|
+
} catch (cause) {
|
|
1545
|
+
throw new TraceAnalysisValidationError(`${operation}: invalid arguments: ${cause instanceof z.ZodError ? z.prettifyError(cause) : String(cause)}`, { cause });
|
|
1546
|
+
}
|
|
1547
|
+
}
|
|
1548
|
+
function parseStoreOutput(operation, schema, value) {
|
|
1549
|
+
try {
|
|
1550
|
+
return schema.parse(value);
|
|
1551
|
+
} catch (cause) {
|
|
1552
|
+
throw new TraceAnalysisStoreContractError(operation, `invalid store result: ${cause instanceof z.ZodError ? z.prettifyError(cause) : String(cause)}`, { cause });
|
|
1553
|
+
}
|
|
1554
|
+
}
|
|
1555
|
+
function toTraceJsonSchema(schema) {
|
|
1556
|
+
return z.toJSONSchema(schema, { target: "draft-7" });
|
|
1557
|
+
}
|
|
1558
|
+
//#endregion
|
|
1559
|
+
//#region src/trace-analyst/store-boundary.ts
|
|
1560
|
+
/** Apply the public validation, cancellation, not-found, and size rules to any adapter. */
|
|
1561
|
+
function createBoundedTraceAnalysisStore(source, options = {}) {
|
|
1562
|
+
const budgets = resolveTraceBudgets(options.budgets);
|
|
1563
|
+
return {
|
|
1564
|
+
async hasTrace(traceId, context) {
|
|
1565
|
+
throwIfAborted(context);
|
|
1566
|
+
const { trace_id } = parseTraceInput("hasTrace", traceStoreInputSchemas.hasTrace, { trace_id: traceId });
|
|
1567
|
+
const result = await source.hasTrace(trace_id, context);
|
|
1568
|
+
throwIfAborted(context);
|
|
1569
|
+
return parseStoreOutput("hasTrace", traceStoreOutputSchemas.hasTrace, result);
|
|
1570
|
+
},
|
|
1571
|
+
async hasSpans(input, context) {
|
|
1572
|
+
throwIfAborted(context);
|
|
1573
|
+
const parsed = parseTraceInput("hasSpans", traceStoreInputSchemas.hasSpans, input);
|
|
1574
|
+
assertUniqueIds(parsed.span_ids, "hasSpans.span_ids");
|
|
1575
|
+
const result = await source.hasSpans(parsed, context);
|
|
1576
|
+
throwIfAborted(context);
|
|
1577
|
+
return validateExistingSpanIds(parseStoreOutput("hasSpans", traceStoreOutputSchemas.hasSpans, result), parsed.span_ids);
|
|
1578
|
+
},
|
|
1579
|
+
async getOverview(filters, context) {
|
|
1580
|
+
throwIfAborted(context);
|
|
1581
|
+
const input = parseTraceInput("getOverview", traceStoreInputSchemas.getOverview, { filters });
|
|
1582
|
+
const result = await source.getOverview(input.filters, context);
|
|
1583
|
+
throwIfAborted(context);
|
|
1584
|
+
return boundOverview(parseStoreOutput("getOverview", traceStoreOutputSchemas.getOverview, result), budgets.perCallByteCeiling);
|
|
1585
|
+
},
|
|
1586
|
+
async queryTraces(input, context) {
|
|
1587
|
+
throwIfAborted(context);
|
|
1588
|
+
const parsed = parseTraceInput("queryTraces", traceStoreInputSchemas.queryTraces, input);
|
|
1589
|
+
const offset = parsed.offset ?? 0;
|
|
1590
|
+
const result = await source.queryTraces({
|
|
1591
|
+
...parsed,
|
|
1592
|
+
offset
|
|
1593
|
+
}, context);
|
|
1594
|
+
throwIfAborted(context);
|
|
1595
|
+
return boundTracePage(parseStoreOutput("queryTraces", traceStoreOutputSchemas.queryTraces, result), {
|
|
1596
|
+
limit: parsed.limit,
|
|
1597
|
+
offset
|
|
1598
|
+
}, budgets.perCallByteCeiling);
|
|
1599
|
+
},
|
|
1600
|
+
async countTraces(filters, context) {
|
|
1601
|
+
throwIfAborted(context);
|
|
1602
|
+
const input = parseTraceInput("countTraces", traceStoreInputSchemas.countTraces, { filters });
|
|
1603
|
+
const result = await source.countTraces(input.filters, context);
|
|
1604
|
+
throwIfAborted(context);
|
|
1605
|
+
return parseStoreOutput("countTraces", traceStoreOutputSchemas.countTraces, result);
|
|
1606
|
+
},
|
|
1607
|
+
async viewTrace(input, context) {
|
|
1608
|
+
throwIfAborted(context);
|
|
1609
|
+
const parsed = parseTraceInput("viewTrace", traceStoreInputSchemas.viewTrace, input);
|
|
1610
|
+
await requireTrace(source, parsed.trace_id, context);
|
|
1611
|
+
const perAttributeCap = parsed.per_attribute_byte_cap ?? budgets.perAttributeViewBudget;
|
|
1612
|
+
const result = await source.viewTrace({
|
|
1613
|
+
...parsed,
|
|
1614
|
+
per_attribute_byte_cap: perAttributeCap
|
|
1615
|
+
}, context);
|
|
1616
|
+
throwIfAborted(context);
|
|
1617
|
+
return boundTraceView(parseStoreOutput("viewTrace", traceStoreOutputSchemas.viewTrace, result), parsed.trace_id, perAttributeCap, budgets.perCallByteCeiling);
|
|
1618
|
+
},
|
|
1619
|
+
async viewSpans(input, context) {
|
|
1620
|
+
throwIfAborted(context);
|
|
1621
|
+
const parsed = parseTraceInput("viewSpans", traceStoreInputSchemas.viewSpans, input);
|
|
1622
|
+
assertUniqueIds(parsed.span_ids, "viewSpans.span_ids");
|
|
1623
|
+
await requireTrace(source, parsed.trace_id, context);
|
|
1624
|
+
const existingSpanIds = new Set(validateExistingSpanIds(parseStoreOutput("hasSpans", traceStoreOutputSchemas.hasSpans, await source.hasSpans({
|
|
1625
|
+
trace_id: parsed.trace_id,
|
|
1626
|
+
span_ids: parsed.span_ids
|
|
1627
|
+
}, context)), parsed.span_ids));
|
|
1628
|
+
throwIfAborted(context);
|
|
1629
|
+
const perAttributeCap = parsed.per_attribute_byte_cap ?? budgets.perAttributeSpanBudget;
|
|
1630
|
+
const result = await source.viewSpans({
|
|
1631
|
+
...parsed,
|
|
1632
|
+
per_attribute_byte_cap: perAttributeCap
|
|
1633
|
+
}, context);
|
|
1634
|
+
throwIfAborted(context);
|
|
1635
|
+
return boundSpansView(parseStoreOutput("viewSpans", traceStoreOutputSchemas.viewSpans, result), parsed.trace_id, parsed.span_ids, existingSpanIds, perAttributeCap, budgets.perCallByteCeiling);
|
|
1636
|
+
},
|
|
1637
|
+
async searchTrace(input, context) {
|
|
1638
|
+
throwIfAborted(context);
|
|
1639
|
+
const parsed = parseTraceInput("searchTrace", traceStoreInputSchemas.searchTrace, input);
|
|
1640
|
+
compileSearchRegex(parsed.regex_pattern);
|
|
1641
|
+
await requireTrace(source, parsed.trace_id, context);
|
|
1642
|
+
const result = await source.searchTrace(parsed, context);
|
|
1643
|
+
throwIfAborted(context);
|
|
1644
|
+
return boundTraceSearch(parseStoreOutput("searchTrace", traceStoreOutputSchemas.searchTrace, result), parsed.trace_id, parsed.max_matches, budgets);
|
|
1645
|
+
},
|
|
1646
|
+
async searchSpan(input, context) {
|
|
1647
|
+
throwIfAborted(context);
|
|
1648
|
+
const parsed = parseTraceInput("searchSpan", traceStoreInputSchemas.searchSpan, input);
|
|
1649
|
+
compileSearchRegex(parsed.regex_pattern);
|
|
1650
|
+
await requireTrace(source, parsed.trace_id, context);
|
|
1651
|
+
await requireSpan(source, parsed.trace_id, parsed.span_id, context);
|
|
1652
|
+
const result = await source.searchSpan(parsed, context);
|
|
1653
|
+
throwIfAborted(context);
|
|
1654
|
+
return boundSpanSearch(parseStoreOutput("searchSpan", traceStoreOutputSchemas.searchSpan, result), parsed.trace_id, parsed.span_id, parsed.max_matches, budgets);
|
|
1655
|
+
}
|
|
1656
|
+
};
|
|
1657
|
+
}
|
|
1658
|
+
async function requireTrace(source, traceId, context) {
|
|
1659
|
+
throwIfAborted(context);
|
|
1660
|
+
const exists = await source.hasTrace(traceId, context);
|
|
1661
|
+
throwIfAborted(context);
|
|
1662
|
+
if (!parseStoreOutput("hasTrace", traceStoreOutputSchemas.hasTrace, exists)) throw new TraceNotFoundError(traceId);
|
|
1663
|
+
}
|
|
1664
|
+
async function requireSpan(source, traceId, spanId, context) {
|
|
1665
|
+
throwIfAborted(context);
|
|
1666
|
+
const existing = await source.hasSpans({
|
|
1667
|
+
trace_id: traceId,
|
|
1668
|
+
span_ids: [spanId]
|
|
1669
|
+
}, context);
|
|
1670
|
+
throwIfAborted(context);
|
|
1671
|
+
if (validateExistingSpanIds(parseStoreOutput("hasSpans", traceStoreOutputSchemas.hasSpans, existing), [spanId]).length === 0) throw new SpanNotFoundError(traceId, spanId);
|
|
1672
|
+
}
|
|
1673
|
+
function validateExistingSpanIds(found, requested) {
|
|
1674
|
+
const requestedSet = new Set(requested);
|
|
1675
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1676
|
+
for (const id of found) {
|
|
1677
|
+
if (!requestedSet.has(id)) throw new TraceAnalysisStoreContractError("hasSpans", `hasSpans returned unrequested span id ${JSON.stringify(id)}`);
|
|
1678
|
+
if (seen.has(id)) throw new TraceAnalysisStoreContractError("hasSpans", `hasSpans returned duplicate span id ${JSON.stringify(id)}`);
|
|
1679
|
+
seen.add(id);
|
|
1680
|
+
}
|
|
1681
|
+
return [...found];
|
|
1682
|
+
}
|
|
1683
|
+
function assertUniqueIds(ids, label) {
|
|
1684
|
+
if (new Set(ids).size !== ids.length) throw new TraceAnalysisValidationError(`${label} must not contain duplicates`);
|
|
1685
|
+
}
|
|
1686
|
+
function throwIfAborted(context) {
|
|
1687
|
+
context?.signal?.throwIfAborted();
|
|
1688
|
+
}
|
|
1689
|
+
//#endregion
|
|
1690
|
+
//#region src/trace-analyst/tools.ts
|
|
1691
|
+
const TRACE_ANALYST_TOOL_NAMESPACE = "traces";
|
|
1692
|
+
/** Bind all seven trace reads without exposing an agent framework type. */
|
|
1693
|
+
function buildTraceAnalysisToolDescriptors(options) {
|
|
1694
|
+
const store = createBoundedTraceAnalysisStore(options.store, { budgets: options.budgets });
|
|
1695
|
+
return [
|
|
1696
|
+
{
|
|
1697
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1698
|
+
name: "getDatasetOverview",
|
|
1699
|
+
description: "Dataset rollup: total traces, raw_jsonl_bytes, services, agents, models, tools, and sample_trace_ids. Always call this first without a regex_pattern.",
|
|
1700
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.getOverview),
|
|
1701
|
+
handler: async (args, context) => {
|
|
1702
|
+
const { filters } = parseTraceInput("getDatasetOverview", traceStoreInputSchemas.getOverview, args ?? {});
|
|
1703
|
+
return store.getOverview(filters, context);
|
|
1704
|
+
}
|
|
1705
|
+
},
|
|
1706
|
+
{
|
|
1707
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1708
|
+
name: "queryTraces",
|
|
1709
|
+
description: `Paginated trace summaries, at most ${TRACE_ANALYSIS_LIMITS.queryTraces} per call. Each summary carries raw_jsonl_bytes; narrow with indexed filters before regex_pattern.`,
|
|
1710
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.queryTraces),
|
|
1711
|
+
handler: async (args, context) => {
|
|
1712
|
+
const input = parseTraceInput("queryTraces", traceStoreInputSchemas.queryTraces, args);
|
|
1713
|
+
return store.queryTraces(input, context);
|
|
1714
|
+
}
|
|
1715
|
+
},
|
|
1716
|
+
{
|
|
1717
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1718
|
+
name: "countTraces",
|
|
1719
|
+
description: "Count traces matching filters. Use as a cheap pre-flight before a regex_pattern scan.",
|
|
1720
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.countTraces),
|
|
1721
|
+
handler: async (args, context) => {
|
|
1722
|
+
const { filters } = parseTraceInput("countTraces", traceStoreInputSchemas.countTraces, args ?? {});
|
|
1723
|
+
return store.countTraces(filters, context);
|
|
1724
|
+
}
|
|
1725
|
+
},
|
|
1726
|
+
{
|
|
1727
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1728
|
+
name: "viewTrace",
|
|
1729
|
+
description: "Return all spans for one trace with bounded attributes. Oversized responses carry an oversized summary instead of spans; continue with searchTrace or viewSpans.",
|
|
1730
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.viewTrace.omit({ per_attribute_byte_cap: true })),
|
|
1731
|
+
handler: async (args, context) => {
|
|
1732
|
+
const input = parseTraceInput("viewTrace", traceStoreInputSchemas.viewTrace.omit({ per_attribute_byte_cap: true }), args);
|
|
1733
|
+
return store.viewTrace(input, context);
|
|
1734
|
+
}
|
|
1735
|
+
},
|
|
1736
|
+
{
|
|
1737
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1738
|
+
name: "viewSpans",
|
|
1739
|
+
description: `Read 1..${TRACE_ANALYSIS_LIMITS.viewSpans} specific spans. Every requested id is accounted for in spans, missing_span_ids, or omitted_span_ids; retry omitted ids.`,
|
|
1740
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.viewSpans.omit({ per_attribute_byte_cap: true })),
|
|
1741
|
+
handler: async (args, context) => {
|
|
1742
|
+
const input = parseTraceInput("viewSpans", traceStoreInputSchemas.viewSpans.omit({ per_attribute_byte_cap: true }), args);
|
|
1743
|
+
return store.viewSpans(input, context);
|
|
1744
|
+
}
|
|
1745
|
+
},
|
|
1746
|
+
{
|
|
1747
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1748
|
+
name: "searchTrace",
|
|
1749
|
+
description: `Regex search across one trace, bounded to ${TRACE_ANALYSIS_LIMITS.searchMatches} hits. When has_more is true, refine the regex instead of treating the result as complete.`,
|
|
1750
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.searchTrace),
|
|
1751
|
+
handler: async (args, context) => {
|
|
1752
|
+
const input = parseTraceInput("searchTrace", traceStoreInputSchemas.searchTrace, args);
|
|
1753
|
+
return store.searchTrace(input, context);
|
|
1754
|
+
}
|
|
1755
|
+
},
|
|
1756
|
+
{
|
|
1757
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1758
|
+
name: "searchSpan",
|
|
1759
|
+
description: `Regex search inside one span, bounded to ${TRACE_ANALYSIS_LIMITS.searchMatches} hits. Use after viewSpans omits or truncates a large span payload.`,
|
|
1760
|
+
parameters: toTraceJsonSchema(traceStoreInputSchemas.searchSpan),
|
|
1761
|
+
handler: async (args, context) => {
|
|
1762
|
+
const input = parseTraceInput("searchSpan", traceStoreInputSchemas.searchSpan, args);
|
|
1763
|
+
return store.searchSpan(input, context);
|
|
1764
|
+
}
|
|
1765
|
+
}
|
|
1766
|
+
];
|
|
1767
|
+
}
|
|
1768
|
+
function traceAnalystFunctionGroup(options) {
|
|
1769
|
+
return {
|
|
1770
|
+
namespace: TRACE_ANALYST_TOOL_NAMESPACE,
|
|
1771
|
+
title: "Trace Analysis",
|
|
1772
|
+
selectionCriteria: "Use for any inspection of OTLP-shaped trace data.",
|
|
1773
|
+
description: "Discovery, narrowing, and bounded deep reads over a JSONL trace dataset. Always call getDatasetOverview first.",
|
|
1774
|
+
functions: buildTraceAnalysisToolDescriptors(options)
|
|
1775
|
+
};
|
|
1776
|
+
}
|
|
1777
|
+
//#endregion
|
|
1778
|
+
//#region src/analyst/tool-groups.ts
|
|
1779
|
+
const TOOL_NAMES_BY_GROUP = {
|
|
1780
|
+
all: /* @__PURE__ */ new Set(),
|
|
1781
|
+
discovery: /* @__PURE__ */ new Set([
|
|
1782
|
+
"getDatasetOverview",
|
|
1783
|
+
"queryTraces",
|
|
1784
|
+
"countTraces"
|
|
1785
|
+
]),
|
|
1786
|
+
discoveryAndRead: /* @__PURE__ */ new Set([
|
|
1787
|
+
"getDatasetOverview",
|
|
1788
|
+
"queryTraces",
|
|
1789
|
+
"countTraces",
|
|
1790
|
+
"viewTrace",
|
|
1791
|
+
"viewSpans"
|
|
1792
|
+
]),
|
|
1793
|
+
discoveryAndSearch: /* @__PURE__ */ new Set([
|
|
1794
|
+
"getDatasetOverview",
|
|
1795
|
+
"queryTraces",
|
|
1796
|
+
"countTraces",
|
|
1797
|
+
"searchTrace",
|
|
1798
|
+
"searchSpan"
|
|
1799
|
+
]),
|
|
1800
|
+
targeted: /* @__PURE__ */ new Set([
|
|
1801
|
+
"getDatasetOverview",
|
|
1802
|
+
"queryTraces",
|
|
1803
|
+
"viewSpans",
|
|
1804
|
+
"searchSpan"
|
|
1805
|
+
]),
|
|
1806
|
+
singleTrace: /* @__PURE__ */ new Set([
|
|
1807
|
+
"getDatasetOverview",
|
|
1808
|
+
"viewTrace",
|
|
1809
|
+
"viewSpans",
|
|
1810
|
+
"searchTrace",
|
|
1811
|
+
"searchSpan"
|
|
1812
|
+
])
|
|
1813
|
+
};
|
|
1814
|
+
/**
|
|
1815
|
+
* Build the tool set for a named group bound to a specific trace store.
|
|
1816
|
+
*
|
|
1817
|
+
* `all` returns every tool. Other groups filter the canonical descriptors
|
|
1818
|
+
* by name to the documented subset. An unrecognised group name throws —
|
|
1819
|
+
* silently returning all tools would defeat the cost-control point.
|
|
1820
|
+
*/
|
|
1821
|
+
function buildTraceToolsForGroup(group, store) {
|
|
1822
|
+
const all = buildTraceAnalysisToolDescriptors({ store });
|
|
1823
|
+
if (group === "all") return all;
|
|
1824
|
+
const allow = TOOL_NAMES_BY_GROUP[group];
|
|
1825
|
+
if (!allow) throw new Error(`unknown trace tool group: ${group}`);
|
|
1826
|
+
return all.filter((tool) => allow.has(tool.name));
|
|
1827
|
+
}
|
|
1828
|
+
//#endregion
|
|
1829
|
+
//#region src/analyst/kind-factory.ts
|
|
1830
|
+
/** Run one definition and retain its answer, findings, and full investigation record. */
|
|
1831
|
+
async function runTraceAnalyst(args) {
|
|
1832
|
+
const { definition, context } = args;
|
|
1833
|
+
validateDefinition(definition);
|
|
1834
|
+
const minimumEvidenceCitations = definition.minimumEvidenceCitations ?? 1;
|
|
1835
|
+
const settlementTimeoutMs = validateUsageSettlementTimeout(args.settlementTimeoutMs);
|
|
1836
|
+
const costLedger = context.costLedger ?? new CostLedger(context.budgetUsd);
|
|
1837
|
+
const costTags = {
|
|
1838
|
+
...context.tags ?? {},
|
|
1839
|
+
analystId: definition.id,
|
|
1840
|
+
...context.correlationId ? { analystRunId: context.correlationId } : {}
|
|
1841
|
+
};
|
|
1842
|
+
try {
|
|
1843
|
+
const preparedContext = await definition.prepareContext?.(args.store, context);
|
|
1844
|
+
if (preparedContext !== void 0 && typeof preparedContext !== "string") throw new TypeError(`trace analyst '${definition.id}' prepareContext must return a string`);
|
|
1845
|
+
const instructions = [
|
|
1846
|
+
definition.instructions.trim(),
|
|
1847
|
+
renderPriorFindings(context.priorFindings),
|
|
1848
|
+
renderUpstreamFindings(context.upstreamFindings),
|
|
1849
|
+
RAW_FINDING_SCHEMA_PROMPT,
|
|
1850
|
+
minimumEvidenceCitations > 1 ? `Every finding requires at least ${minimumEvidenceCitations} distinct evidence citations.` : "",
|
|
1851
|
+
preparedContext ? `PREPARED CONTEXT:\n${preparedContext}` : "",
|
|
1852
|
+
"Return a direct prose answer and a strict findings array. Use trace tools to investigate. Do not infer trace facts from the question alone."
|
|
1853
|
+
].filter(Boolean).join("\n\n");
|
|
1854
|
+
const completed = await args.engine.analyze({
|
|
1855
|
+
analystId: definition.id,
|
|
1856
|
+
question: deriveQuestion(context, definition),
|
|
1857
|
+
instructions,
|
|
1858
|
+
tools: buildTraceToolsForGroup(definition.toolGroup, args.store),
|
|
1859
|
+
limits: resolveTraceAnalystLimits(definition.limits),
|
|
1860
|
+
costLedger,
|
|
1861
|
+
costPhase: context.costPhase ?? "trace-analysis",
|
|
1862
|
+
costTags,
|
|
1863
|
+
...context.signal ? { signal: context.signal } : {},
|
|
1864
|
+
...context.log ? { log: context.log } : {}
|
|
1865
|
+
});
|
|
1866
|
+
const findings = await acceptFindings(definition, completed.findings, args.store, context, minimumEvidenceCitations);
|
|
1867
|
+
if (definition.requireStructuredFindings && findings.length === 0) throw new Error(`trace analyst '${definition.id}' returned no valid structured findings: ${truncateForContext(completed.answer, 600)}`);
|
|
1868
|
+
context.log?.(`trace analyst ${definition.id} completed`, {
|
|
1869
|
+
engine: args.engine.id,
|
|
1870
|
+
model_calls: completed.modelCalls,
|
|
1871
|
+
tool_calls: completed.toolCalls,
|
|
1872
|
+
submitted_findings: completed.findings.length,
|
|
1873
|
+
accepted_findings: findings.length
|
|
1874
|
+
});
|
|
1875
|
+
return {
|
|
1876
|
+
...completed,
|
|
1877
|
+
findings
|
|
1878
|
+
};
|
|
1879
|
+
} finally {
|
|
1880
|
+
const usage = await settleUsageReceiptFromCostLedger(costLedger, {
|
|
1881
|
+
channel: "analyst",
|
|
1882
|
+
tags: costTags,
|
|
1883
|
+
timeoutMs: settlementTimeoutMs
|
|
1884
|
+
});
|
|
1885
|
+
if (!usage.settled) context.log?.(`trace analyst ${definition.id} provider settlement timed out`, {
|
|
1886
|
+
pending_calls: usage.pendingCalls,
|
|
1887
|
+
timeout_ms: settlementTimeoutMs
|
|
1888
|
+
});
|
|
1889
|
+
context.recordUsage?.(usage.receipt);
|
|
1890
|
+
}
|
|
1891
|
+
}
|
|
1892
|
+
/** Adapt a research definition to the common Analyst registry contract. */
|
|
1893
|
+
function createTraceAnalyst(definition, options) {
|
|
1894
|
+
validateDefinition(definition);
|
|
1895
|
+
const version = options.versionSuffix ? `${definition.version}+${options.versionSuffix}` : definition.version;
|
|
1896
|
+
const settlementTimeoutMs = validateUsageSettlementTimeout(options.settlementTimeoutMs);
|
|
1897
|
+
const limits = resolveTraceAnalystLimits(definition.limits);
|
|
1898
|
+
const engineIdentity = snapshotExactExecutionComponentIdentity({
|
|
1899
|
+
id: options.engine.id,
|
|
1900
|
+
version: options.engine.version,
|
|
1901
|
+
config: options.engine.executionConfig
|
|
1902
|
+
}, "createTraceAnalyst engine");
|
|
1903
|
+
return {
|
|
1904
|
+
id: definition.id,
|
|
1905
|
+
description: definition.description,
|
|
1906
|
+
inputKind: "trace-store",
|
|
1907
|
+
cost: {
|
|
1908
|
+
kind: "llm",
|
|
1909
|
+
...options.engine.model ? { models: [options.engine.model] } : {},
|
|
1910
|
+
settlement_timeout_ms: settlementTimeoutMs
|
|
1911
|
+
},
|
|
1912
|
+
version,
|
|
1913
|
+
executionConfig: {
|
|
1914
|
+
kind: "trace-analyst",
|
|
1915
|
+
model: options.engine.model ?? null,
|
|
1916
|
+
engine: options.engine.id,
|
|
1917
|
+
engine_identity: engineIdentity,
|
|
1918
|
+
instructions_digest: hashCanonical(definition.instructions.trim()),
|
|
1919
|
+
question: typeof definition.question === "string" ? hashCanonical(definition.question.trim()) : definition.question === void 0 ? "context-derived" : "version-bound",
|
|
1920
|
+
tool_group: definition.toolGroup,
|
|
1921
|
+
max_iterations: limits.maxIterations,
|
|
1922
|
+
max_llm_calls: limits.maxLlmCalls,
|
|
1923
|
+
max_tool_calls: limits.maxToolCalls,
|
|
1924
|
+
max_output_chars: limits.maxOutputChars,
|
|
1925
|
+
minimum_evidence_citations: definition.minimumEvidenceCitations ?? 1,
|
|
1926
|
+
require_structured_findings: definition.requireStructuredFindings ?? false,
|
|
1927
|
+
prepare_context: definition.prepareContext === void 0 ? "disabled" : "version-bound",
|
|
1928
|
+
post_process: definition.postProcess === void 0 ? "disabled" : "version-bound",
|
|
1929
|
+
evidence_verification: EVIDENCE_VERIFICATION_VERSION,
|
|
1930
|
+
settlement_timeout_ms: settlementTimeoutMs
|
|
1931
|
+
},
|
|
1932
|
+
async analyze(store, context) {
|
|
1933
|
+
const completed = await runTraceAnalyst({
|
|
1934
|
+
definition,
|
|
1935
|
+
engine: options.engine,
|
|
1936
|
+
store,
|
|
1937
|
+
context,
|
|
1938
|
+
settlementTimeoutMs: options.settlementTimeoutMs
|
|
1939
|
+
});
|
|
1940
|
+
return completed.findings.map((finding) => toAnalystFinding(definition, version, finding, {
|
|
1941
|
+
analysis_engine: options.engine.id,
|
|
1942
|
+
analysis_model: options.engine.model,
|
|
1943
|
+
analysis_model_calls: completed.modelCalls,
|
|
1944
|
+
analysis_tool_calls: completed.toolCalls,
|
|
1945
|
+
analysis_runtime: completed.runtime
|
|
1946
|
+
}));
|
|
1947
|
+
}
|
|
1948
|
+
};
|
|
1949
|
+
}
|
|
1950
|
+
async function acceptFindings(definition, submitted, store, context, minimumEvidenceCitations) {
|
|
1951
|
+
const expectedSubjects = KIND_EXPECTED_SUBJECTS[definition.id];
|
|
1952
|
+
const accepted = [];
|
|
1953
|
+
for (const row of submitted) {
|
|
1954
|
+
const parsed = parseRawFinding(row, context.log);
|
|
1955
|
+
if (!parsed) continue;
|
|
1956
|
+
const processed = definition.postProcess ? definition.postProcess(parsed, context) : parsed;
|
|
1957
|
+
if (!processed) continue;
|
|
1958
|
+
const validated = parseRawFinding(processed, context.log);
|
|
1959
|
+
if (!validated) continue;
|
|
1960
|
+
if (expectedSubjects && validated.subject !== void 0) {
|
|
1961
|
+
const subject = parseFindingSubject(validated.subject);
|
|
1962
|
+
if (subject === null || !expectedSubjects.includes(subject.kind)) {
|
|
1963
|
+
context.log?.("finding rejected: subject is not valid for analyst", {
|
|
1964
|
+
analyst_id: definition.id,
|
|
1965
|
+
subject: validated.subject,
|
|
1966
|
+
allowed: expectedSubjects
|
|
1967
|
+
});
|
|
1968
|
+
continue;
|
|
1969
|
+
}
|
|
1970
|
+
}
|
|
1971
|
+
const distinctEvidence = new Set(validated.evidence.map((citation) => citation.uri.trim())).size;
|
|
1972
|
+
if (distinctEvidence < minimumEvidenceCitations) {
|
|
1973
|
+
context.log?.("finding rejected: insufficient evidence citations", {
|
|
1974
|
+
analyst_id: definition.id,
|
|
1975
|
+
required: minimumEvidenceCitations,
|
|
1976
|
+
distinct: distinctEvidence
|
|
1977
|
+
});
|
|
1978
|
+
continue;
|
|
1979
|
+
}
|
|
1980
|
+
if (!await evidenceIsResolvable(validated, store, context)) continue;
|
|
1981
|
+
accepted.push(validated);
|
|
1982
|
+
}
|
|
1983
|
+
return accepted;
|
|
1984
|
+
}
|
|
1985
|
+
async function evidenceIsResolvable(finding, store, context) {
|
|
1986
|
+
const knownFindings = new Map([...context.priorFindings ?? [], ...context.upstreamFindings ?? []].map((entry) => [entry.finding_id, entry]));
|
|
1987
|
+
for (const citation of finding.evidence) {
|
|
1988
|
+
if (citation.excerpt !== void 0 && citation.excerpt.trim().length < MINIMUM_EXCERPT_LENGTH) {
|
|
1989
|
+
rejectEvidence(context, citation.uri, "excerpt is too short to verify");
|
|
1990
|
+
return false;
|
|
1991
|
+
}
|
|
1992
|
+
const traceLocation = parseTraceSpanEvidenceUri(citation.uri);
|
|
1993
|
+
if (traceLocation) {
|
|
1994
|
+
const storeContext = context.signal ? { signal: context.signal } : void 0;
|
|
1995
|
+
if (!(await store.hasSpans({
|
|
1996
|
+
trace_id: traceLocation.traceId,
|
|
1997
|
+
span_ids: [traceLocation.spanId]
|
|
1998
|
+
}, storeContext)).includes(traceLocation.spanId)) {
|
|
1999
|
+
rejectEvidence(context, citation.uri, "trace span does not exist");
|
|
2000
|
+
return false;
|
|
2001
|
+
}
|
|
2002
|
+
if (citation.excerpt !== void 0) {
|
|
2003
|
+
const span = (await store.viewSpans({
|
|
2004
|
+
trace_id: traceLocation.traceId,
|
|
2005
|
+
span_ids: [traceLocation.spanId]
|
|
2006
|
+
}, storeContext)).spans.find((entry) => entry.span_id === traceLocation.spanId);
|
|
2007
|
+
if (!span || !containsExactText([span.attributes, span.status_message], citation.excerpt)) {
|
|
2008
|
+
rejectEvidence(context, citation.uri, "excerpt is not present in the cited span content");
|
|
2009
|
+
return false;
|
|
2010
|
+
}
|
|
2011
|
+
}
|
|
2012
|
+
continue;
|
|
2013
|
+
}
|
|
2014
|
+
const findingId = parseFindingEvidenceUri(citation.uri);
|
|
2015
|
+
const referenced = findingId ? knownFindings.get(findingId) : void 0;
|
|
2016
|
+
if (!referenced) {
|
|
2017
|
+
rejectEvidence(context, citation.uri, "citation is not a supplied finding or trace span");
|
|
2018
|
+
return false;
|
|
2019
|
+
}
|
|
2020
|
+
if (citation.excerpt !== void 0 && !containsExactText([
|
|
2021
|
+
referenced.claim,
|
|
2022
|
+
referenced.rationale,
|
|
2023
|
+
referenced.recommended_action,
|
|
2024
|
+
referenced.validation_plan,
|
|
2025
|
+
referenced.evidence_refs?.map((evidence) => evidence.excerpt)
|
|
2026
|
+
], citation.excerpt)) {
|
|
2027
|
+
rejectEvidence(context, citation.uri, "excerpt is not present in the cited finding content");
|
|
2028
|
+
return false;
|
|
2029
|
+
}
|
|
2030
|
+
}
|
|
2031
|
+
return true;
|
|
2032
|
+
}
|
|
2033
|
+
function parseFindingEvidenceUri(uri) {
|
|
2034
|
+
const match = /^finding:\/\/([^/?#]+)$/.exec(uri);
|
|
2035
|
+
if (!match) return null;
|
|
2036
|
+
try {
|
|
2037
|
+
return decodeURIComponent(match[1]) || null;
|
|
2038
|
+
} catch {
|
|
2039
|
+
return null;
|
|
2040
|
+
}
|
|
2041
|
+
}
|
|
2042
|
+
/** Excerpts must quote enough content to be checkable evidence, not an
|
|
2043
|
+
* incidental substring of an id, status, or timestamp. */
|
|
2044
|
+
const MINIMUM_EXCERPT_LENGTH = 8;
|
|
2045
|
+
/** Bumped whenever the evidence-acceptance rules change, so two differently
|
|
2046
|
+
* strict builds cannot seal identical execution plans. */
|
|
2047
|
+
const EVIDENCE_VERIFICATION_VERSION = "resolvable-excerpt-v1";
|
|
2048
|
+
/** Matches only within the passed content-bearing values — callers must not
|
|
2049
|
+
* hand this whole spans or findings, or identifier fields become quotable. */
|
|
2050
|
+
function containsExactText(value, expected, depth = 0) {
|
|
2051
|
+
if (!expected || depth > 20) return false;
|
|
2052
|
+
if (typeof value === "string") return value.includes(expected);
|
|
2053
|
+
if (Array.isArray(value)) return value.some((entry) => containsExactText(entry, expected, depth + 1));
|
|
2054
|
+
if (typeof value === "object" && value !== null) return Object.values(value).some((entry) => containsExactText(entry, expected, depth + 1));
|
|
2055
|
+
return false;
|
|
2056
|
+
}
|
|
2057
|
+
function rejectEvidence(context, uri, reason) {
|
|
2058
|
+
context.log?.("finding rejected: unresolved evidence", {
|
|
2059
|
+
uri,
|
|
2060
|
+
reason
|
|
2061
|
+
});
|
|
2062
|
+
}
|
|
2063
|
+
function validateDefinition(definition) {
|
|
2064
|
+
for (const [name, value] of [
|
|
2065
|
+
["id", definition.id],
|
|
2066
|
+
["description", definition.description],
|
|
2067
|
+
["area", definition.area],
|
|
2068
|
+
["version", definition.version],
|
|
2069
|
+
["instructions", definition.instructions]
|
|
2070
|
+
]) if (typeof value !== "string" || !value.trim()) throw new TypeError(`trace analyst ${name} must be a non-empty string`);
|
|
2071
|
+
const minimumEvidenceCitations = definition.minimumEvidenceCitations ?? 1;
|
|
2072
|
+
if (!Number.isSafeInteger(minimumEvidenceCitations) || minimumEvidenceCitations < 1) throw new TypeError("minimumEvidenceCitations must be a positive safe integer");
|
|
2073
|
+
resolveTraceAnalystLimits(definition.limits);
|
|
2074
|
+
}
|
|
2075
|
+
function deriveQuestion(context, definition) {
|
|
2076
|
+
const base = (typeof definition.question === "function" ? definition.question(context) : definition.question)?.trim() || `Analyze this trace dataset and report ${definition.area} findings. ${definition.description}`;
|
|
2077
|
+
const focus = context.tags?.focus?.trim();
|
|
2078
|
+
return focus ? `${base}\nFocus: ${focus}` : base;
|
|
2079
|
+
}
|
|
2080
|
+
function toAnalystFinding(definition, version, raw, metadata) {
|
|
2081
|
+
return makeFinding({
|
|
2082
|
+
analyst_id: definition.id,
|
|
2083
|
+
area: definition.area,
|
|
2084
|
+
subject: raw.subject,
|
|
2085
|
+
claim: raw.claim,
|
|
2086
|
+
rationale: raw.rationale,
|
|
2087
|
+
severity: raw.severity,
|
|
2088
|
+
confidence: raw.confidence,
|
|
2089
|
+
evidence_refs: evidenceRefsFromRawFinding(raw),
|
|
2090
|
+
recommended_action: raw.recommended_action,
|
|
2091
|
+
metadata: {
|
|
2092
|
+
definition_version: version,
|
|
2093
|
+
...metadata
|
|
2094
|
+
}
|
|
2095
|
+
});
|
|
2096
|
+
}
|
|
2097
|
+
function renderPriorFindings(prior) {
|
|
2098
|
+
if (!prior || prior.length === 0) return "";
|
|
2099
|
+
const maxRows = 40;
|
|
2100
|
+
const rows = prior.slice(0, maxRows).map((finding) => {
|
|
2101
|
+
const subject = finding.subject ? ` [${finding.subject}]` : "";
|
|
2102
|
+
return `- id=${finding.finding_id} ${finding.severity}${subject} ${truncateForContext(finding.claim, 160)}`;
|
|
2103
|
+
});
|
|
2104
|
+
if (prior.length > maxRows) rows.push(`- ${prior.length - maxRows} older findings omitted`);
|
|
2105
|
+
return [
|
|
2106
|
+
"PRIOR FINDINGS:",
|
|
2107
|
+
"Reuse a matching finding id through id_basis and raise confidence only when current evidence confirms recurrence.",
|
|
2108
|
+
...rows
|
|
2109
|
+
].join("\n");
|
|
2110
|
+
}
|
|
2111
|
+
function renderUpstreamFindings(upstream) {
|
|
2112
|
+
if (!upstream || upstream.length === 0) return "";
|
|
2113
|
+
const maxRows = 40;
|
|
2114
|
+
const rows = upstream.slice(0, maxRows).map((finding) => {
|
|
2115
|
+
const subject = finding.subject ? ` [${finding.subject}]` : "";
|
|
2116
|
+
const evidence = finding.evidence_refs[0]?.uri ? ` evidence=${truncateForContext(finding.evidence_refs[0].uri, 120)}` : "";
|
|
2117
|
+
return `- id=${finding.finding_id} source=${finding.analyst_id} ${finding.severity}${subject} claim=${truncateForContext(finding.claim, 160)}${evidence}`;
|
|
2118
|
+
});
|
|
2119
|
+
if (upstream.length > maxRows) rows.push(`- ${upstream.length - maxRows} additional upstream findings omitted`);
|
|
2120
|
+
return [
|
|
2121
|
+
"UPSTREAM FINDINGS:",
|
|
2122
|
+
"Build on these findings instead of repeating them. Cite dependencies as finding://<id>.",
|
|
2123
|
+
...rows
|
|
2124
|
+
].join("\n");
|
|
2125
|
+
}
|
|
2126
|
+
function truncateForContext(value, max) {
|
|
2127
|
+
if (value.length <= max) return value;
|
|
2128
|
+
return `${value.slice(0, max - 3).trimEnd()}...`;
|
|
2129
|
+
}
|
|
2130
|
+
//#endregion
|
|
2131
|
+
export { spanEpochMillis as $, coerceJson as A, renderFindingSubject as B, TraceNotFoundError as C, RawAnalystFindingSchema as D, RawAnalystEvidenceSchema as E, FINDING_SUBJECT_SYNTAX as F, resolveTraceAnalystLimits as G, snapshotExactExecutionPlan as H, FindingSubjectStringSchema as I, extractOtlpAttributes as J, asString as K, KIND_EXPECTED_SUBJECTS as L, stripCodeFences as M, FINDING_SUBJECT_GRAMMAR_PROMPT as N, evidenceRefsFromRawFinding as O, FINDING_SUBJECT_KINDS as P, readOtlpStatus as Q, findingSubjectGrammarPromptFor as R, TraceFileTooLargeError as S, RAW_FINDING_SCHEMA_PROMPT as T, deepFreezeCanonicalJson as U, snapshotExactExecutionComponentIdentity as V, DEFAULT_TRACE_ANALYST_LIMITS as W, inferOtlpKind as X, firstStringAttr as Y, projectOtlpFlatLine as Z, TraceAnalysisLimitError as _, buildTraceToolsForGroup as a, TraceFileMalformedError as b, traceAnalystFunctionGroup as c, truncateForBudget as d, stringField as et, validateInteger as f, SpanNotFoundError as g, TRACE_ANALYSIS_LIMITS as h, runTraceAnalyst as i, traceSpanKindToOpenInferenceKind as it, coerceToFindingRows as j, parseRawFinding as k, createBoundedTraceAnalysisStore as l, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as m, renderPriorFindings as n, classifyOtlpSpanRole as nt, TRACE_ANALYST_TOOL_NAMESPACE as o, DEFAULT_TRACE_ANALYST_BUDGETS as p, compareSpanTime as q, renderUpstreamFindings as r, isOtlpModelCall as rt, buildTraceAnalysisToolDescriptors as s, createTraceAnalyst as t, applyToolSpanOtlpAttributes as tt, compileSearchRegex as u, TraceAnalysisStoreContractError as v, ANALYST_SEVERITIES as w, TraceFileMissingError as x, TraceAnalysisValidationError as y, parseFindingSubject as z };
|
|
2132
|
+
|
|
2133
|
+
//# sourceMappingURL=kind-factory-CFxA0JQX.js.map
|