opentel-mcp 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +87 -0
- package/README.md +138 -2
- package/package.json +1 -1
- package/src/config.js +15 -0
- package/src/cost/calculator.js +66 -0
- package/src/error-recording/config.js +74 -0
- package/src/error-recording/types.d.ts +55 -0
- package/src/fingerprint/attributes.js +16 -0
- package/src/fingerprint/classify/validation-paths.js +55 -1
- package/src/fingerprint/compose.js +27 -1
- package/src/fingerprint/types.d.ts +1 -1
- package/src/index.d.ts +31 -0
- package/src/instrument.js +273 -25
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,92 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.13.0
|
|
4
|
+
|
|
5
|
+
Closes the two open items from `docs/known-gaps.md` entry 10 (a
|
|
6
|
+
raw-content audit of every attribute/span event this package emits) that
|
|
7
|
+
needed a design decision before a fix — raw exception content on
|
|
8
|
+
`recordException`/`setStatus`, and an unvalidated tool-result model field
|
|
9
|
+
reaching `mcp.tool.model`/`gen_ai.response.model` — plus the two
|
|
10
|
+
lower-risk fixes from the same entry that didn't need one. Full design:
|
|
11
|
+
ADR 019 (`docs/adr/019-raw-content-on-spans.md`). See the README's new
|
|
12
|
+
"What this library records" and "Error recording" sections for the
|
|
13
|
+
operator-facing consequence of each.
|
|
14
|
+
|
|
15
|
+
### Added — `errorRecording.mode` config (ADR 019 Part 1)
|
|
16
|
+
|
|
17
|
+
- New top-level `errorRecording` option, sibling to `fingerprinting` /
|
|
18
|
+
`costTracking` / `thrashDetection` / `schemaDrift`: `'full'` (default,
|
|
19
|
+
byte-for-byte unchanged from every prior release —
|
|
20
|
+
`recordException(err)` + `setStatus({ message: err.message })` with the
|
|
21
|
+
raw error), `'normalized'` (reuses the existing
|
|
22
|
+
`normalizeMessage()`/`parseAndNormalizeStack()` fingerprinting pipeline
|
|
23
|
+
— no new scrubbing logic — to strip known-sensitive-shaped substrings
|
|
24
|
+
from the message and the local `cwd` prefix from the stack, without
|
|
25
|
+
mutating the original `err`, which both call sites still rethrow), or
|
|
26
|
+
`'none'` (records neither — the same `setStatus({ code: ERROR })`
|
|
27
|
+
no-message pattern already used for tool-level `isError: true`
|
|
28
|
+
failures). Applies to both thrown-exception paths, `tools/call` and
|
|
29
|
+
`tools/list`.
|
|
30
|
+
- New `OTEL_MCP_ERROR_RECORDING_MODE` env var — same
|
|
31
|
+
option-then-env-then-default precedence, and the same silent fallback
|
|
32
|
+
to the default on an unrecognized value, as every other `OTEL_MCP_*`
|
|
33
|
+
config.
|
|
34
|
+
- `error.type`/`exception.type` (both read from `err.name`) are now
|
|
35
|
+
capped at 128 characters unconditionally, in every mode — the same cap
|
|
36
|
+
`mcp.failure.error_class` uses below, reusing its exported constant
|
|
37
|
+
rather than a second, possibly-drifting copy.
|
|
38
|
+
- Default stays `'full'` for all of `0.x`; ADR 019 Part 1 records the
|
|
39
|
+
intent to flip it to `'normalized'` at `1.0`, not before — not decided
|
|
40
|
+
in this release.
|
|
41
|
+
|
|
42
|
+
### Added — `mcp.tool.model` / `gen_ai.response.model` validation (ADR 019 Part 2)
|
|
43
|
+
|
|
44
|
+
**This is new behavior that can change what a call reports, not just a
|
|
45
|
+
new diagnostic.** Before this release, a tool result's `model` field
|
|
46
|
+
reached the span completely unvalidated, whatever it was. If you have a
|
|
47
|
+
provider/deployment whose model identifiers use a character outside
|
|
48
|
+
`[A-Za-z0-9._:/@-]`, or that (implausibly, but possibly) exceed 256
|
|
49
|
+
characters, upgrading will make `mcp.tool.model`/`gen_ai.response.model`
|
|
50
|
+
disappear from those calls' spans and `mcp.tool.pricing_status` flip from
|
|
51
|
+
whatever it was to `"unknown"` — even for an otherwise-legitimate, real
|
|
52
|
+
model id. The allowlist was deliberately built generous (see below) and
|
|
53
|
+
verified against known provider conventions, but it's still new, and a
|
|
54
|
+
one-time `diag.warn()` names when this happens (shape only, never the
|
|
55
|
+
value) so it's discoverable rather than a silent metric/span change.
|
|
56
|
+
|
|
57
|
+
- A tool result's declared model field is now checked against a 256-
|
|
58
|
+
character length cap and a `[A-Za-z0-9._:/@-]` allowlist — verified
|
|
59
|
+
against every `DEFAULT_PRICING` key and `normalizeModelName()`'s
|
|
60
|
+
documented `provider/model` input contract, deliberately generous —
|
|
61
|
+
before it can reach `mcp.tool.model`, `gen_ai.response.model`,
|
|
62
|
+
`calculateCost()`, or either metric label
|
|
63
|
+
(`mcp.tool.tokens.total`/`mcp.tool.cost.total`).
|
|
64
|
+
- A rejected value is never a silent drop: `mcp.tool.pricing_status` is
|
|
65
|
+
set to `"unknown"` (the same status a legitimately unrecognized model
|
|
66
|
+
already produces), and a one-time `diag.warn()` fires per
|
|
67
|
+
`instrumentMcpServer()` call, reporting shape only — length, and which
|
|
68
|
+
check failed — never the rejected value itself.
|
|
69
|
+
- `budgetTracker.recordUnpriced()` now receives the validated (possibly
|
|
70
|
+
`undefined`) model rather than the raw tool-result value, closing a
|
|
71
|
+
second leak path through its own pre-existing warning that would
|
|
72
|
+
otherwise have echoed a rejected value verbatim.
|
|
73
|
+
- `costTracking.pricing`/`pricingTable` override keys are unaffected —
|
|
74
|
+
operator-authored config, never subject to this gate.
|
|
75
|
+
|
|
76
|
+
### Fixed — `mcp.failure.error_class` uncapped length, `mcp.failure.validation_paths` dynamic-key leak (known-gaps entry 10)
|
|
77
|
+
|
|
78
|
+
- `mcp.failure.error_class` is now capped at 128 characters
|
|
79
|
+
(`fingerprint/compose.js`'s new `MAX_ERROR_CLASS_LENGTH`) — length-
|
|
80
|
+
bounded only, not pattern-scrubbed; a non-string `.name` is coerced to
|
|
81
|
+
a string before capping rather than thrown. ADR 004's note calling the
|
|
82
|
+
underlying value "low-cardinality" is updated to flag that as an
|
|
83
|
+
assumption, not an enforced property.
|
|
84
|
+
- `mcp.failure.validation_paths` format 1 (the raw Zod issues array, SDK
|
|
85
|
+
≤1.29.0) now redacts a non-identifier-shaped path segment — a
|
|
86
|
+
`z.record()` schema's runtime key — to the placeholder `<KEY>` instead
|
|
87
|
+
of surfacing it verbatim, matching the identifier-only gate formats 2/3
|
|
88
|
+
already applied. Numeric (array-index) segments are never redacted.
|
|
89
|
+
|
|
3
90
|
## 0.12.0
|
|
4
91
|
|
|
5
92
|
Agent Thrash Detection gains a new, narrower session-identity fallback for
|
package/README.md
CHANGED
|
@@ -461,6 +461,59 @@ examples/fingerprint-demo.js`). Not yet wired through
|
|
|
461
461
|
`computeFingerprint` directly rather than configuring the automatic
|
|
462
462
|
per-call-site wrapping; tracked in the roadmap below.
|
|
463
463
|
|
|
464
|
+
## Error recording (v0.13.0+)
|
|
465
|
+
|
|
466
|
+
Every thrown `tools/call`/`tools/list` error goes through
|
|
467
|
+
`span.recordException(err)` (an OpenTelemetry SDK method, not one of this
|
|
468
|
+
library's own attributes) plus `span.setStatus({ code: ERROR, message:
|
|
469
|
+
err.message })` — unconditionally, whether or not `fingerprinting` is
|
|
470
|
+
enabled. By default that means `err.message` and `err.stack` land on the
|
|
471
|
+
span exactly as thrown. `errorRecording.mode` controls this:
|
|
472
|
+
|
|
473
|
+
| Mode | `exception.message` / status message | `exception.stacktrace` | When to use |
|
|
474
|
+
|---|---|---|---|
|
|
475
|
+
| `'full'` (default) | Raw `err.message`, unmodified | Raw `err.stack`, unmodified | Today's behavior, unchanged — matches what every other OTel-instrumented library in the same trace does for the same kind of event |
|
|
476
|
+
| `'normalized'` | `normalizeMessage(err.message)` — the exact scrubbing pipeline (`src/fingerprint/normalize/message.js`) fingerprinting already runs before hashing: UUIDs, emails, URLs, IPs, timestamps, filesystem paths, hex runs, quoted ids | Reconstructed from `parseAndNormalizeStack()` (`src/fingerprint/normalize/stack.js`) — keeps every function name/file/line, strips only the local `cwd` prefix (and collapses `node_modules` package versions) | Tool results come from third-party or unaudited MCP servers and you want the same scrubbing fingerprinting already trusts, applied to the raw exception content too |
|
|
477
|
+
| `'none'` | Not set — `span.setStatus({ code: ERROR })` with no message, the same pattern already used for tool-level `isError: true` failures | Not set | You rely entirely on `mcp.failure.*` (category/fingerprint/signature — already hashed/normalized) and don't want any free-text exception content on the span at all |
|
|
478
|
+
|
|
479
|
+
No mode mutates the original `err` — both call sites rethrow it
|
|
480
|
+
afterward, so `'normalized'`/`'none'` build the exception event
|
|
481
|
+
independently rather than editing `err.message`/`err.stack` in place.
|
|
482
|
+
`error.type`/`exception.type` (`err.name`) is capped at 128 characters
|
|
483
|
+
unconditionally in every mode — the same cap `mcp.failure.error_class`
|
|
484
|
+
uses — since a length cap on a class-identifier field costs a
|
|
485
|
+
well-behaved tool nothing, unlike message/stack content.
|
|
486
|
+
|
|
487
|
+
**`'normalized'` is targeted scrubbing, not general-purpose redaction —
|
|
488
|
+
read this before treating it as a PII filter.** `normalizeMessage()`
|
|
489
|
+
matches specific, structured shapes: UUIDs, email addresses, URLs,
|
|
490
|
+
IPv4/IPv6 addresses, ISO-8601/Unix timestamps, filesystem paths, long hex
|
|
491
|
+
runs, and quoted alphanumeric ids (8–64 chars, mixed letters/digits). It
|
|
492
|
+
does not recognize sensitive content in general. An API key in a format
|
|
493
|
+
none of those patterns match (a bare, unquoted token with no digit in it,
|
|
494
|
+
or a custom prefix scheme), or a customer's name embedded in ordinary
|
|
495
|
+
prose ("could not process request for Jane Smith"), passes through
|
|
496
|
+
`'normalized'` mode completely unchanged — identical to what `'full'`
|
|
497
|
+
mode would put on the span. Treat `'normalized'` as "the same scrubbing
|
|
498
|
+
fingerprinting already trusts for hashing," not as a guarantee that
|
|
499
|
+
whatever a tool's error messages contain is safe to record; if a tool's
|
|
500
|
+
errors routinely carry sensitive free text these patterns don't happen to
|
|
501
|
+
match, `'none'` is the only mode that keeps message/stack content off the
|
|
502
|
+
span entirely.
|
|
503
|
+
|
|
504
|
+
**Default stays `'full'` through all of `0.x`.** Changing it would alter
|
|
505
|
+
what every trace backend renders for the single highest-traffic failure
|
|
506
|
+
path in this library, silently, for every existing deployment that
|
|
507
|
+
doesn't opt in — see ADR 019 Part 1 (`docs/adr/019-raw-content-on-spans.md`)
|
|
508
|
+
for the full argument, including why the default is expected to flip to
|
|
509
|
+
`'normalized'` at `1.0`, not before.
|
|
510
|
+
|
|
511
|
+
### Configuration
|
|
512
|
+
|
|
513
|
+
| Option | Env var | Type | Default | Description |
|
|
514
|
+
|---|---|---|---|---|
|
|
515
|
+
| `mode` | `OTEL_MCP_ERROR_RECORDING_MODE` | `'full'` \| `'normalized'` \| `'none'` | `'full'` | See table above. An unrecognized value falls back to `'full'` silently, same as every other `OTEL_MCP_*` env var |
|
|
516
|
+
|
|
464
517
|
## Cost & Token Attribution (v0.5.0)
|
|
465
518
|
|
|
466
519
|
MCP tools increasingly wrap LLM calls themselves — a tool that
|
|
@@ -569,7 +622,7 @@ stricter, not the runtime.
|
|
|
569
622
|
| `mcp.tool.tokens.input` | Custom | Input tokens consumed | 1000 |
|
|
570
623
|
| `mcp.tool.tokens.output` | Custom | Output tokens produced | 500 |
|
|
571
624
|
| `mcp.tool.tokens.total` | Custom | input + output | 1500 |
|
|
572
|
-
| `mcp.tool.model` | Custom | Detected model name | "claude-sonnet-5" |
|
|
625
|
+
| `mcp.tool.model` | Custom | Detected model name, gated by a length/shape check (v0.13.0) — see below | "claude-sonnet-5" |
|
|
573
626
|
| `gen_ai.response.model` | Standard (GenAI semconv)[^5] | Same value as `mcp.tool.model`, co-emitted for dashboard compatibility | "claude-sonnet-5" |
|
|
574
627
|
| `mcp.tool.pricing_status` | Custom[^7] | `"known"` \| `"unknown"` \| `"user_override"` — set whenever token usage was extracted, even with no model detected | "known" |
|
|
575
628
|
| `mcp.tool.cost.usd` | Custom | Estimated cost, from `calculateCost()` | 0.0105 |
|
|
@@ -589,6 +642,37 @@ attributes only appear when a cost was calculated *and* a configured
|
|
|
589
642
|
limit was crossed. Source of truth: `src/attributes.js` and
|
|
590
643
|
`src/instrument.js`'s `applyCostAttribution()`.
|
|
591
644
|
|
|
645
|
+
### Model identifier validation (v0.13.0)
|
|
646
|
+
|
|
647
|
+
A tool result's declared model field (`result.model`, `result.usage.model`,
|
|
648
|
+
`result._meta.model`, or the same read out of JSON in
|
|
649
|
+
`result.content[0].text`) is tool-result content, not library-computed
|
|
650
|
+
metadata like everything else `applyCostAttribution()` puts on a span — so
|
|
651
|
+
before it can reach `mcp.tool.model`, `gen_ai.response.model`,
|
|
652
|
+
`calculateCost()`, or a metric label
|
|
653
|
+
(`mcp.tool.tokens.total`/`mcp.tool.cost.total`), it has to pass a length
|
|
654
|
+
cap (256 characters) and character allowlist
|
|
655
|
+
(`/^[A-Za-z0-9._:/@-]{1,256}$/`). The allowlist is deliberately generous —
|
|
656
|
+
verified against every `DEFAULT_PRICING` key *and* `normalizeModelName()`'s
|
|
657
|
+
documented `provider/model` input contract (e.g.
|
|
658
|
+
`"Anthropic/Claude-Opus-4-7"`), plus headroom for conventions like a full
|
|
659
|
+
Bedrock ARN or a Vertex AI `@version` suffix — because silently rejecting a
|
|
660
|
+
legitimate model name is a worse failure than admitting a few characters
|
|
661
|
+
no known provider convention actually uses.
|
|
662
|
+
|
|
663
|
+
**A rejected value is never a silent drop.** `mcp.tool.pricing_status` is
|
|
664
|
+
set to `"unknown"` — the same status a legitimately unrecognized model
|
|
665
|
+
already produces (see the footnote above) — and a `diag.warn()` fires
|
|
666
|
+
once per `instrumentMcpServer()` call, reporting shape only (length, and
|
|
667
|
+
whether length or character class failed), **never the rejected value
|
|
668
|
+
itself**: echoing a malformed model id into log output would just move
|
|
669
|
+
this exact problem from spans into logs instead of closing it. Token
|
|
670
|
+
counts (`mcp.tool.tokens.*`) are unaffected either way — they're set
|
|
671
|
+
before this gate runs. `costTracking.pricing`/`pricingTable` override keys
|
|
672
|
+
(operator-authored config) are never subject to this check — only
|
|
673
|
+
tool-result content is. Full design: ADR 019 Part 2
|
|
674
|
+
(`docs/adr/019-raw-content-on-spans.md`).
|
|
675
|
+
|
|
592
676
|
### Metrics
|
|
593
677
|
|
|
594
678
|
Two more `mcp.tool.*` metrics, via the same API-only pattern as the four
|
|
@@ -1746,6 +1830,56 @@ useful today, for free, to any client that already sets `_meta` in this
|
|
|
1746
1830
|
shape; the client-side half is tracked as future work, not implied as
|
|
1747
1831
|
solved by this release.
|
|
1748
1832
|
|
|
1833
|
+
## What this library records — read this before routing spans anywhere sensitive data isn't already allowed
|
|
1834
|
+
|
|
1835
|
+
This is a description of what happens today, not a claim about a specific
|
|
1836
|
+
threat model this library has verified it covers.
|
|
1837
|
+
|
|
1838
|
+
opentel-mcp forwards what your tools and their thrown exceptions actually
|
|
1839
|
+
produce. It does not invent, infer, or independently verify tool-result
|
|
1840
|
+
content — if a tool's handler throws `new Error('user ' + email + ' not
|
|
1841
|
+
found')`, or a tool result puts a credential in a field this library
|
|
1842
|
+
reads, that content is exactly what reaches your telemetry backend. Three
|
|
1843
|
+
channels carry it:
|
|
1844
|
+
|
|
1845
|
+
- **`span.recordException(err)` / `span.setStatus({ message })`** on every
|
|
1846
|
+
thrown `tools/call`/`tools/list` error — `exception.message` and
|
|
1847
|
+
`exception.stacktrace` are set from `err.message`/`err.stack` verbatim
|
|
1848
|
+
by default. `errorRecording.mode` controls this; see "Error recording"
|
|
1849
|
+
above.
|
|
1850
|
+
- **`mcp.tool.model` / `gen_ai.response.model`** — a tool result's own
|
|
1851
|
+
declared model field, admitted once it passes a length/character-shape
|
|
1852
|
+
check (v0.13.0). The check bounds shape, not content: a well-formed but
|
|
1853
|
+
still arbitrary string from a tool result reaches the span. See "Cost &
|
|
1854
|
+
Token Attribution" → "Model identifier validation" above.
|
|
1855
|
+
- **`mcp.failure.error_class`** (length-capped at 128 characters, not
|
|
1856
|
+
pattern-scrubbed) and **`mcp.failure.validation_paths`** (a schema's
|
|
1857
|
+
dynamic/record keys replaced with the placeholder `<KEY>`) — both
|
|
1858
|
+
already hardened; see `docs/known-gaps.md` entry 10.
|
|
1859
|
+
|
|
1860
|
+
**This isn't unique to opentel-mcp.** OpenTelemetry's own semantic
|
|
1861
|
+
conventions mark `exception.message` as an attribute that "may contain
|
|
1862
|
+
sensitive information" and still specify recording it by default —
|
|
1863
|
+
every other OTel-instrumented library sharing the same trace (an HTTP
|
|
1864
|
+
client, a DB driver, a queue consumer) records `err.message`/`err.stack`
|
|
1865
|
+
the same way, unscrubbed, through the same `recordException()` call.
|
|
1866
|
+
opentel-mcp matches that ecosystem default rather than silently diverging
|
|
1867
|
+
from it. What it adds on top: `errorRecording.mode` lets you choose
|
|
1868
|
+
`'normalized'` (targeted scrubbing — the same structured patterns
|
|
1869
|
+
fingerprinting matches: UUIDs, emails, URLs, IPs, timestamps, paths,
|
|
1870
|
+
quoted ids; **not** a general-purpose PII filter, so freeform sensitive
|
|
1871
|
+
text — a customer's name in prose, an API key in a format none of those
|
|
1872
|
+
patterns match — still reaches the span unchanged, see "Error recording"
|
|
1873
|
+
above) or `'none'` (record neither) instead of `'full'` — a level of
|
|
1874
|
+
operator control most peer instrumentations don't offer for this path.
|
|
1875
|
+
Full reasoning, including why the default doesn't change in this
|
|
1876
|
+
release: ADR 019 (`docs/adr/019-raw-content-on-spans.md`).
|
|
1877
|
+
|
|
1878
|
+
This is a data-handling property of an observability library, not a
|
|
1879
|
+
security control opentel-mcp is claiming to provide — it doesn't
|
|
1880
|
+
authenticate, encrypt, or restrict who can read your traces; that's your
|
|
1881
|
+
tracing backend's job.
|
|
1882
|
+
|
|
1749
1883
|
## Configuration
|
|
1750
1884
|
|
|
1751
1885
|
All options passed to `instrumentMcpServer(server, options)`. Source of
|
|
@@ -1762,9 +1896,11 @@ truth: `src/config.js`.
|
|
|
1762
1896
|
| `costTracking` | object | see below | Controls cost/token attribution[^9] — see "Cost & Token Attribution" above |
|
|
1763
1897
|
| `thrashDetection` | object | see below | Controls Agent Thrash Detection[^10] — see "Agent Thrash Detection" above. Requires `fingerprinting: true` |
|
|
1764
1898
|
| `schemaDrift` | object | see below | Controls tool schema drift detection[^11] — see "Tool schema drift detection" above. `enabled: true` by default changes `instrumentMcpServer()`'s throw behavior on upgrade — see that section's "Behavior change on upgrade" |
|
|
1899
|
+
| `errorRecording` | object | see below | Controls what raw exception content reaches `recordException()`/`setStatus()`[^12] — see "Error recording" above |
|
|
1765
1900
|
|
|
1766
1901
|
[^10]: `{ enabled?, threshold?, windowMs?, maxTrackedKeys?, entryTtlMs?, reEmitAfter?, assumeSingleSession? }`, all fields optional, individually defaulted, and individually overridable via an `OTEL_MCP_THRASH_*` env var — see "Agent Thrash Detection" → "Configuration" above for the full table.
|
|
1767
1902
|
[^11]: `{ enabled?, maxTrackedTools? }`, all fields optional, individually defaulted, and individually overridable via an `OTEL_MCP_SCHEMA_DRIFT_*` env var — see "Tool schema drift detection" → "Configuration" above for the full table.
|
|
1903
|
+
[^12]: `{ mode?: 'full' | 'normalized' | 'none' }`, defaulted to `'full'`, individually overridable via `OTEL_MCP_ERROR_RECORDING_MODE` — see "Error recording" → "Configuration" above for the full table.
|
|
1768
1904
|
|
|
1769
1905
|
[^9]: `{ enabled?: boolean; pricingTable?: PricingTable; extractor?: UsageExtractor; budget?: { perSessionUsd?: number; perToolUsd?: number } }`, all fields optional and individually defaulted — `{ enabled: true, pricingTable: DEFAULT_PRICING, extractor: defaultExtractor }` with budget tracking off.
|
|
1770
1906
|
[^2]: Required only when `setupNodeSdk` is `true`. Has no effect otherwise — the host app's registered `TracerProvider` owns the resource; passing it anyway logs a one-time `diag.warn`.
|
|
@@ -2181,7 +2317,7 @@ pragmatic choice rather than a spec-pure one.
|
|
|
2181
2317
|
revision 2026-07-28, v0.10.0+; see "MCP v2 support" above for what's
|
|
2182
2318
|
covered and `docs/known-gaps.md` entries 6 and 8 for what isn't yet)
|
|
2183
2319
|
- @opentelemetry/api ^1.9.0
|
|
2184
|
-
-
|
|
2320
|
+
- 948 tests, 944 passing + 4 intentionally skipped (`npm test`) — see `test/`
|
|
2185
2321
|
- `npm run typecheck` (`tsc --noEmit`) type-checks the public `.d.ts`
|
|
2186
2322
|
surface (`src/index.d.ts` and friends) — see CONTRIBUTING.md
|
|
2187
2323
|
|
package/package.json
CHANGED
package/src/config.js
CHANGED
|
@@ -9,6 +9,7 @@ import { defaultExtractor } from './cost/extractor.js';
|
|
|
9
9
|
import { normalizeModelName } from './cost/calculator.js';
|
|
10
10
|
import { resolveThrashConfig } from './thrash/config.js';
|
|
11
11
|
import { resolveSchemaDriftConfig } from './schema-drift/config.js';
|
|
12
|
+
import { resolveErrorRecordingConfig } from './error-recording/config.js';
|
|
12
13
|
|
|
13
14
|
/**
|
|
14
15
|
* @typedef {object} CostTrackingOptions
|
|
@@ -84,6 +85,19 @@ import { resolveSchemaDriftConfig } from './schema-drift/config.js';
|
|
|
84
85
|
* `enabled: false` (or the default when this option is omitted) is a true no-op: tools/list is not
|
|
85
86
|
* wrapped at all, unlike thrashDetection/costTracking whose disabled state still wraps tools/call for
|
|
86
87
|
* other reasons and merely skips inner logic.
|
|
88
|
+
* @property {Partial<import('./error-recording/config.js').ErrorRecordingConfig>} [errorRecording] - Controls
|
|
89
|
+
* what a THROWN exception (not a tool-level `isError: true` result, which never carries a JS Error and is
|
|
90
|
+
* unaffected) puts on the mcp.tool.call / tools/list span (ADR 019, docs/adr/019-raw-content-on-spans.md
|
|
91
|
+
* Part 1, v0.13.0): `{ mode: 'full' | 'normalized' | 'none' }`. Defaults to `{ mode: 'full' }` —
|
|
92
|
+
* byte-identical to every release before v0.13.0 (span.recordException(err) plus setStatus({ message:
|
|
93
|
+
* err.message }), uncapped). `'normalized'` reuses fingerprint/normalize/message.js's normalizeMessage()
|
|
94
|
+
* for the exception message and fingerprint/normalize/stack.js's parseAndNormalizeStack() for the
|
|
95
|
+
* stacktrace (cwd-stripped, node_modules-version-collapsed) — no new scrubbing pipeline. `'none'` sets
|
|
96
|
+
* only the ERROR status code, no message, no exception event — the same pattern already used for a
|
|
97
|
+
* tool-level isError: true failure. Also settable via the `OTEL_MCP_ERROR_RECORDING_MODE` environment
|
|
98
|
+
* variable (lower precedence than this option), same OTEL_MCP_<FEATURE>_ prefix pattern as
|
|
99
|
+
* OTEL_MCP_THRASH_ and OTEL_MCP_SCHEMA_DRIFT_; an unrecognized value from either source falls back to
|
|
100
|
+
* 'full' silently, never a throw.
|
|
87
101
|
* @property {string} [instanceKey] - Host-supplied stable identifier for one logical service (ADR 012,
|
|
88
102
|
* docs/adr/012-tracker-lifecycle-and-shared-state.md, Option C — Phase 2: this option and its wiring).
|
|
89
103
|
* When provided, the four in-memory trackers this library keeps per instrumented server — budget
|
|
@@ -259,6 +273,7 @@ export function resolveOptions(options) {
|
|
|
259
273
|
},
|
|
260
274
|
thrashDetection: resolveThrashConfig(opts.thrashDetection),
|
|
261
275
|
schemaDrift: resolveSchemaDriftConfig(opts.schemaDrift),
|
|
276
|
+
errorRecording: resolveErrorRecordingConfig(opts.errorRecording),
|
|
262
277
|
instanceKey: resolveInstanceKey(opts.instanceKey, process.env[ENV_INSTANCE_KEY]),
|
|
263
278
|
};
|
|
264
279
|
}
|
package/src/cost/calculator.js
CHANGED
|
@@ -18,6 +18,72 @@ export function normalizeModelName(model) {
|
|
|
18
18
|
return slashIndex === -1 ? lower : lower.slice(slashIndex + 1);
|
|
19
19
|
}
|
|
20
20
|
|
|
21
|
+
// ADR 019 Part 2 (docs/adr/019-raw-content-on-spans.md, v0.13.0 Phase 2):
|
|
22
|
+
// usage.model is tool-RESULT content, not library-computed metadata like
|
|
23
|
+
// everything else applyCostAttribution() (instrument.js) puts on a span —
|
|
24
|
+
// so it's gated through this allowlist before it can ever reach
|
|
25
|
+
// mcp.tool.model/gen_ai.response.model or a metric label (metrics.js's
|
|
26
|
+
// recordTokens()/recordCost() both key a label off it too, so an
|
|
27
|
+
// unvalidated value would be a metric-cardinality hazard, not just a span
|
|
28
|
+
// content one).
|
|
29
|
+
//
|
|
30
|
+
// Verified empirically against two real contracts, not assumed from
|
|
31
|
+
// provider docs: (1) every DEFAULT_PRICING key (pricing.js) matches a
|
|
32
|
+
// narrower `[a-z0-9-]` set, but (2) normalizeModelName() above already
|
|
33
|
+
// documents AND tests (test/cost/calculator.test.js) a REQUIRED
|
|
34
|
+
// "provider/model" input shape ("Anthropic/Claude-Opus-4-7") that no
|
|
35
|
+
// DEFAULT_PRICING key itself contains — a narrow, table-derived allowlist
|
|
36
|
+
// would reject input this library already advertises support for. `.`/
|
|
37
|
+
// `:`/`@` are additionally admitted for real-world provider conventions
|
|
38
|
+
// this investigation did not verify live (Bedrock's
|
|
39
|
+
// `anthropic.claude-3-sonnet-20240229-v1:0` / ARN form, Vertex AI's
|
|
40
|
+
// `text-bison@001`) — deliberately generous: silently rejecting a
|
|
41
|
+
// legitimate model is a worse failure mode than admitting a few
|
|
42
|
+
// characters no currently-known convention actually uses. See
|
|
43
|
+
// isValidModelId()'s only caller (instrument.js's applyCostAttribution())
|
|
44
|
+
// for what "rejected" actually does, which is never a silent drop.
|
|
45
|
+
export const MODEL_ID_MAX_LENGTH = 256;
|
|
46
|
+
const MODEL_ID_RE = new RegExp(`^[A-Za-z0-9._:/@-]{1,${MODEL_ID_MAX_LENGTH}}$`);
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Whether `value` is shaped like a real model identifier: a non-empty
|
|
50
|
+
* string, at most {@link MODEL_ID_MAX_LENGTH} characters, containing only
|
|
51
|
+
* alphanumerics and `.`/`_`/`:`/`/`/`@`/`-` — see this module's own
|
|
52
|
+
* comment above `MODEL_ID_MAX_LENGTH` for why this specific set. Does NOT
|
|
53
|
+
* check whether `value` resolves to a `DEFAULT_PRICING` entry —
|
|
54
|
+
* `mcp.tool.pricing_status: "unknown"` already exists to represent a
|
|
55
|
+
* real, valid model name with no known price (ADR 016 point 4); this
|
|
56
|
+
* function is a shape gate, not a pricing-table membership check.
|
|
57
|
+
*
|
|
58
|
+
* @param {unknown} value
|
|
59
|
+
* @returns {value is string}
|
|
60
|
+
*/
|
|
61
|
+
export function isValidModelId(value) {
|
|
62
|
+
return typeof value === 'string' && MODEL_ID_RE.test(value);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Describes, in shape-only terms, why `value` failed {@link isValidModelId}
|
|
67
|
+
* — a length, or "not a string"/"empty string" — and NEVER the value
|
|
68
|
+
* itself. Used only for a one-time diagnostic (instrument.js's
|
|
69
|
+
* applyCostAttribution(), ADR 019 Part 2): echoing the rejected value
|
|
70
|
+
* there would relocate this whole gate's problem from spans into logs
|
|
71
|
+
* instead of closing it, so this deliberately returns shape and nothing
|
|
72
|
+
* else — see that function's own warnRejectedModel() for the full
|
|
73
|
+
* reasoning.
|
|
74
|
+
*
|
|
75
|
+
* @param {unknown} value
|
|
76
|
+
* @returns {string}
|
|
77
|
+
*/
|
|
78
|
+
export function describeInvalidModelId(value) {
|
|
79
|
+
if (typeof value !== 'string') return 'not a string';
|
|
80
|
+
if (value.length === 0) return 'empty string';
|
|
81
|
+
if (value.length > MODEL_ID_MAX_LENGTH) {
|
|
82
|
+
return `length ${value.length}, expected 1-${MODEL_ID_MAX_LENGTH} identifier-shaped characters`;
|
|
83
|
+
}
|
|
84
|
+
return `length ${value.length}, contains a disallowed character`;
|
|
85
|
+
}
|
|
86
|
+
|
|
21
87
|
/**
|
|
22
88
|
* @param {unknown} value
|
|
23
89
|
* @returns {value is number}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module error-recording/config
|
|
3
|
+
* Options parsing and defaults for exception-recording mode (ADR 019,
|
|
4
|
+
* docs/adr/019-raw-content-on-spans.md — Part 1, v0.13.0 Phase 1).
|
|
5
|
+
*
|
|
6
|
+
* Mirrors src/thrash/config.js's and src/schema-drift/config.js's pattern
|
|
7
|
+
* exactly: precedence, highest to lowest, is an explicit field on the
|
|
8
|
+
* `partial` argument, then the field's `OTEL_MCP_ERROR_RECORDING_*` env
|
|
9
|
+
* var, then the hardcoded default. An invalid or unparseable value from
|
|
10
|
+
* either source is treated exactly like an absent one — silent fallback
|
|
11
|
+
* to the next source, never a throw. `mode` is this module's only field
|
|
12
|
+
* (unlike thrash/schema-drift's several), so `pick()`'s multi-type
|
|
13
|
+
* dispatch (see those two modules' `BOOLEAN_FIELDS` Set) has nothing to
|
|
14
|
+
* dispatch between here — this module reads the one env var directly.
|
|
15
|
+
*
|
|
16
|
+
* The small `resolveMode()` helper below is not imported from either
|
|
17
|
+
* sibling config module — those don't export it, and this project's own
|
|
18
|
+
* convention (see schema-drift/config.js's own docblock, and
|
|
19
|
+
* fingerprint/hash.js's FALLBACK_FINGERPRINT duplication precedent it
|
|
20
|
+
* cites) is to duplicate a small, self-contained primitive rather than
|
|
21
|
+
* couple otherwise-unrelated feature config modules together.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* @typedef {import('./types.d.ts').ErrorRecordingConfig} ErrorRecordingConfig
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
const ENV_PREFIX = 'OTEL_MCP_ERROR_RECORDING_';
|
|
29
|
+
const ENV_MODE = `${ENV_PREFIX}MODE`;
|
|
30
|
+
|
|
31
|
+
const VALID_MODES = new Set(['full', 'normalized', 'none']);
|
|
32
|
+
|
|
33
|
+
/** @type {ErrorRecordingConfig} */
|
|
34
|
+
const DEFAULTS = {
|
|
35
|
+
mode: 'full',
|
|
36
|
+
};
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Resolves `mode` to one of `VALID_MODES`, or `fallback` if `value` isn't
|
|
40
|
+
* one — an unrecognized string (a typo, e.g. `'normalised'`) degrades to
|
|
41
|
+
* the default exactly like a missing value, never a throw, matching this
|
|
42
|
+
* codebase's general "invalid config degrades to the next source" env-var
|
|
43
|
+
* discipline (thrash/config.js, schema-drift/config.js).
|
|
44
|
+
*
|
|
45
|
+
* @param {unknown} value
|
|
46
|
+
* @param {ErrorRecordingConfig['mode']} fallback
|
|
47
|
+
* @returns {ErrorRecordingConfig['mode']}
|
|
48
|
+
*/
|
|
49
|
+
function resolveMode(value, fallback) {
|
|
50
|
+
return typeof value === 'string' && VALID_MODES.has(value) ? /** @type {ErrorRecordingConfig['mode']} */ (value) : fallback;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* @param {Partial<ErrorRecordingConfig> | undefined} partial
|
|
55
|
+
* @returns {unknown} `partial.mode` if present, else the env var's raw string, else undefined (caller applies the default).
|
|
56
|
+
*/
|
|
57
|
+
function pickMode(partial) {
|
|
58
|
+
if (partial && partial.mode !== undefined) return partial.mode;
|
|
59
|
+
return process.env[ENV_MODE];
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Validates and applies defaults to raw error-recording options. Never
|
|
64
|
+
* throws — the one field independently falls back to its default on
|
|
65
|
+
* anything other than a valid, recognized mode string.
|
|
66
|
+
*
|
|
67
|
+
* @param {Partial<ErrorRecordingConfig>} [partial]
|
|
68
|
+
* @returns {ErrorRecordingConfig}
|
|
69
|
+
*/
|
|
70
|
+
export function resolveErrorRecordingConfig(partial) {
|
|
71
|
+
return {
|
|
72
|
+
mode: resolveMode(pickMode(partial), DEFAULTS.mode),
|
|
73
|
+
};
|
|
74
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared type definitions for exception-recording mode (ADR 019,
|
|
3
|
+
* docs/adr/019-raw-content-on-spans.md — Part 1, v0.13.0 Phase 1).
|
|
4
|
+
*
|
|
5
|
+
* Hand-written, not a compiled build artifact — this project ships plain
|
|
6
|
+
* JS with no TypeScript build step (see CONTRIBUTING.md). {@link ErrorRecordingConfig}
|
|
7
|
+
* is re-exported from src/index.d.ts (via instrumentMcpServer()'s
|
|
8
|
+
* `options.errorRecording`), the same pattern src/thrash/types.d.ts's
|
|
9
|
+
* `ThrashConfig` and src/schema-drift/types.d.ts's `SchemaDriftConfig`
|
|
10
|
+
* already establish: a hand-written interface here, kept in sync with
|
|
11
|
+
* src/error-recording/config.js's own JSDoc `@typedef` by hand, not by a
|
|
12
|
+
* build step.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Resolved error-recording config — see src/error-recording/config.js's
|
|
17
|
+
* `resolveErrorRecordingConfig()`, which this mirrors field-for-field. All
|
|
18
|
+
* fields are required here (this is the RESOLVED shape, after
|
|
19
|
+
* defaults/env vars have been applied); {@link instrumentMcpServer}'s
|
|
20
|
+
* `errorRecording` option accepts `Partial<ErrorRecordingConfig>` — see
|
|
21
|
+
* src/index.d.ts. Same pattern as `ThrashConfig`/`SchemaDriftConfig`.
|
|
22
|
+
*/
|
|
23
|
+
export interface ErrorRecordingConfig {
|
|
24
|
+
/**
|
|
25
|
+
* Controls what a thrown exception (as opposed to a tool-level
|
|
26
|
+
* `isError: true` result, which never carries a JS `Error` and is
|
|
27
|
+
* unaffected by this option) puts on the `tools/call`/`tools/list`
|
|
28
|
+
* span, per ADR 019 Part 1:
|
|
29
|
+
*
|
|
30
|
+
* - `'full'` — `span.recordException(err)` plus `span.setStatus({
|
|
31
|
+
* message: err.message })`, exactly as every release before
|
|
32
|
+
* v0.13.0. `exception.message`/`exception.stacktrace` carry
|
|
33
|
+
* `err.message`/`err.stack` verbatim, uncapped — matching the OTel
|
|
34
|
+
* ecosystem's own default (see ADR 019's "What OpenTelemetry's own
|
|
35
|
+
* semantic conventions say").
|
|
36
|
+
* - `'normalized'` — the same `exception` event shape, but
|
|
37
|
+
* `exception.message` is run through `normalizeMessage()`
|
|
38
|
+
* (`src/fingerprint/normalize/message.js` — the same
|
|
39
|
+
* UUID/email/URL/IP/timestamp/path/hex/quoted-id scrubbing already
|
|
40
|
+
* used before fingerprint hashing) and `exception.stacktrace` is
|
|
41
|
+
* rebuilt from `parseAndNormalizeStack()`'s frames
|
|
42
|
+
* (`src/fingerprint/normalize/stack.js` — cwd-stripped,
|
|
43
|
+
* `node_modules` version-collapsed), never the raw `err.stack`.
|
|
44
|
+
* - `'none'` — `span.setStatus({ code: ERROR })` only: no `message`,
|
|
45
|
+
* no `exception` event at all. The same pattern already used for a
|
|
46
|
+
* tool-level `isError: true` failure.
|
|
47
|
+
*
|
|
48
|
+
* `error.type` is capped at 128 characters regardless of this
|
|
49
|
+
* setting — see ADR 019 Part 1, "`error.type` — closing ADR 004's
|
|
50
|
+
* open note as part of this design."
|
|
51
|
+
*
|
|
52
|
+
* @default 'full'
|
|
53
|
+
*/
|
|
54
|
+
mode: 'full' | 'normalized' | 'none';
|
|
55
|
+
}
|
|
@@ -21,6 +21,15 @@ export const ATTRIBUTE_KEYS = Object.freeze({
|
|
|
21
21
|
SIGNATURE: 'mcp.failure.signature',
|
|
22
22
|
CATEGORY: 'mcp.failure.category',
|
|
23
23
|
ORIGIN: 'mcp.failure.origin',
|
|
24
|
+
/**
|
|
25
|
+
* `result.inputs.errorClass` — `err.name` (or a non-Error throwable's
|
|
26
|
+
* `.name`), e.g. `"TypeError"`, `"ZodError"`. Capped at 128 characters
|
|
27
|
+
* by `computeFingerprint()` (`fingerprint/compose.js`'s
|
|
28
|
+
* `MAX_ERROR_CLASS_LENGTH`) — length-bounded only, not pattern-scrubbed
|
|
29
|
+
* the way `normalizedMessage` is: this is the application/library's own
|
|
30
|
+
* error class name, whatever it set `.name` to, not free text this
|
|
31
|
+
* library controls the shape of (docs/known-gaps.md entry 10).
|
|
32
|
+
*/
|
|
24
33
|
ERROR_CLASS: 'mcp.failure.error_class',
|
|
25
34
|
/**
|
|
26
35
|
* ADR 007's channel dimension (`classifyFailureChannel()`,
|
|
@@ -42,6 +51,13 @@ export const ATTRIBUTE_KEYS = Object.freeze({
|
|
|
42
51
|
* CHANNEL, additive and never part of the fingerprint hash: the path
|
|
43
52
|
* text is already implicit in the hashed normalized message (see ADR
|
|
44
53
|
* 009), so hashing it again would be redundant, not more correct.
|
|
54
|
+
*
|
|
55
|
+
* A path segment that isn't a schema-declared identifier (a
|
|
56
|
+
* `z.record()`/map schema's runtime key, e.g. an email used as an
|
|
57
|
+
* object key) is redacted to `<KEY>` rather than surfaced — see
|
|
58
|
+
* `PATH_SEGMENT_RE`'s comment in validation-paths.js (docs/known-gaps.md
|
|
59
|
+
* entry 10) for why a placeholder, not a dropped segment or a dropped
|
|
60
|
+
* path.
|
|
45
61
|
*/
|
|
46
62
|
VALIDATION_PATHS: 'mcp.failure.validation_paths',
|
|
47
63
|
});
|
|
@@ -147,6 +147,51 @@ function extractJsonArraySubstring(text) {
|
|
|
147
147
|
return null; // unbalanced -- never guess a truncated substring
|
|
148
148
|
}
|
|
149
149
|
|
|
150
|
+
// Known-gaps entry 10 (docs/known-gaps.md): unlike the two RENDERED
|
|
151
|
+
// formats below (RENDERED_DOT_PATH_RE, V2_ISSUE_START_RE), this JSON
|
|
152
|
+
// format's `path` array comes straight from a parsed ZodError with no
|
|
153
|
+
// character restriction at all -- for a `z.record()`/map-shaped schema,
|
|
154
|
+
// a failing issue's `path` includes the actual runtime KEY the caller
|
|
155
|
+
// passed (e.g. an email address used as an object key), which is caller
|
|
156
|
+
// data, not a schema-declared field name. `PATH_SEGMENT_RE` gates each
|
|
157
|
+
// string segment against the same identifier-only shape the rendered
|
|
158
|
+
// formats already require (RENDERED_DOT_PATH_RE without the `.`/`[...]`
|
|
159
|
+
// chaining, since here each segment is already a separate array element,
|
|
160
|
+
// not a joined string to split) -- a segment that doesn't match is, by
|
|
161
|
+
// construction, not a schema-declared property name a tool author wrote,
|
|
162
|
+
// so it's treated as a dynamic key and redacted.
|
|
163
|
+
//
|
|
164
|
+
// Numeric segments (array indices) are never redacted: Zod only ever
|
|
165
|
+
// produces a `number` path segment for an array index, never for an
|
|
166
|
+
// object/record key (JSON object keys are always strings, even for a
|
|
167
|
+
// numeric-looking one) -- so a `number` segment is structurally always a
|
|
168
|
+
// small integer index, never caller-chosen content.
|
|
169
|
+
//
|
|
170
|
+
// What "redacted" means was a real choice among three, argued here rather
|
|
171
|
+
// than picked silently:
|
|
172
|
+
// - Drop the whole path. Loses the entire signal for exactly the
|
|
173
|
+
// validation failures a record-shaped schema exists to catch -- the
|
|
174
|
+
// one case this gap is about is also the one case this option throws
|
|
175
|
+
// away completely.
|
|
176
|
+
// - Drop just the offending segment. WRONG, not just lossy: joining the
|
|
177
|
+
// remaining segments produces a path that names a DIFFERENT, real
|
|
178
|
+
// field. `["users", "<dynamic-key>", "email"]` dropped down to
|
|
179
|
+
// `users.email` looks like a legitimate, confidently-reported path to
|
|
180
|
+
// a field named `email` directly under `users` -- a fabricated
|
|
181
|
+
// dot-path this feature's own "never guess" discipline (see this
|
|
182
|
+
// module's docblock) exists specifically to rule out.
|
|
183
|
+
// - Replace the segment with a fixed placeholder. Preserves the path's
|
|
184
|
+
// shape and depth (still says "the failure was inside a record-like
|
|
185
|
+
// value at this position") without leaking the key's content -- the
|
|
186
|
+
// same shape/anomaly-signal-without-content-capture tradeoff
|
|
187
|
+
// `mcp.tool.argument_count` already makes (`src/attributes.js`), and
|
|
188
|
+
// the same placeholder-substitution mechanism `normalizeMessage()`'s
|
|
189
|
+
// `NORMALIZE_STEPS` already uses for EMAIL/UUID/URL/etc.
|
|
190
|
+
// (`normalize/patterns.js`). Chosen: it's the only option that
|
|
191
|
+
// neither destroys the signal nor fabricates a wrong one.
|
|
192
|
+
const PATH_SEGMENT_RE = /^[\w$]+$/;
|
|
193
|
+
const REDACTED_PATH_SEGMENT = '<KEY>';
|
|
194
|
+
|
|
150
195
|
/**
|
|
151
196
|
* Parses `jsonText` and, only if EVERY element confidently looks like a
|
|
152
197
|
* Zod issue (a plain object with both a `path` array of string/number
|
|
@@ -159,6 +204,12 @@ function extractJsonArraySubstring(text) {
|
|
|
159
204
|
* more likely a sign this isn't really a Zod issues array at all than a
|
|
160
205
|
* reason to keep only the matching ones.
|
|
161
206
|
*
|
|
207
|
+
* Any string segment that isn't identifier-shaped (`PATH_SEGMENT_RE`) is
|
|
208
|
+
* replaced with `REDACTED_PATH_SEGMENT` before joining — see the
|
|
209
|
+
* constants' own comment above for why a placeholder, not a dropped
|
|
210
|
+
* segment or a dropped path. Numeric segments (array indices) are never
|
|
211
|
+
* redacted.
|
|
212
|
+
*
|
|
162
213
|
* @param {string} jsonText
|
|
163
214
|
* @returns {string[] | null}
|
|
164
215
|
*/
|
|
@@ -182,7 +233,10 @@ function parseZodIssuesArray(jsonText) {
|
|
|
182
233
|
if (!Array.isArray(path) || path.length === 0) return null;
|
|
183
234
|
if (!path.every((segment) => typeof segment === 'string' || typeof segment === 'number')) return null;
|
|
184
235
|
|
|
185
|
-
|
|
236
|
+
const redactedPath = path.map((segment) =>
|
|
237
|
+
typeof segment === 'number' || PATH_SEGMENT_RE.test(segment) ? segment : REDACTED_PATH_SEGMENT,
|
|
238
|
+
);
|
|
239
|
+
paths.push(redactedPath.join('.'));
|
|
186
240
|
}
|
|
187
241
|
return paths;
|
|
188
242
|
}
|
|
@@ -23,6 +23,31 @@ import { DEFAULT_CLASSIFIERS, runClassifiers } from './classify/index.js';
|
|
|
23
23
|
const HASH_INPUT_VERSION = 'v1';
|
|
24
24
|
const DEFAULT_STACK_FRAMES = 5;
|
|
25
25
|
|
|
26
|
+
// Known-gaps entry 10 (docs/known-gaps.md): `errorClass` is `err.name` (or
|
|
27
|
+
// a non-Error throwable's `.name`), and unlike `normalizedMessage` above it
|
|
28
|
+
// gets no NORMALIZE_STEPS scrubbing — it lands on `mcp.failure.error_class`
|
|
29
|
+
// (fingerprint/attributes.js) and `error.type` (instrument.js) exactly as
|
|
30
|
+
// whatever thrown value set it. In practice `.name` is almost always a
|
|
31
|
+
// short, fixed class identifier ("TypeError", "ZodError"), but nothing in
|
|
32
|
+
// the language enforces that — a tool author (or library it depends on)
|
|
33
|
+
// can assign any string. Capped here, not scrubbed like a message: a class
|
|
34
|
+
// name isn't expected to contain structured PII shapes (emails, ids, ...)
|
|
35
|
+
// the way a free-text message is, so pattern-matching would be reaching
|
|
36
|
+
// for a problem this value doesn't really have. Length is the actual
|
|
37
|
+
// unbounded dimension, so that's what's bounded. This also caps what feeds
|
|
38
|
+
// the fingerprint hash below, which is fine: two DIFFERENT long class
|
|
39
|
+
// names sharing an identical first 128 characters colliding into the same
|
|
40
|
+
// fingerprint is not a real-world scenario a normal class identifier ever
|
|
41
|
+
// produces.
|
|
42
|
+
//
|
|
43
|
+
// Exported as of v0.13.0 (ADR 019 Part 1, docs/adr/019-raw-content-on-spans.md):
|
|
44
|
+
// instrument.js's independent `error.type`/`exception.type` read of the
|
|
45
|
+
// same underlying err.name (ATTR_ERROR_TYPE, and the 'normalized'
|
|
46
|
+
// errorRecording.mode's exception event) reuses this exact constant
|
|
47
|
+
// rather than a second, possibly-drifting copy of the number 128 — "one
|
|
48
|
+
// shared, capped value," per that ADR's own wording.
|
|
49
|
+
export const MAX_ERROR_CLASS_LENGTH = 128;
|
|
50
|
+
|
|
26
51
|
/**
|
|
27
52
|
* @param {FingerprintContext} [ctx]
|
|
28
53
|
* @returns {FingerprintResult}
|
|
@@ -120,8 +145,9 @@ export function computeFingerprint(err, ctx, opts = {}) {
|
|
|
120
145
|
category = 'dependency';
|
|
121
146
|
}
|
|
122
147
|
|
|
148
|
+
const rawErrorClass = typeof coerced.name === 'string' ? coerced.name : String(coerced.name);
|
|
123
149
|
const inputs = {
|
|
124
|
-
errorClass:
|
|
150
|
+
errorClass: rawErrorClass.slice(0, MAX_ERROR_CLASS_LENGTH),
|
|
125
151
|
category,
|
|
126
152
|
origin: ctx.origin,
|
|
127
153
|
toolName: ctx.toolName ?? null,
|
|
@@ -80,7 +80,7 @@ export interface NormalizedStackFrame {
|
|
|
80
80
|
|
|
81
81
|
/** The normalized, low-cardinality inputs that get hashed into a fingerprint. */
|
|
82
82
|
export interface FingerprintInputs {
|
|
83
|
-
/** e.g. `"TypeError"`, `"ZodError"`, `"MCPToolError"`. */
|
|
83
|
+
/** e.g. `"TypeError"`, `"ZodError"`, `"MCPToolError"`. Capped at 128 characters (compose.js's MAX_ERROR_CLASS_LENGTH) — unlike `normalizedMessage`, not pattern-scrubbed, only length-bounded. */
|
|
84
84
|
readonly errorClass: string;
|
|
85
85
|
readonly category: FailureCategory;
|
|
86
86
|
readonly origin: FailureOrigin;
|
package/src/index.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { CostTrackingOptions } from './cost/types.d.ts';
|
|
2
2
|
import type { ThrashConfig, ThrashSummary } from './thrash/types.d.ts';
|
|
3
3
|
import type { SchemaDriftConfig } from './schema-drift/types.d.ts';
|
|
4
|
+
import type { ErrorRecordingConfig } from './error-recording/types.d.ts';
|
|
4
5
|
import type { ObservationState } from './observation/types.d.ts';
|
|
5
6
|
|
|
6
7
|
/**
|
|
@@ -113,6 +114,24 @@ export interface InstrumentOptions {
|
|
|
113
114
|
*/
|
|
114
115
|
schemaDrift?: Partial<SchemaDriftConfig>;
|
|
115
116
|
|
|
117
|
+
/**
|
|
118
|
+
* Controls what a THROWN exception (not a tool-level `isError: true` result, which never carries a JS
|
|
119
|
+
* `Error` and is unaffected by this option) puts on the `tools/call`/`tools/list` span (ADR 019,
|
|
120
|
+
* `docs/adr/019-raw-content-on-spans.md` Part 1, v0.13.0).
|
|
121
|
+
*
|
|
122
|
+
* Defaults to `{ mode: 'full' }` — byte-identical to every release before v0.13.0:
|
|
123
|
+
* `span.recordException(err)` plus `span.setStatus({ message: err.message })`, both uncapped.
|
|
124
|
+
* `'normalized'` reuses `normalizeMessage()` (`src/fingerprint/normalize/message.js`) for the exception
|
|
125
|
+
* message and `parseAndNormalizeStack()` (`src/fingerprint/normalize/stack.js`) for the stacktrace
|
|
126
|
+
* (cwd-stripped, `node_modules` version-collapsed) — no new scrubbing pipeline. `'none'` sets only the
|
|
127
|
+
* `ERROR` status code, no message, no `exception` event.
|
|
128
|
+
*
|
|
129
|
+
* Also settable via the `OTEL_MCP_ERROR_RECORDING_MODE` environment variable (lower precedence than this
|
|
130
|
+
* option); an unrecognized value from either source falls back to `'full'` silently, never a throw. See
|
|
131
|
+
* {@link ErrorRecordingConfig} (`src/error-recording/types.d.ts`).
|
|
132
|
+
*/
|
|
133
|
+
errorRecording?: Partial<ErrorRecordingConfig>;
|
|
134
|
+
|
|
116
135
|
/**
|
|
117
136
|
* Host-supplied stable identifier for one logical service (ADR 012,
|
|
118
137
|
* `docs/adr/012-tracker-lifecycle-and-shared-state.md`, Option C). When provided, the four in-memory
|
|
@@ -395,6 +414,18 @@ export type { ThrashConfig, ThrashDetectedEvent, ThrashSummary, ThrashOffender }
|
|
|
395
414
|
|
|
396
415
|
export type { SchemaDriftConfig, SchemaDriftKind, SchemaDriftEvent } from './schema-drift/types.d.ts';
|
|
397
416
|
|
|
417
|
+
// --- Exception-recording mode (src/error-recording/) ---
|
|
418
|
+
//
|
|
419
|
+
// Re-exported here so TypeScript consumers get this type from the package
|
|
420
|
+
// root instead of reaching into src/error-recording/* directly. See
|
|
421
|
+
// src/error-recording/types.d.ts for the full shape documentation and ADR
|
|
422
|
+
// 019 (docs/adr/019-raw-content-on-spans.md). Same posture as Agent
|
|
423
|
+
// Thrash Detection / schema drift above — no runtime value re-exported:
|
|
424
|
+
// resolveErrorRecordingConfig() is internal to src/config.js's wiring,
|
|
425
|
+
// not part of the public API.
|
|
426
|
+
|
|
427
|
+
export type { ErrorRecordingConfig } from './error-recording/types.d.ts';
|
|
428
|
+
|
|
398
429
|
// --- Two-axis observation contract (src/observation/) ---
|
|
399
430
|
//
|
|
400
431
|
// Re-exported here so TypeScript consumers get these types from the
|
package/src/instrument.js
CHANGED
|
@@ -12,11 +12,13 @@ import { resourceFromAttributes } from '@opentelemetry/resources';
|
|
|
12
12
|
import { resolveOptions } from './config.js';
|
|
13
13
|
import { StderrSpanExporter } from './exporters/stderr.js';
|
|
14
14
|
import { setupMeter } from './metrics.js';
|
|
15
|
-
import { computeFingerprint } from './fingerprint/compose.js';
|
|
15
|
+
import { computeFingerprint, MAX_ERROR_CLASS_LENGTH } from './fingerprint/compose.js';
|
|
16
16
|
import { toSpanAttributes, ATTRIBUTE_KEYS } from './fingerprint/attributes.js';
|
|
17
17
|
import { classifyFailureChannel } from './fingerprint/classify/channel.js';
|
|
18
18
|
import { extractValidationPaths } from './fingerprint/classify/validation-paths.js';
|
|
19
|
-
import {
|
|
19
|
+
import { normalizeMessage } from './fingerprint/normalize/message.js';
|
|
20
|
+
import { parseAndNormalizeStack } from './fingerprint/normalize/stack.js';
|
|
21
|
+
import { calculateCost, normalizeModelName, isValidModelId, describeInvalidModelId } from './cost/calculator.js';
|
|
20
22
|
import { DEFAULT_PRICING_LAST_VERIFIED } from './cost/pricing.js';
|
|
21
23
|
import { createBudgetTracker } from './cost/budget.js';
|
|
22
24
|
import { extractTraceContext } from './tracecontext/extract.js';
|
|
@@ -464,6 +466,15 @@ export function instrumentMcpServer(input, options) {
|
|
|
464
466
|
// resolveThrashSessionId() persist state across calls. See that
|
|
465
467
|
// function's docblock for why this flag exists at all.
|
|
466
468
|
const thrashSessionState = { hasSeenRealSessionId: false, hasWarnedFallbackUsed: false };
|
|
469
|
+
// Same "plain holder, not instanceKey-shared" reasoning as
|
|
470
|
+
// thrashSessionState immediately above (ADR 019 Part 2, v0.13.0 Phase
|
|
471
|
+
// 2): warnRejectedModel() (see applyCostAttribution()) is a low-stakes
|
|
472
|
+
// diagnostic, not correctness-critical accumulated state the way
|
|
473
|
+
// budgetTracker/thrashDetector are — firing again on a fresh instance
|
|
474
|
+
// under a fresh-Server-per-request deployment is the same known,
|
|
475
|
+
// accepted characteristic docs/known-gaps.md entry 6 already documents
|
|
476
|
+
// for thrashSessionState, not a new problem this introduces.
|
|
477
|
+
const costAttributionState = { warnedRejectedModel: false };
|
|
467
478
|
// Additive to instrumentMcpServer()'s existing return contract (the same
|
|
468
479
|
// input object, for chaining — see this function's own docblock): a
|
|
469
480
|
// getThrashSummary() method attached the same way shutdown() is, just
|
|
@@ -563,9 +574,19 @@ export function instrumentMcpServer(input, options) {
|
|
|
563
574
|
toolOutcomeCounter,
|
|
564
575
|
server,
|
|
565
576
|
kind,
|
|
577
|
+
resolved.errorRecording,
|
|
578
|
+
costAttributionState,
|
|
566
579
|
);
|
|
567
580
|
} else if (schema === v1.ListToolsRequestSchema && schemaDriftDetector) {
|
|
568
|
-
handler = wrapToolsListHandler(
|
|
581
|
+
handler = wrapToolsListHandler(
|
|
582
|
+
handler,
|
|
583
|
+
tracer,
|
|
584
|
+
schemaDriftDetector,
|
|
585
|
+
schemaDriftEmitter,
|
|
586
|
+
SCHEMA_DRIFT_SCOPE,
|
|
587
|
+
kind,
|
|
588
|
+
resolved.errorRecording,
|
|
589
|
+
);
|
|
569
590
|
}
|
|
570
591
|
return originalSetRequestHandler(schema, handler);
|
|
571
592
|
};
|
|
@@ -605,9 +626,19 @@ export function instrumentMcpServer(input, options) {
|
|
|
605
626
|
toolOutcomeCounter,
|
|
606
627
|
server,
|
|
607
628
|
kind,
|
|
629
|
+
resolved.errorRecording,
|
|
630
|
+
costAttributionState,
|
|
608
631
|
);
|
|
609
632
|
} else if (maybeHandler === undefined && method === TOOLS_LIST_METHOD && schemaDriftDetector) {
|
|
610
|
-
handlerOrSchemas = wrapToolsListHandler(
|
|
633
|
+
handlerOrSchemas = wrapToolsListHandler(
|
|
634
|
+
handlerOrSchemas,
|
|
635
|
+
tracer,
|
|
636
|
+
schemaDriftDetector,
|
|
637
|
+
schemaDriftEmitter,
|
|
638
|
+
SCHEMA_DRIFT_SCOPE,
|
|
639
|
+
kind,
|
|
640
|
+
resolved.errorRecording,
|
|
641
|
+
);
|
|
611
642
|
}
|
|
612
643
|
return maybeHandler === undefined
|
|
613
644
|
? originalSetRequestHandler(method, handlerOrSchemas)
|
|
@@ -748,6 +779,51 @@ function extractSessionAndRequestId(kind, extraOrCtx) {
|
|
|
748
779
|
return { sessionId: extraOrCtx?.sessionId, requestId: extraOrCtx?.requestId };
|
|
749
780
|
}
|
|
750
781
|
|
|
782
|
+
/**
|
|
783
|
+
* One-time diagnostic for a tool result's `model` field that failed
|
|
784
|
+
* isValidModelId()'s length/shape gate (ADR 019 Part 2,
|
|
785
|
+
* docs/adr/019-raw-content-on-spans.md, v0.13.0 Phase 2). Distinct from
|
|
786
|
+
* budgetTracker.recordUnpriced()'s own warning (src/cost/budget.js):
|
|
787
|
+
* that one describes a LATER, different condition (a real, valid model
|
|
788
|
+
* name that just doesn't resolve to a price) and only fires when a
|
|
789
|
+
* budget is actually configured. This one fires for any
|
|
790
|
+
* `costTracking.enabled` deployment, budget or not — "your cost/token
|
|
791
|
+
* attribution is silently missing a model" is worth knowing regardless
|
|
792
|
+
* of whether a budget guardrail is in play.
|
|
793
|
+
*
|
|
794
|
+
* CRITICAL, and the entire reason this function exists rather than just
|
|
795
|
+
* inlining a `diag.warn()` call with `value` in it: reports SHAPE ONLY —
|
|
796
|
+
* via describeInvalidModelId() (src/cost/calculator.js), never the
|
|
797
|
+
* rejected value itself. `diag.warn()` output routinely gets piped into
|
|
798
|
+
* a deployment's own logging pipeline; printing the offending string
|
|
799
|
+
* here would relocate this whole gate's problem from spans into logs
|
|
800
|
+
* instead of closing it.
|
|
801
|
+
*
|
|
802
|
+
* Fires at most once per `state` object — see applyCostAttribution()'s
|
|
803
|
+
* own docblock for why `state` (costAttributionState) is constructed
|
|
804
|
+
* fresh per instrumentMcpServer() call rather than instanceKey-shared.
|
|
805
|
+
* Never throws, matching every other diagnostic in this codebase.
|
|
806
|
+
*
|
|
807
|
+
* @param {unknown} value - The rejected `usage.model` value. Read only for its shape (typeof, length) —
|
|
808
|
+
* never logged, never returned, never otherwise observable outside this function.
|
|
809
|
+
* @param {{ warnedRejectedModel: boolean }} state
|
|
810
|
+
*/
|
|
811
|
+
function warnRejectedModel(value, state) {
|
|
812
|
+
if (state.warnedRejectedModel) return;
|
|
813
|
+
state.warnedRejectedModel = true;
|
|
814
|
+
|
|
815
|
+
try {
|
|
816
|
+
diag.warn(
|
|
817
|
+
`opentel-mcp: a tool result's model field failed validation (${describeInvalidModelId(value)}) and was ` +
|
|
818
|
+
"excluded from mcp.tool.model / gen_ai.response.model / cost attribution; mcp.tool.pricing_status was " +
|
|
819
|
+
"set to 'unknown'. This warning fires once per instrumentMcpServer() call and deliberately never logs " +
|
|
820
|
+
'the value itself — see docs/adr/019-raw-content-on-spans.md Part 2.',
|
|
821
|
+
);
|
|
822
|
+
} catch {
|
|
823
|
+
// Never throw — see this function's own docblock.
|
|
824
|
+
}
|
|
825
|
+
}
|
|
826
|
+
|
|
751
827
|
/**
|
|
752
828
|
* Best-effort cost/token attribution for one tool call's result, added on
|
|
753
829
|
* top of the span and metrics that always fire (see wrapToolCallHandler
|
|
@@ -794,6 +870,22 @@ function extractSessionAndRequestId(kind, extraOrCtx) {
|
|
|
794
870
|
* directly instead of inferring it from an attribute's absence. Passed to
|
|
795
871
|
* recordTokens/recordCost too, for the same metric-level visibility.
|
|
796
872
|
*
|
|
873
|
+
* v0.13.0 (ADR 019 Part 2, docs/adr/019-raw-content-on-spans.md): `usage.model`
|
|
874
|
+
* is tool-RESULT content, not library-computed metadata — before it's used
|
|
875
|
+
* ANYWHERE below (span attributes, calculateCost(), pricingOverrideKeys
|
|
876
|
+
* lookup, metric labels), it's gated through isValidModelId()
|
|
877
|
+
* (src/cost/calculator.js): a length/character allowlist, never a
|
|
878
|
+
* pricing-table membership check. A value that fails this gate is treated
|
|
879
|
+
* exactly like "no model detected" for every downstream purpose — the
|
|
880
|
+
* existing mcp.tool.pricing_status: "unknown" / budgetTracker.recordUnpriced()
|
|
881
|
+
* machinery already handles that case correctly, so a rejected value is
|
|
882
|
+
* never a silent drop, just a redirection into a path this function
|
|
883
|
+
* already had. warnRejectedModel() below additionally fires a one-time,
|
|
884
|
+
* SHAPE-ONLY diagnostic (never the rejected value itself — see that
|
|
885
|
+
* function's own docblock) distinct from budgetTracker.recordUnpriced()'s
|
|
886
|
+
* own warning, which describes a different, later condition and is gated
|
|
887
|
+
* on a budget being configured; this one isn't.
|
|
888
|
+
*
|
|
797
889
|
* @param {import('@opentelemetry/api').Span} span
|
|
798
890
|
* @param {ReturnType<import('./metrics.js').setupMeter> | null} metricsRecorder
|
|
799
891
|
* @param {string | undefined} toolName
|
|
@@ -801,9 +893,12 @@ function extractSessionAndRequestId(kind, extraOrCtx) {
|
|
|
801
893
|
* @param {*} result
|
|
802
894
|
* @param {import('./config.js').CostTrackingOptions & { pricingOverrideKeys: Set<string> }} costTracking
|
|
803
895
|
* @param {ReturnType<import('./cost/budget.js').createBudgetTracker>} budgetTracker
|
|
896
|
+
* @param {{ warnedRejectedModel: boolean }} costAttributionState - ADR 019 Part 2: one-time-warning state
|
|
897
|
+
* for warnRejectedModel() below, constructed once per instrumentMcpServer() call (same non-instanceKey-shared
|
|
898
|
+
* granularity as thrashSessionState — see that variable's own comment in instrumentMcpServer()).
|
|
804
899
|
* @returns {{ tokensIn: number, tokensOut: number, costUsd: number } | null}
|
|
805
900
|
*/
|
|
806
|
-
function applyCostAttribution(span, metricsRecorder, toolName, sessionId, result, costTracking, budgetTracker) {
|
|
901
|
+
function applyCostAttribution(span, metricsRecorder, toolName, sessionId, result, costTracking, budgetTracker, costAttributionState) {
|
|
807
902
|
if (!costTracking.enabled) return null;
|
|
808
903
|
|
|
809
904
|
try {
|
|
@@ -813,33 +908,55 @@ function applyCostAttribution(span, metricsRecorder, toolName, sessionId, result
|
|
|
813
908
|
span.setAttribute(ATTR_MCP_TOOL_TOKENS_INPUT, usage.inputTokens);
|
|
814
909
|
span.setAttribute(ATTR_MCP_TOOL_TOKENS_OUTPUT, usage.outputTokens);
|
|
815
910
|
span.setAttribute(ATTR_MCP_TOOL_TOKENS_TOTAL, usage.totalTokens);
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
911
|
+
|
|
912
|
+
// ADR 019 Part 2: usage.model is gated here, ONCE, before it reaches
|
|
913
|
+
// anything below — a rejected value (usage.model was present but
|
|
914
|
+
// isn't identifier-shaped) is treated identically to "no model
|
|
915
|
+
// detected" for every downstream purpose (validModel stays
|
|
916
|
+
// undefined), and warnRejectedModel() fires the one-time, shape-only
|
|
917
|
+
// diagnostic. usage.model === undefined (no model detected at all —
|
|
918
|
+
// the pre-existing, unrelated case) never calls warnRejectedModel:
|
|
919
|
+
// there's nothing rejected, just nothing found.
|
|
920
|
+
const modelIsValid = isValidModelId(usage.model);
|
|
921
|
+
if (usage.model !== undefined && !modelIsValid) {
|
|
922
|
+
warnRejectedModel(usage.model, costAttributionState);
|
|
923
|
+
}
|
|
924
|
+
const validModel = modelIsValid ? usage.model : undefined;
|
|
925
|
+
|
|
926
|
+
if (validModel) {
|
|
927
|
+
span.setAttribute(ATTR_MCP_TOOL_MODEL, validModel);
|
|
928
|
+
span.setAttribute(ATTR_GEN_AI_RESPONSE_MODEL, validModel);
|
|
819
929
|
}
|
|
820
930
|
|
|
821
931
|
let costUsd = null;
|
|
822
|
-
if (
|
|
823
|
-
costUsd = calculateCost(usage.inputTokens, usage.outputTokens,
|
|
932
|
+
if (validModel) {
|
|
933
|
+
costUsd = calculateCost(usage.inputTokens, usage.outputTokens, validModel, costTracking.pricingTable);
|
|
824
934
|
}
|
|
825
935
|
|
|
826
936
|
// ADR 016 point 4: provenance-based, computed against the same
|
|
827
937
|
// normalized key calculateCost() itself looked up — 'unknown' covers
|
|
828
|
-
//
|
|
829
|
-
//
|
|
938
|
+
// "no model detected at all," "a rejected model field" (ADR 019 Part
|
|
939
|
+
// 2 — validModel is undefined either way), and "model detected but
|
|
940
|
+
// didn't resolve to a cost" (unrecognized, or a malformed override
|
|
941
|
+
// entry).
|
|
830
942
|
const pricingStatus =
|
|
831
|
-
!
|
|
943
|
+
!validModel || costUsd === null
|
|
832
944
|
? MCP_TOOL_PRICING_STATUS_UNKNOWN
|
|
833
|
-
: costTracking.pricingOverrideKeys.has(normalizeModelName(
|
|
945
|
+
: costTracking.pricingOverrideKeys.has(normalizeModelName(validModel))
|
|
834
946
|
? MCP_TOOL_PRICING_STATUS_USER_OVERRIDE
|
|
835
947
|
: MCP_TOOL_PRICING_STATUS_KNOWN;
|
|
836
948
|
span.setAttribute(ATTR_MCP_TOOL_PRICING_STATUS, pricingStatus);
|
|
837
|
-
|
|
949
|
+
// validModel, not usage.model: mcp.tool.model is also a metric LABEL
|
|
950
|
+
// here (metrics.js's recordTokens()/recordCost()), so a rejected
|
|
951
|
+
// value must never reach it either — an unvalidated string becoming a
|
|
952
|
+
// label is a cardinality hazard on top of the content-exposure one
|
|
953
|
+
// ADR 019 Part 2 is about.
|
|
954
|
+
metricsRecorder?.recordTokens(toolName, validModel, usage.totalTokens, pricingStatus);
|
|
838
955
|
|
|
839
956
|
if (costUsd !== null) {
|
|
840
957
|
span.setAttribute(ATTR_MCP_TOOL_COST_USD, costUsd);
|
|
841
958
|
span.setAttribute(ATTR_MCP_TOOL_COST_CURRENCY, MCP_TOOL_COST_CURRENCY_USD);
|
|
842
|
-
metricsRecorder?.recordCost(toolName,
|
|
959
|
+
metricsRecorder?.recordCost(toolName, validModel, costUsd, pricingStatus);
|
|
843
960
|
|
|
844
961
|
const budgetResult = budgetTracker.recordAndCheck(sessionId, toolName, costUsd);
|
|
845
962
|
if (budgetResult.exceeded) {
|
|
@@ -853,7 +970,21 @@ function applyCostAttribution(span, metricsRecorder, toolName, sessionId, result
|
|
|
853
970
|
// never saw it. recordUnpriced() no-ops unless a budget is actually
|
|
854
971
|
// configured, and warns at most once per tracker instance — see its
|
|
855
972
|
// own docblock (src/cost/budget.js).
|
|
856
|
-
|
|
973
|
+
//
|
|
974
|
+
// validModel, deliberately NOT usage.model: recordUnpriced()'s OWN
|
|
975
|
+
// pre-existing warning (cost/budget.js) names whatever `model` it's
|
|
976
|
+
// given verbatim (`isNonEmptyString(model) ? \`"${model}"\` : ...`)
|
|
977
|
+
// — it was written for "a real, short model name that just has no
|
|
978
|
+
// price," where that's safe. Passing the RAW rejected usage.model
|
|
979
|
+
// through unchanged here would let THAT warning re-leak the exact
|
|
980
|
+
// content warnRejectedModel() above was just careful not to —
|
|
981
|
+
// relocating ADR 019 Part 2's problem into a different diag.warn()
|
|
982
|
+
// call instead of closing it. validModel is undefined for a
|
|
983
|
+
// rejected value, which recordUnpriced() already renders as
|
|
984
|
+
// "(no model detected)" — an approximation ("rejected" collapses
|
|
985
|
+
// into "not detected"), accepted deliberately: warnRejectedModel()
|
|
986
|
+
// above is the one place that's supposed to say more, safely.
|
|
987
|
+
budgetTracker.recordUnpriced(validModel);
|
|
857
988
|
}
|
|
858
989
|
|
|
859
990
|
return { tokensIn: usage.inputTokens, tokensOut: usage.outputTokens, costUsd: costUsd ?? 0 };
|
|
@@ -1216,6 +1347,96 @@ function applyToolOutcomeThrown(toolOutcomeCounter) {
|
|
|
1216
1347
|
}
|
|
1217
1348
|
}
|
|
1218
1349
|
|
|
1350
|
+
/**
|
|
1351
|
+
* Records a thrown exception onto `span` per `errorRecordingConfig.mode`
|
|
1352
|
+
* (ADR 019 Part 1, docs/adr/019-raw-content-on-spans.md, v0.13.0 Phase 1).
|
|
1353
|
+
* The only two call sites are the thrown-exception catch blocks in
|
|
1354
|
+
* wrapToolCallHandler()/wrapToolsListHandler() below — a tool-level
|
|
1355
|
+
* `isError: true` result never carries a JS Error at all and is
|
|
1356
|
+
* unaffected by this option (see its own `span.setStatus({ code:
|
|
1357
|
+
* SpanStatusCode.ERROR })` with no message, a few lines above in
|
|
1358
|
+
* wrapToolCallHandler — the exact pattern 'none' mode below reuses).
|
|
1359
|
+
*
|
|
1360
|
+
* - `'full'` (default): `span.recordException(err)` +
|
|
1361
|
+
* `setStatus({ message: err?.message })` — byte-identical to every
|
|
1362
|
+
* release before v0.13.0. This is the literal two-line call the
|
|
1363
|
+
* catch blocks used inline before this function existed; taken
|
|
1364
|
+
* unconditionally, with no new branching evaluated first, whenever
|
|
1365
|
+
* mode is `'full'`.
|
|
1366
|
+
* - `'none'`: `setStatus({ code: ERROR })` only — no message, no
|
|
1367
|
+
* `exception` event.
|
|
1368
|
+
* - `'normalized'`: a hand-built `exception` event carrying the same
|
|
1369
|
+
* three keys `recordException()` itself would set
|
|
1370
|
+
* (`exception.type`/`exception.message`/`exception.stacktrace`), but
|
|
1371
|
+
* with `exception.message` run through `normalizeMessage()` and
|
|
1372
|
+
* `exception.stacktrace` rebuilt from `parseAndNormalizeStack()`'s
|
|
1373
|
+
* frames (cwd-stripped, `node_modules`-version-collapsed) — never the
|
|
1374
|
+
* raw `err.stack`. `err` itself is never mutated: both call sites
|
|
1375
|
+
* rethrow the original object afterward, and `computeFingerprint()` a
|
|
1376
|
+
* few lines later (when fingerprinting is enabled) reads
|
|
1377
|
+
* `err.message`/`err.stack` independently, on the real, unmutated
|
|
1378
|
+
* object.
|
|
1379
|
+
*
|
|
1380
|
+
* `error.type` (`ATTR_ERROR_TYPE`, set by each call site immediately
|
|
1381
|
+
* after this call returns) is capped at `MAX_ERROR_CLASS_LENGTH`
|
|
1382
|
+
* regardless of mode — see each call site's own comment. The native
|
|
1383
|
+
* `exception.type` a `'full'`-mode `span.recordException(err)` call sets
|
|
1384
|
+
* internally is deliberately NOT capped here: `'full'` mode is defined
|
|
1385
|
+
* as byte-identical to pre-v0.13.0 behavior, so the native OTel SDK call
|
|
1386
|
+
* is left completely untouched, uncapped exception.type included. Only
|
|
1387
|
+
* `'normalized'` mode's hand-built event applies the cap to
|
|
1388
|
+
* `exception.type` too (mirroring `recordException()`'s own
|
|
1389
|
+
* `err.code?.toString() ?? err.name` derivation for what goes in that
|
|
1390
|
+
* field), since that event is already being constructed by hand either
|
|
1391
|
+
* way — there is no "untouched native call" to preserve for that mode.
|
|
1392
|
+
*
|
|
1393
|
+
* Never throws — matches this library's fail-open discipline for
|
|
1394
|
+
* everything downstream of a real thrown error, the same guarantee
|
|
1395
|
+
* `fingerprint/compose.js`'s `computeFingerprint()` already makes for
|
|
1396
|
+
* the exact same `err`.
|
|
1397
|
+
*
|
|
1398
|
+
* @param {import('@opentelemetry/api').Span} span
|
|
1399
|
+
* @param {unknown} err
|
|
1400
|
+
* @param {import('./error-recording/config.js').ErrorRecordingConfig} errorRecordingConfig
|
|
1401
|
+
*/
|
|
1402
|
+
function recordThrownException(span, err, errorRecordingConfig) {
|
|
1403
|
+
try {
|
|
1404
|
+
if (errorRecordingConfig.mode === 'none') {
|
|
1405
|
+
span.setStatus({ code: SpanStatusCode.ERROR });
|
|
1406
|
+
return;
|
|
1407
|
+
}
|
|
1408
|
+
|
|
1409
|
+
if (errorRecordingConfig.mode === 'normalized') {
|
|
1410
|
+
const rawMessage = err?.message;
|
|
1411
|
+
const message = typeof rawMessage === 'string' ? normalizeMessage(rawMessage) : undefined;
|
|
1412
|
+
|
|
1413
|
+
const rawStack = err?.stack;
|
|
1414
|
+
const { frames } = parseAndNormalizeStack(typeof rawStack === 'string' ? rawStack : undefined, {
|
|
1415
|
+
cwd: process.cwd(),
|
|
1416
|
+
});
|
|
1417
|
+
const stacktrace =
|
|
1418
|
+
frames.length > 0
|
|
1419
|
+
? frames.map((frame) => `${frame.fn || '<anonymous>'}@${frame.file}:${frame.line ?? '?'}`).join('\n')
|
|
1420
|
+
: undefined;
|
|
1421
|
+
|
|
1422
|
+
const exceptionType = String(err?.code ?? err?.name ?? 'Error').slice(0, MAX_ERROR_CLASS_LENGTH);
|
|
1423
|
+
|
|
1424
|
+
const attrs = { 'exception.type': exceptionType };
|
|
1425
|
+
if (message !== undefined) attrs['exception.message'] = message;
|
|
1426
|
+
if (stacktrace !== undefined) attrs['exception.stacktrace'] = stacktrace;
|
|
1427
|
+
span.addEvent('exception', attrs);
|
|
1428
|
+
span.setStatus({ code: SpanStatusCode.ERROR, message });
|
|
1429
|
+
return;
|
|
1430
|
+
}
|
|
1431
|
+
|
|
1432
|
+
// 'full' (default) — byte-identical to pre-v0.13.0 behavior.
|
|
1433
|
+
span.recordException(err);
|
|
1434
|
+
span.setStatus({ code: SpanStatusCode.ERROR, message: err?.message });
|
|
1435
|
+
} catch {
|
|
1436
|
+
// Never throw — see this function's own docblock.
|
|
1437
|
+
}
|
|
1438
|
+
}
|
|
1439
|
+
|
|
1219
1440
|
/**
|
|
1220
1441
|
* Wraps a tools/call handler in a span covering its execution, plus the
|
|
1221
1442
|
* mcp.tool.* metrics (see src/metrics.js). This sits as the innermost layer
|
|
@@ -1309,6 +1530,11 @@ function applyToolOutcomeThrown(toolOutcomeCounter) {
|
|
|
1309
1530
|
* @param {'v1' | 'v2'} kind - ADR 015 Phase 2: which SDK `server` came from, resolved once by
|
|
1310
1531
|
* detectServerKind() at instrument time — determines how `sessionId`/`requestId` are read off
|
|
1311
1532
|
* the handler's second argument (`extra` for v1, `ctx` for v2 — see extractSessionAndRequestId()).
|
|
1533
|
+
* @param {import('./error-recording/config.js').ErrorRecordingConfig} errorRecordingConfig - ADR 019 Part 1
|
|
1534
|
+
* (v0.13.0): controls what the thrown-exception branch below puts on the span — see
|
|
1535
|
+
* recordThrownException()'s own docblock.
|
|
1536
|
+
* @param {{ warnedRejectedModel: boolean }} costAttributionState - ADR 019 Part 2 (v0.13.0): threaded
|
|
1537
|
+
* through to applyCostAttribution() — see that function's own docblock and warnRejectedModel()'s.
|
|
1312
1538
|
*/
|
|
1313
1539
|
function wrapToolCallHandler(
|
|
1314
1540
|
handler,
|
|
@@ -1325,6 +1551,8 @@ function wrapToolCallHandler(
|
|
|
1325
1551
|
toolOutcomeCounter,
|
|
1326
1552
|
server,
|
|
1327
1553
|
kind,
|
|
1554
|
+
errorRecordingConfig,
|
|
1555
|
+
costAttributionState,
|
|
1328
1556
|
) {
|
|
1329
1557
|
return (request, extra) => {
|
|
1330
1558
|
const toolName = request?.params?.name;
|
|
@@ -1417,7 +1645,16 @@ function wrapToolCallHandler(
|
|
|
1417
1645
|
span.setAttribute(ATTRIBUTE_KEYS.VALIDATION_PATHS, validationPaths);
|
|
1418
1646
|
}
|
|
1419
1647
|
}
|
|
1420
|
-
const usage = applyCostAttribution(
|
|
1648
|
+
const usage = applyCostAttribution(
|
|
1649
|
+
span,
|
|
1650
|
+
metricsRecorder,
|
|
1651
|
+
toolName,
|
|
1652
|
+
sessionId,
|
|
1653
|
+
result,
|
|
1654
|
+
costTracking,
|
|
1655
|
+
budgetTracker,
|
|
1656
|
+
costAttributionState,
|
|
1657
|
+
);
|
|
1421
1658
|
// ADR 007: a 'protocol.output' failure recovered above (a
|
|
1422
1659
|
// McpServer-disguised output-schema bug) must not be counted as
|
|
1423
1660
|
// thrash here either, same as the thrown branch below —
|
|
@@ -1446,7 +1683,7 @@ function wrapToolCallHandler(
|
|
|
1446
1683
|
);
|
|
1447
1684
|
} else {
|
|
1448
1685
|
span.setStatus({ code: SpanStatusCode.OK });
|
|
1449
|
-
applyCostAttribution(span, metricsRecorder, toolName, sessionId, result, costTracking, budgetTracker);
|
|
1686
|
+
applyCostAttribution(span, metricsRecorder, toolName, sessionId, result, costTracking, budgetTracker, costAttributionState);
|
|
1450
1687
|
if (thrashSessionId !== null) {
|
|
1451
1688
|
applyThrashSuccessClear(thrashConfig, thrashDetector, thrashSessionId, toolName);
|
|
1452
1689
|
}
|
|
@@ -1455,9 +1692,16 @@ function wrapToolCallHandler(
|
|
|
1455
1692
|
return result;
|
|
1456
1693
|
} catch (err) {
|
|
1457
1694
|
applyToolOutcomeThrown(toolOutcomeCounter);
|
|
1458
|
-
span
|
|
1459
|
-
|
|
1460
|
-
|
|
1695
|
+
recordThrownException(span, err, errorRecordingConfig);
|
|
1696
|
+
// ADR 019 Part 1: capped at MAX_ERROR_CLASS_LENGTH unconditionally,
|
|
1697
|
+
// regardless of errorRecordingConfig.mode — a length cap on a
|
|
1698
|
+
// class-identifier field costs a well-behaved tool nothing, so
|
|
1699
|
+
// there's no default-behavior tension to gate it behind (contrast
|
|
1700
|
+
// recordThrownException()'s own mode branching, which does gate
|
|
1701
|
+
// message/stacktrace CONTENT). Same coercion compose.js's
|
|
1702
|
+
// errorClass applies (a non-string .name is stringified, not
|
|
1703
|
+
// thrown on) — see fingerprint/compose.js's rawErrorClass.
|
|
1704
|
+
const errorType = String(err?.name ?? 'Error').slice(0, MAX_ERROR_CLASS_LENGTH);
|
|
1461
1705
|
span.setAttribute(ATTR_ERROR_TYPE, errorType);
|
|
1462
1706
|
|
|
1463
1707
|
let failureCategory = '';
|
|
@@ -1549,8 +1793,13 @@ function wrapToolCallHandler(
|
|
|
1549
1793
|
* @param {string} schemaDriftScope - Fixed per instrumented server instance — see instrumentMcpServer().
|
|
1550
1794
|
* @param {'v1' | 'v2'} kind - ADR 015 Phase 2: see wrapToolCallHandler's own `kind` param and
|
|
1551
1795
|
* extractSessionAndRequestId() — same requestId-only extraction, no sessionId use here.
|
|
1796
|
+
* @param {import('./error-recording/config.js').ErrorRecordingConfig} errorRecordingConfig - ADR 019 Part 1
|
|
1797
|
+
* (v0.13.0): see wrapToolCallHandler's own param and recordThrownException()'s docblock. The thrown-
|
|
1798
|
+
* exception branch below is the only place this handler's own JS Error content reaches a span; a
|
|
1799
|
+
* schema-drift capture failure (caught separately just above) is diag.debug()-only and never reaches
|
|
1800
|
+
* the span at all, so it's unaffected by this option.
|
|
1552
1801
|
*/
|
|
1553
|
-
function wrapToolsListHandler(handler, tracer, schemaDriftDetector, schemaDriftEmitter, schemaDriftScope, kind) {
|
|
1802
|
+
function wrapToolsListHandler(handler, tracer, schemaDriftDetector, schemaDriftEmitter, schemaDriftScope, kind, errorRecordingConfig) {
|
|
1554
1803
|
return (request, extra) => {
|
|
1555
1804
|
return tracer.startActiveSpan(TOOLS_LIST_METHOD, { kind: SpanKind.SERVER }, async (span) => {
|
|
1556
1805
|
span.setAttribute(ATTR_MCP_METHOD_NAME, TOOLS_LIST_METHOD);
|
|
@@ -1580,8 +1829,7 @@ function wrapToolsListHandler(handler, tracer, schemaDriftDetector, schemaDriftE
|
|
|
1580
1829
|
span.setStatus({ code: SpanStatusCode.OK });
|
|
1581
1830
|
return result;
|
|
1582
1831
|
} catch (err) {
|
|
1583
|
-
span
|
|
1584
|
-
span.setStatus({ code: SpanStatusCode.ERROR, message: err?.message });
|
|
1832
|
+
recordThrownException(span, err, errorRecordingConfig);
|
|
1585
1833
|
throw err;
|
|
1586
1834
|
} finally {
|
|
1587
1835
|
span.end();
|