@anvia/langfuse 0.2.9-preview.20260621T024716.ef3553c → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +425 -0
- package/dist/index.d.ts +164 -4
- package/dist/index.js +1172 -107
- package/dist/index.js.map +1 -1
- package/package.json +9 -7
package/README.md
CHANGED
|
@@ -49,6 +49,49 @@ await tracing.flush();
|
|
|
49
49
|
|
|
50
50
|
Use `flush()` after short-lived jobs. Use `shutdown()` when the process is exiting.
|
|
51
51
|
|
|
52
|
+
## Configuration
|
|
53
|
+
|
|
54
|
+
`langfuse.create()` accepts the following options. Any option that is
|
|
55
|
+
left undefined falls back to the matching environment variable, and
|
|
56
|
+
explicit options always win.
|
|
57
|
+
|
|
58
|
+
| Option | Environment variable | Notes |
|
|
59
|
+
| ------------- | ----------------------------- | ---------------------------------------------------------- |
|
|
60
|
+
| `publicKey` | `LANGFUSE_PUBLIC_KEY` | Required for score publishing. |
|
|
61
|
+
| `secretKey` | `LANGFUSE_SECRET_KEY` | Required for score publishing. |
|
|
62
|
+
| `baseUrl` | `LANGFUSE_BASE_URL` | Defaults to `https://cloud.langfuse.com`. |
|
|
63
|
+
| `environment` | `LANGFUSE_TRACING_ENVIRONMENT`| Tag attached to every trace. |
|
|
64
|
+
| `release` | `LANGFUSE_RELEASE` | Tag attached to every trace. |
|
|
65
|
+
| `serviceName` | `LANGFUSE_SERVICE_NAME` | Recorded on the root observation and as the OTel `service.name` resource attribute. |
|
|
66
|
+
|
|
67
|
+
```ts
|
|
68
|
+
const tracing = langfuse.create({
|
|
69
|
+
// All fields are optional and fall back to env vars.
|
|
70
|
+
serviceName: "support-agent",
|
|
71
|
+
});
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Observation metadata
|
|
75
|
+
|
|
76
|
+
The adapter records extra data on Langfuse observations so the UI
|
|
77
|
+
shows everything the agent runtime emits:
|
|
78
|
+
|
|
79
|
+
- **Generation observations** carry `providerRequest` and `modelInfo`
|
|
80
|
+
(with `provider`, `defaultModel`, and `capabilities`) on start, and
|
|
81
|
+
`firstDeltaMs` on end. `usageDetails` always includes
|
|
82
|
+
`cachedInputTokens` and `cacheCreationInputTokens`.
|
|
83
|
+
- **Generation observations** receive a `generation.update({ output: { delta } })`
|
|
84
|
+
call for every streaming delta (`text_delta`, `reasoning_delta`,
|
|
85
|
+
`tool_call`), so the Langfuse UI reflects partial output as the
|
|
86
|
+
model produces it.
|
|
87
|
+
- **Tool observations** carry `toolDefinition` and `toolMetadata` on
|
|
88
|
+
start, and `structuredResult` on end.
|
|
89
|
+
- **Root run observation** carries `serviceName` and the configured
|
|
90
|
+
`metadata` from `AgentRunStartArgs`.
|
|
91
|
+
|
|
92
|
+
User-supplied `trace.metadata` always wins over the built-in fields
|
|
93
|
+
above (the spread order preserves it).
|
|
94
|
+
|
|
52
95
|
## Eval Scores
|
|
53
96
|
|
|
54
97
|
```ts
|
|
@@ -59,14 +102,396 @@ const reporter = createLangfuseEvalReporter(tracing);
|
|
|
59
102
|
|
|
60
103
|
The reporter reads trace information from eval output when available, then publishes metric scores to Langfuse.
|
|
61
104
|
|
|
105
|
+
### Eval reporter options
|
|
106
|
+
|
|
107
|
+
```ts
|
|
108
|
+
const reporter = createLangfuseEvalReporter(tracing, {
|
|
109
|
+
publishInvalid: false, // publish invalid outcomes as zero scores
|
|
110
|
+
onMissingTrace: "ignore", // "ignore" | "warn" | "throw"
|
|
111
|
+
truncateInputAt: 2048, // max bytes for case input/expected summaries
|
|
112
|
+
includeMessages: true, // include output.messages in score metadata
|
|
113
|
+
});
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
- `onMissingTrace` decides what happens when no trace can be
|
|
117
|
+
resolved for a case. `"ignore"` (default, also when `strict` is
|
|
118
|
+
not set) drops the score silently. `"warn"` logs a
|
|
119
|
+
`console.warn`. `"throw"` rejects with an error. The legacy
|
|
120
|
+
`strict: true` option continues to work as an alias for
|
|
121
|
+
`"throw"`.
|
|
122
|
+
|
|
123
|
+
- `truncateInputAt` caps the byte size of `caseInputSummary` and
|
|
124
|
+
`caseExpectedSummary` metadata keys. Truncation appends
|
|
125
|
+
`<truncated>` to the cut value.
|
|
126
|
+
|
|
127
|
+
- `includeMessages` controls whether `output.messages` (if present)
|
|
128
|
+
is included in score metadata.
|
|
129
|
+
|
|
130
|
+
### Trace resolution
|
|
131
|
+
|
|
132
|
+
The reporter resolves a trace ID for each case in three tiers:
|
|
133
|
+
|
|
134
|
+
1. `output.trace` (most direct, set by an agent run).
|
|
135
|
+
2. `case.input.trace` (useful when the case input bundles trace
|
|
136
|
+
info).
|
|
137
|
+
3. `case.metadata.traceId` (and optional `observationId`).
|
|
138
|
+
|
|
139
|
+
### Metric annotations
|
|
140
|
+
|
|
141
|
+
`EvalMetric` (in `@anvia/core`) accepts optional `dataType`,
|
|
142
|
+
`configId` / `scoreConfigId`, and `metadata` fields. The reporter
|
|
143
|
+
forwards them to Langfuse, so categorical or boolean metrics are
|
|
144
|
+
sent with the right shape:
|
|
145
|
+
|
|
146
|
+
```ts
|
|
147
|
+
import { defineMetric, EvalOutcome } from "@anvia/core";
|
|
148
|
+
|
|
149
|
+
const judge = defineMetric({
|
|
150
|
+
name: "quality",
|
|
151
|
+
dataType: "CATEGORICAL",
|
|
152
|
+
configId: "quality-config",
|
|
153
|
+
metadata: { source: "judge-llm" },
|
|
154
|
+
evaluate: () => EvalOutcome.pass("good"),
|
|
155
|
+
});
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
`defineMetric` is a small identity helper that signals intent and
|
|
159
|
+
preserves type inference; plain object literals continue to work.
|
|
160
|
+
|
|
161
|
+
### Typed scores and overrides
|
|
162
|
+
|
|
163
|
+
`tracing.score()` accepts a `dataType` (`"NUMERIC" | "CATEGORICAL" | "BOOLEAN"`), a `configId` (or its `scoreConfigId` alias), a per-score `environment` override, and a `timestamp` (Date or ISO 8601 string). The adapter validates `value` against the dataType at the boundary.
|
|
164
|
+
|
|
165
|
+
```ts
|
|
166
|
+
await tracing.score({
|
|
167
|
+
traceId: trace.traceId,
|
|
168
|
+
name: "verdict",
|
|
169
|
+
value: "pass", // string for CATEGORICAL
|
|
170
|
+
dataType: "CATEGORICAL",
|
|
171
|
+
configId: "cfg-1",
|
|
172
|
+
environment: "staging",
|
|
173
|
+
timestamp: new Date(),
|
|
174
|
+
});
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
The score fetch has a default timeout of 30 s, overrideable via
|
|
178
|
+
`langfuse.create({ timeoutMs: ... })`.
|
|
179
|
+
|
|
180
|
+
### Batching and retry (high-volume evals)
|
|
181
|
+
|
|
182
|
+
Enable the in-memory score queue by setting `scoreBatchSize` on
|
|
183
|
+
`langfuse.create()`. When enabled, `tracing.score()` enqueues the
|
|
184
|
+
score and returns immediately. The queue flushes when it reaches
|
|
185
|
+
`scoreBatchSize`, on a debounce timer (`scoreFlushIntervalMs`,
|
|
186
|
+
default 250 ms), and on `flushScores()`, `flush()`, or `shutdown()`.
|
|
187
|
+
|
|
188
|
+
```ts
|
|
189
|
+
const tracing = langfuse.create({
|
|
190
|
+
publicKey,
|
|
191
|
+
secretKey,
|
|
192
|
+
scoreBatchSize: 20, // enable queue; flushes at 20 items
|
|
193
|
+
scoreFlushIntervalMs: 500, // or after 500ms
|
|
194
|
+
scoreMaxRetries: 3, // retry 429 / 5xx with backoff
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
await tracing.score({ traceId, name: "quality", value: 1 });
|
|
198
|
+
await tracing.score({ traceId, name: "latency", value: 0.4 });
|
|
199
|
+
await tracing.flushScores(); // drain the queue
|
|
200
|
+
console.log(tracing.scoreQueueDepth()); // 0
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
`flush()` and `shutdown()` also drain the score queue. After all
|
|
204
|
+
retries are exhausted, the queue throws a `LangfuseScoreError` whose
|
|
205
|
+
`scores` property contains the failed payloads so you can inspect
|
|
206
|
+
what was lost.
|
|
207
|
+
|
|
208
|
+
## Event observations & trace handle
|
|
209
|
+
|
|
210
|
+
After a run starts, you can record ad-hoc checkpoints and attach
|
|
211
|
+
extra attributes to the active trace without threading the run
|
|
212
|
+
observer through every function call. The tracing instance exposes
|
|
213
|
+
the most recent trace through `getCurrentTrace()`:
|
|
214
|
+
|
|
215
|
+
```ts
|
|
216
|
+
import { langfuse } from "@anvia/langfuse";
|
|
217
|
+
|
|
218
|
+
const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
|
|
219
|
+
|
|
220
|
+
await tracing.startRun({
|
|
221
|
+
agentName: "support",
|
|
222
|
+
prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
|
|
223
|
+
history: [],
|
|
224
|
+
maxTurns: 3,
|
|
225
|
+
});
|
|
226
|
+
|
|
227
|
+
const trace = tracing.getCurrentTrace();
|
|
228
|
+
|
|
229
|
+
trace?.addEvent("retrieval.done", { docCount: 4 });
|
|
230
|
+
trace?.addEvent("validation.passed");
|
|
231
|
+
trace?.addAttributes({ quality: "high" });
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
`addEvent` creates a Langfuse `event` observation under the active
|
|
235
|
+
root and ends it immediately. `addAttributes` updates the root
|
|
236
|
+
observation's metadata. Both calls bubble up to Langfuse via the
|
|
237
|
+
existing OpenTelemetry span processor.
|
|
238
|
+
|
|
239
|
+
If you have the run observer returned by `startRun`, you can also
|
|
240
|
+
use its `event?(...)` hook (added in `@anvia/core`) to record
|
|
241
|
+
checkpoints:
|
|
242
|
+
|
|
243
|
+
```ts
|
|
244
|
+
const run = await tracing.startRun({
|
|
245
|
+
agentName: "support",
|
|
246
|
+
prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
|
|
247
|
+
history: [],
|
|
248
|
+
maxTurns: 3,
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
await run.event?.({
|
|
252
|
+
name: "retrieval.done",
|
|
253
|
+
attributes: { docCount: 4 },
|
|
254
|
+
});
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
The trace handle is cleared when the run `end`s or `error`s. If you
|
|
258
|
+
call `addEvent` or `addAttributes` after that, the underlying
|
|
259
|
+
Langfuse SDK will reject the call because the root observation is
|
|
260
|
+
no longer accepting children. We let the error propagate so it is
|
|
261
|
+
visible.
|
|
262
|
+
|
|
263
|
+
`getCurrentTrace()` is per-tracing-instance and last-write-wins:
|
|
264
|
+
when multiple runs start on the same instance, the handle always
|
|
265
|
+
points at the most recent run. For true concurrency, manage your
|
|
266
|
+
own context (e.g. capture the run observer or build your own
|
|
267
|
+
mapping keyed on user/session).
|
|
268
|
+
|
|
269
|
+
## Datasets & Experiment runs
|
|
270
|
+
|
|
271
|
+
Bridge `@anvia/core/evals` to Langfuse's dataset / experiment-run
|
|
272
|
+
workflow. The dataset client surfaces the four endpoints you need
|
|
273
|
+
to create a dataset, fetch its items, upsert items, and post a
|
|
274
|
+
batched dataset-run-items payload.
|
|
275
|
+
|
|
276
|
+
```ts
|
|
277
|
+
import {
|
|
278
|
+
createLangfuseDatasetClient,
|
|
279
|
+
runEvalAsExperiment,
|
|
280
|
+
langfuse,
|
|
281
|
+
} from "@anvia/langfuse";
|
|
282
|
+
|
|
283
|
+
const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
|
|
284
|
+
const client = createLangfuseDatasetClient(tracing);
|
|
285
|
+
|
|
286
|
+
await client.createDataset({ name: "support-smoke" });
|
|
287
|
+
await client.upsertItems("support-smoke", [
|
|
288
|
+
{ id: "c-1", input: { q: "hi" }, expected: "hello" },
|
|
289
|
+
{ id: "c-2", input: { q: "bye" } },
|
|
290
|
+
]);
|
|
291
|
+
|
|
292
|
+
const dataset = await client.getDataset("support-smoke");
|
|
293
|
+
console.log(dataset.items);
|
|
294
|
+
|
|
295
|
+
await client.runExperiment({
|
|
296
|
+
datasetName: "support-smoke",
|
|
297
|
+
runName: "smoke-2026-06",
|
|
298
|
+
run: (item) => ({
|
|
299
|
+
output: `answer-for-${item.id}`,
|
|
300
|
+
trace: { traceId: `trace-${item.id}` },
|
|
301
|
+
}),
|
|
302
|
+
});
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
For eval suites, `runEvalAsExperiment` runs the suite and posts a
|
|
306
|
+
dataset run alongside the metric scores:
|
|
307
|
+
|
|
308
|
+
```ts
|
|
309
|
+
import { runEvalAsExperiment } from "@anvia/langfuse";
|
|
310
|
+
|
|
311
|
+
const { suite, datasetRun } = await runEvalAsExperiment(
|
|
312
|
+
{
|
|
313
|
+
name: "smoke",
|
|
314
|
+
cases: [
|
|
315
|
+
{ id: "c-1", input: "a", expected: "A" },
|
|
316
|
+
{ id: "c-2", input: "b", expected: "B" },
|
|
317
|
+
],
|
|
318
|
+
target: async (input) => input.toUpperCase(),
|
|
319
|
+
metrics: [/* ... */],
|
|
320
|
+
reporters: [/* existing reporters still score */],
|
|
321
|
+
},
|
|
322
|
+
{
|
|
323
|
+
tracing,
|
|
324
|
+
datasetName: "smoke-set",
|
|
325
|
+
runName: "smoke-run",
|
|
326
|
+
},
|
|
327
|
+
);
|
|
328
|
+
|
|
329
|
+
console.log(suite.passed, datasetRun.posted);
|
|
330
|
+
```
|
|
331
|
+
|
|
332
|
+
### Options
|
|
333
|
+
|
|
334
|
+
- `createLangfuseDatasetClient(tracing, options)`:
|
|
335
|
+
- `publicKey`, `secretKey`, `baseUrl`: optionally override the
|
|
336
|
+
tracing instance's resolved values. Falls back to env vars
|
|
337
|
+
(`LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_BASE_URL`).
|
|
338
|
+
- `pageSize` (default `50`): dataset item pagination.
|
|
339
|
+
- `timeoutMs` (default `30_000`): per-request timeout via
|
|
340
|
+
`AbortSignal.timeout`.
|
|
341
|
+
- `client.runExperiment({ ... })`:
|
|
342
|
+
- `items` (optional): supply items directly to skip the
|
|
343
|
+
`getDataset` round trip.
|
|
344
|
+
- `run`: invoked per item, returns `{ output, trace? }`.
|
|
345
|
+
Per-item failures are caught and surfaced in the result's
|
|
346
|
+
`errors` array; only successful items reach the batched POST.
|
|
347
|
+
|
|
348
|
+
### Failure handling
|
|
349
|
+
|
|
350
|
+
`runExperiment` continues on per-item errors. The `errors` array
|
|
351
|
+
contains `{ itemId, error }` entries for failed items; the POST
|
|
352
|
+
still goes out with the successful subset. Non-2xx on the
|
|
353
|
+
batched POST throws (and the items that already errored are
|
|
354
|
+
also included in the thrown error's context).
|
|
355
|
+
|
|
356
|
+
## Prompt management
|
|
357
|
+
|
|
358
|
+
Pull prompts from Langfuse's prompt store and link generations to
|
|
359
|
+
the prompt version. The tracing instance attaches the prompt name
|
|
360
|
+
and version to the root trace and every generation in the run
|
|
361
|
+
when a prompt ref is configured.
|
|
362
|
+
|
|
363
|
+
```ts
|
|
364
|
+
import { createLangfusePromptClient, langfuse } from "@anvia/langfuse";
|
|
365
|
+
|
|
366
|
+
const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
|
|
367
|
+
const prompts = createLangfusePromptClient(tracing);
|
|
368
|
+
|
|
369
|
+
const prompt = await prompts.getPrompt("support.system");
|
|
370
|
+
console.log(prompt.prompt, prompt.version);
|
|
371
|
+
|
|
372
|
+
await tracing.startRun({
|
|
373
|
+
agentName: "support",
|
|
374
|
+
prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
|
|
375
|
+
history: [],
|
|
376
|
+
maxTurns: 3,
|
|
377
|
+
promptRef: { name: "support.system", version: prompt.version },
|
|
378
|
+
});
|
|
379
|
+
```
|
|
380
|
+
|
|
381
|
+
`promptRef` is also accepted on `trace.metadata` (keys
|
|
382
|
+
`promptName` and `promptVersion`) for back-compat with users who
|
|
383
|
+
already attach metadata to `withTrace(...)`.
|
|
384
|
+
|
|
385
|
+
### Prompt client options
|
|
386
|
+
|
|
387
|
+
- `cacheTtlMs` (default `60_000`): in-memory TTL per
|
|
388
|
+
`${name}::${version}::${label}` key.
|
|
389
|
+
- `timeoutMs` (default `30_000`): per-request timeout via
|
|
390
|
+
`AbortSignal.timeout`.
|
|
391
|
+
- `publicKey`, `secretKey`, `baseUrl`: override the tracing
|
|
392
|
+
instance's resolved values, falling back to env vars.
|
|
393
|
+
|
|
394
|
+
### Per-call options
|
|
395
|
+
|
|
396
|
+
- `getPrompt(name, { version?, label?, cacheTtlMs?, refresh? })`:
|
|
397
|
+
- `version` and `label` filter the upstream request.
|
|
398
|
+
- `cacheTtlMs` overrides the client default for this call.
|
|
399
|
+
- `refresh: true` skips the cache and re-fetches.
|
|
400
|
+
|
|
401
|
+
### Helpers
|
|
402
|
+
|
|
403
|
+
- `getPromptText(name, options?)`: returns the prompt string;
|
|
404
|
+
throws if the prompt is a chat prompt.
|
|
405
|
+
- `getPromptChat(name, options?)`: returns the chat message
|
|
406
|
+
array; throws if the prompt is a text prompt.
|
|
407
|
+
- `refresh()`: clears the cache.
|
|
408
|
+
|
|
409
|
+
## PII redaction
|
|
410
|
+
|
|
411
|
+
Mask personally identifiable information in observations before it
|
|
412
|
+
leaves the process. The default pattern set catches emails,
|
|
413
|
+
credit cards (with Luhn validation), phone numbers, IPv4
|
|
414
|
+
addresses, JWTs, and common API key shapes.
|
|
415
|
+
|
|
416
|
+
```ts
|
|
417
|
+
import { langfuse } from "@anvia/langfuse";
|
|
418
|
+
|
|
419
|
+
const tracing = langfuse.create({
|
|
420
|
+
publicKey: "pk",
|
|
421
|
+
secretKey: "sk",
|
|
422
|
+
redactInputs: true,
|
|
423
|
+
redactOutputs: "deep",
|
|
424
|
+
});
|
|
425
|
+
```
|
|
426
|
+
|
|
427
|
+
- `redactInputs` redacts text on root inputs, chat history, and tool
|
|
428
|
+
arguments.
|
|
429
|
+
- `redactOutputs` redacts generation output text and tool results.
|
|
430
|
+
- `"deep"` recurses into nested objects and arrays (in addition to
|
|
431
|
+
top-level strings).
|
|
432
|
+
|
|
433
|
+
### Customizing
|
|
434
|
+
|
|
435
|
+
```ts
|
|
436
|
+
const tracing = langfuse.create({
|
|
437
|
+
publicKey: "pk",
|
|
438
|
+
secretKey: "sk",
|
|
439
|
+
redactOutputs: true,
|
|
440
|
+
redaction: {
|
|
441
|
+
replacement: "[HIDDEN]",
|
|
442
|
+
patterns: [
|
|
443
|
+
{ name: "ssn", regex: /\b\d{3}-\d{2}-\d{4}\b/g },
|
|
444
|
+
],
|
|
445
|
+
},
|
|
446
|
+
});
|
|
447
|
+
```
|
|
448
|
+
|
|
449
|
+
`createPiiRedactor(options)` is also exported for ad-hoc use:
|
|
450
|
+
|
|
451
|
+
```ts
|
|
452
|
+
import { createPiiRedactor } from "@anvia/langfuse";
|
|
453
|
+
|
|
454
|
+
const redactor = createPiiRedactor();
|
|
455
|
+
const safe = redactor.redactString("email alice@example.com today");
|
|
456
|
+
const safeMessages = redactor.redactMessages(messages);
|
|
457
|
+
const safeObject = redactor.redactObject(spanInput);
|
|
458
|
+
```
|
|
459
|
+
|
|
62
460
|
## Exports
|
|
63
461
|
|
|
64
462
|
- `langfuse`
|
|
463
|
+
- `createLangfuseDatasetClient`
|
|
65
464
|
- `createLangfuseEvalReporter`
|
|
465
|
+
- `createLangfusePromptClient`
|
|
466
|
+
- `runEvalAsExperiment`
|
|
467
|
+
- `LangfuseScoreError`
|
|
66
468
|
- `LangfuseTracing`
|
|
469
|
+
- `LangfuseTraceHandle`
|
|
67
470
|
- `LangfuseTracingOptions`
|
|
68
471
|
- `LangfuseScoreArgs`
|
|
472
|
+
- `LangfuseScoreDataType`
|
|
69
473
|
- `LangfuseEvalReporterOptions`
|
|
474
|
+
- `LangfuseDatasetClient`
|
|
475
|
+
- `LangfuseDatasetClientOptions`
|
|
476
|
+
- `LangfuseDataset`
|
|
477
|
+
- `LangfuseDatasetItem`
|
|
478
|
+
- `LangfuseRunExperimentOptions`
|
|
479
|
+
- `LangfuseRunExperimentResult`
|
|
480
|
+
- `LangfuseRunItemResult`
|
|
481
|
+
- `LangfuseRunItemError`
|
|
482
|
+
- `LangfusePromptClient`
|
|
483
|
+
- `LangfusePromptClientOptions`
|
|
484
|
+
- `LangfusePromptGetOptions`
|
|
485
|
+
- `LangfusePrompt`
|
|
486
|
+
- `LangfuseChatMessage`
|
|
487
|
+
- `RunEvalAsExperimentOptions`
|
|
488
|
+
- `RunEvalAsExperimentResult`
|
|
489
|
+
- `createPiiRedactor`
|
|
490
|
+
- `DEFAULT_PATTERNS`
|
|
491
|
+
- `PiiRedactor`
|
|
492
|
+
- `RedactorPattern`
|
|
493
|
+
- `LangfuseRedactionOptions`
|
|
494
|
+
- `LangfuseRedactionMode`
|
|
70
495
|
|
|
71
496
|
## Development
|
|
72
497
|
|
package/dist/index.d.ts
CHANGED
|
@@ -1,36 +1,196 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { JsonValue } from '@anvia/core/completion';
|
|
1
|
+
import { JsonValue, Message } from '@anvia/core/completion';
|
|
3
2
|
import { AgentObserver } from '@anvia/core/observability';
|
|
3
|
+
import { EvalReporter, EvalSuiteResult, RunEvalSuiteOptions } from '@anvia/core/evals';
|
|
4
4
|
|
|
5
|
+
type LangfuseRedactionMode = boolean | "deep";
|
|
6
|
+
type LangfuseRedactionOptions$1 = {
|
|
7
|
+
patterns?: RedactorPattern$1[];
|
|
8
|
+
replacement?: string;
|
|
9
|
+
};
|
|
10
|
+
type RedactorPattern$1 = {
|
|
11
|
+
name: string;
|
|
12
|
+
regex: RegExp;
|
|
13
|
+
};
|
|
5
14
|
type LangfuseTracingOptions = {
|
|
6
15
|
publicKey?: string | undefined;
|
|
7
16
|
secretKey?: string | undefined;
|
|
8
17
|
baseUrl?: string | undefined;
|
|
9
18
|
environment?: string | undefined;
|
|
10
19
|
release?: string | undefined;
|
|
20
|
+
serviceName?: string | undefined;
|
|
21
|
+
timeoutMs?: number | undefined;
|
|
22
|
+
scoreBatchSize?: number | undefined;
|
|
23
|
+
scoreFlushIntervalMs?: number | undefined;
|
|
24
|
+
scoreMaxRetries?: number | undefined;
|
|
25
|
+
redactInputs?: LangfuseRedactionMode | undefined;
|
|
26
|
+
redactOutputs?: LangfuseRedactionMode | undefined;
|
|
27
|
+
redaction?: LangfuseRedactionOptions$1 | undefined;
|
|
11
28
|
};
|
|
29
|
+
type LangfuseScoreDataType = "NUMERIC" | "CATEGORICAL" | "BOOLEAN";
|
|
12
30
|
type LangfuseScoreArgs = {
|
|
13
31
|
traceId?: string | undefined;
|
|
14
32
|
observationId?: string | undefined;
|
|
15
33
|
name: string;
|
|
16
|
-
value: number;
|
|
34
|
+
value: number | string;
|
|
35
|
+
dataType?: LangfuseScoreDataType | undefined;
|
|
17
36
|
comment?: string | undefined;
|
|
18
37
|
metadata?: Record<string, JsonValue | undefined> | undefined;
|
|
38
|
+
configId?: string | undefined;
|
|
39
|
+
scoreConfigId?: string | undefined;
|
|
40
|
+
environment?: string | undefined;
|
|
41
|
+
timestamp?: Date | string | undefined;
|
|
42
|
+
};
|
|
43
|
+
type LangfuseTraceHandle = {
|
|
44
|
+
readonly traceId: string;
|
|
45
|
+
readonly observationId: string;
|
|
46
|
+
addAttributes(attributes: Record<string, JsonValue | undefined>): void;
|
|
47
|
+
addEvent(name: string, attributes?: Record<string, JsonValue | undefined>): void;
|
|
19
48
|
};
|
|
20
49
|
type LangfuseTracing = AgentObserver & {
|
|
21
50
|
flush(): Promise<void>;
|
|
22
51
|
shutdown(): Promise<void>;
|
|
23
52
|
score(args: LangfuseScoreArgs): Promise<void>;
|
|
53
|
+
flushScores(): Promise<void>;
|
|
54
|
+
scoreQueueDepth(): number;
|
|
55
|
+
getCurrentTrace(): LangfuseTraceHandle | undefined;
|
|
24
56
|
};
|
|
25
57
|
type LangfuseEvalReporterOptions = {
|
|
26
58
|
publishInvalid?: boolean | undefined;
|
|
27
59
|
strict?: boolean | undefined;
|
|
60
|
+
onMissingTrace?: "ignore" | "warn" | "throw" | undefined;
|
|
61
|
+
truncateInputAt?: number | undefined;
|
|
62
|
+
includeMessages?: boolean | undefined;
|
|
63
|
+
};
|
|
64
|
+
type LangfuseDatasetClientOptions = {
|
|
65
|
+
baseUrl?: string | undefined;
|
|
66
|
+
publicKey?: string | undefined;
|
|
67
|
+
secretKey?: string | undefined;
|
|
68
|
+
pageSize?: number | undefined;
|
|
69
|
+
timeoutMs?: number | undefined;
|
|
70
|
+
};
|
|
71
|
+
type LangfuseDatasetItem<Input = unknown, Expected = unknown> = {
|
|
72
|
+
id: string;
|
|
73
|
+
input: Input;
|
|
74
|
+
expected?: Expected | undefined;
|
|
75
|
+
metadata?: Record<string, JsonValue | undefined> | undefined;
|
|
76
|
+
};
|
|
77
|
+
type LangfuseDataset<Input = unknown, Expected = unknown> = {
|
|
78
|
+
name: string;
|
|
79
|
+
description?: string | undefined;
|
|
80
|
+
metadata?: Record<string, JsonValue | undefined> | undefined;
|
|
81
|
+
items: LangfuseDatasetItem<Input, Expected>[];
|
|
82
|
+
};
|
|
83
|
+
type LangfuseRunItemResult<Output = unknown> = {
|
|
84
|
+
output: Output;
|
|
85
|
+
trace?: {
|
|
86
|
+
traceId: string;
|
|
87
|
+
observationId?: string | undefined;
|
|
88
|
+
} | undefined;
|
|
89
|
+
};
|
|
90
|
+
type LangfuseRunItemError = {
|
|
91
|
+
itemId: string;
|
|
92
|
+
error: unknown;
|
|
93
|
+
};
|
|
94
|
+
type LangfuseRunExperimentOptions<Input = unknown, Output = unknown, Expected = unknown> = {
|
|
95
|
+
datasetName: string;
|
|
96
|
+
runName: string;
|
|
97
|
+
description?: string | undefined;
|
|
98
|
+
metadata?: Record<string, JsonValue | undefined> | undefined;
|
|
99
|
+
items?: LangfuseDatasetItem<Input, Expected>[] | undefined;
|
|
100
|
+
run: (item: LangfuseDatasetItem<Input, Expected>) => LangfuseRunItemResult<Output> | Promise<LangfuseRunItemResult<Output>>;
|
|
101
|
+
};
|
|
102
|
+
type LangfuseRunExperimentResult = {
|
|
103
|
+
runName: string;
|
|
104
|
+
datasetName: string;
|
|
105
|
+
posted: number;
|
|
106
|
+
errors: LangfuseRunItemError[];
|
|
107
|
+
};
|
|
108
|
+
type LangfuseDatasetClient = {
|
|
109
|
+
createDataset(dataset: {
|
|
110
|
+
name: string;
|
|
111
|
+
description?: string | undefined;
|
|
112
|
+
metadata?: Record<string, JsonValue | undefined> | undefined;
|
|
113
|
+
}): Promise<LangfuseDataset<unknown, unknown>>;
|
|
114
|
+
getDataset<Input, Expected>(name: string): Promise<LangfuseDataset<Input, Expected>>;
|
|
115
|
+
upsertItems<Input, Expected>(name: string, items: LangfuseDatasetItem<Input, Expected>[]): Promise<void>;
|
|
116
|
+
runExperiment<Input, Output, Expected>(options: LangfuseRunExperimentOptions<Input, Output, Expected>): Promise<LangfuseRunExperimentResult>;
|
|
28
117
|
};
|
|
118
|
+
type LangfusePromptClientOptions = {
|
|
119
|
+
baseUrl?: string | undefined;
|
|
120
|
+
publicKey?: string | undefined;
|
|
121
|
+
secretKey?: string | undefined;
|
|
122
|
+
cacheTtlMs?: number | undefined;
|
|
123
|
+
timeoutMs?: number | undefined;
|
|
124
|
+
};
|
|
125
|
+
type LangfusePromptGetOptions = {
|
|
126
|
+
version?: number | undefined;
|
|
127
|
+
label?: string | undefined;
|
|
128
|
+
cacheTtlMs?: number | undefined;
|
|
129
|
+
refresh?: boolean | undefined;
|
|
130
|
+
};
|
|
131
|
+
type LangfuseChatMessage = {
|
|
132
|
+
role: "system" | "user" | "assistant" | "tool";
|
|
133
|
+
content: string;
|
|
134
|
+
};
|
|
135
|
+
type LangfusePrompt = {
|
|
136
|
+
name: string;
|
|
137
|
+
version: number;
|
|
138
|
+
labels: string[];
|
|
139
|
+
prompt: string | LangfuseChatMessage[];
|
|
140
|
+
type: "text" | "chat";
|
|
141
|
+
tags?: string[];
|
|
142
|
+
resolvedAt: Date;
|
|
143
|
+
};
|
|
144
|
+
type LangfusePromptClient = {
|
|
145
|
+
getPrompt(name: string, options?: LangfusePromptGetOptions): Promise<LangfusePrompt>;
|
|
146
|
+
getPromptText(name: string, options?: LangfusePromptGetOptions): Promise<string>;
|
|
147
|
+
getPromptChat(name: string, options?: LangfusePromptGetOptions): Promise<LangfuseChatMessage[]>;
|
|
148
|
+
refresh(): void;
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
declare function createLangfuseDatasetClient(tracing: Pick<LangfuseTracing, "score">, options?: LangfuseDatasetClientOptions): LangfuseDatasetClient;
|
|
29
152
|
|
|
30
153
|
declare function createLangfuseEvalReporter<Input = unknown, Output = unknown, Expected = unknown>(tracing: Pick<LangfuseTracing, "score">, options?: LangfuseEvalReporterOptions): EvalReporter<Input, Output, Expected>;
|
|
31
154
|
|
|
155
|
+
type RunEvalAsExperimentOptions<Input, Output, Expected = unknown> = Omit<LangfuseRunExperimentOptions<Input, Output, Expected>, "items" | "run"> & {
|
|
156
|
+
tracing: Pick<LangfuseTracing, "score">;
|
|
157
|
+
client?: LangfuseDatasetClient;
|
|
158
|
+
pageSize?: number | undefined;
|
|
159
|
+
timeoutMs?: number | undefined;
|
|
160
|
+
};
|
|
161
|
+
type RunEvalAsExperimentResult<Input, Output, Expected = unknown> = {
|
|
162
|
+
suite: EvalSuiteResult<Input, Output, Expected>;
|
|
163
|
+
datasetRun: LangfuseRunExperimentResult;
|
|
164
|
+
};
|
|
165
|
+
declare function runEvalAsExperiment<Input, Output, Expected = unknown>(evalOptions: RunEvalSuiteOptions<Input, Output, Expected>, experimentOptions: RunEvalAsExperimentOptions<Input, Output, Expected>): Promise<RunEvalAsExperimentResult<Input, Output, Expected>>;
|
|
166
|
+
|
|
167
|
+
declare function createLangfusePromptClient(tracing: Pick<LangfuseTracing, "score">, options?: LangfusePromptClientOptions): LangfusePromptClient;
|
|
168
|
+
|
|
169
|
+
type RedactorPattern = {
|
|
170
|
+
name: string;
|
|
171
|
+
regex: RegExp;
|
|
172
|
+
};
|
|
173
|
+
type LangfuseRedactionOptions = {
|
|
174
|
+
patterns?: RedactorPattern[];
|
|
175
|
+
replacement?: string;
|
|
176
|
+
};
|
|
177
|
+
type PiiRedactor = {
|
|
178
|
+
redactString(input: string): string;
|
|
179
|
+
redactObject<T>(input: T): T;
|
|
180
|
+
redactMessages(input: Message[]): Message[];
|
|
181
|
+
patternNames(): string[];
|
|
182
|
+
};
|
|
183
|
+
declare function createPiiRedactor(options?: LangfuseRedactionOptions): PiiRedactor;
|
|
184
|
+
declare const DEFAULT_PATTERNS: RedactorPattern[];
|
|
185
|
+
|
|
186
|
+
declare class LangfuseScoreError extends Error {
|
|
187
|
+
readonly scores: LangfuseScoreArgs[];
|
|
188
|
+
readonly cause?: unknown;
|
|
189
|
+
constructor(message: string, scores: LangfuseScoreArgs[], cause?: unknown);
|
|
190
|
+
}
|
|
191
|
+
|
|
32
192
|
declare const langfuse: {
|
|
33
193
|
create(options?: LangfuseTracingOptions): LangfuseTracing;
|
|
34
194
|
};
|
|
35
195
|
|
|
36
|
-
export { type LangfuseEvalReporterOptions, type LangfuseScoreArgs, type LangfuseTracing, type LangfuseTracingOptions, createLangfuseEvalReporter, langfuse };
|
|
196
|
+
export { DEFAULT_PATTERNS, type LangfuseChatMessage, type LangfuseDataset, type LangfuseDatasetClient, type LangfuseDatasetClientOptions, type LangfuseDatasetItem, type LangfuseEvalReporterOptions, type LangfusePrompt, type LangfusePromptClient, type LangfusePromptClientOptions, type LangfusePromptGetOptions, type LangfuseRedactionMode, type LangfuseRedactionOptions, type LangfuseRunExperimentOptions, type LangfuseRunExperimentResult, type LangfuseRunItemError, type LangfuseRunItemResult, type LangfuseScoreArgs, type LangfuseScoreDataType, LangfuseScoreError, type LangfuseTraceHandle, type LangfuseTracing, type LangfuseTracingOptions, type PiiRedactor, type RedactorPattern, type RunEvalAsExperimentOptions, type RunEvalAsExperimentResult, createLangfuseDatasetClient, createLangfuseEvalReporter, createLangfusePromptClient, createPiiRedactor, langfuse, runEvalAsExperiment };
|