@anvia/langfuse 0.2.9-preview.20260621T024716.ef3553c → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -49,6 +49,49 @@ await tracing.flush();
49
49
 
50
50
  Use `flush()` after short-lived jobs. Use `shutdown()` when the process is exiting.
51
51
 
52
+ ## Configuration
53
+
54
+ `langfuse.create()` accepts the following options. Any option that is
55
+ left undefined falls back to the matching environment variable, and
56
+ explicit options always win.
57
+
58
+ | Option | Environment variable | Notes |
59
+ | ------------- | ----------------------------- | ---------------------------------------------------------- |
60
+ | `publicKey` | `LANGFUSE_PUBLIC_KEY` | Required for score publishing. |
61
+ | `secretKey` | `LANGFUSE_SECRET_KEY` | Required for score publishing. |
62
+ | `baseUrl` | `LANGFUSE_BASE_URL` | Defaults to `https://cloud.langfuse.com`. |
63
+ | `environment` | `LANGFUSE_TRACING_ENVIRONMENT`| Tag attached to every trace. |
64
+ | `release` | `LANGFUSE_RELEASE` | Tag attached to every trace. |
65
+ | `serviceName` | `LANGFUSE_SERVICE_NAME` | Recorded on the root observation and as the OTel `service.name` resource attribute. |
66
+
67
+ ```ts
68
+ const tracing = langfuse.create({
69
+ // All fields are optional and fall back to env vars.
70
+ serviceName: "support-agent",
71
+ });
72
+ ```
73
+
74
+ ## Observation metadata
75
+
76
+ The adapter records extra data on Langfuse observations so the UI
77
+ shows everything the agent runtime emits:
78
+
79
+ - **Generation observations** carry `providerRequest` and `modelInfo`
80
+ (with `provider`, `defaultModel`, and `capabilities`) on start, and
81
+ `firstDeltaMs` on end. `usageDetails` always includes
82
+ `cachedInputTokens` and `cacheCreationInputTokens`.
83
+ - **Generation observations** receive a `generation.update({ output: { delta } })`
84
+ call for every streaming delta (`text_delta`, `reasoning_delta`,
85
+ `tool_call`), so the Langfuse UI reflects partial output as the
86
+ model produces it.
87
+ - **Tool observations** carry `toolDefinition` and `toolMetadata` on
88
+ start, and `structuredResult` on end.
89
+ - **Root run observation** carries `serviceName` and the configured
90
+ `metadata` from `AgentRunStartArgs`.
91
+
92
+ User-supplied `trace.metadata` always wins over the built-in fields
93
+ above (the spread order preserves it).
94
+
52
95
  ## Eval Scores
53
96
 
54
97
  ```ts
@@ -59,14 +102,396 @@ const reporter = createLangfuseEvalReporter(tracing);
59
102
 
60
103
  The reporter reads trace information from eval output when available, then publishes metric scores to Langfuse.
61
104
 
105
+ ### Eval reporter options
106
+
107
+ ```ts
108
+ const reporter = createLangfuseEvalReporter(tracing, {
109
+ publishInvalid: false, // publish invalid outcomes as zero scores
110
+ onMissingTrace: "ignore", // "ignore" | "warn" | "throw"
111
+ truncateInputAt: 2048, // max bytes for case input/expected summaries
112
+ includeMessages: true, // include output.messages in score metadata
113
+ });
114
+ ```
115
+
116
+ - `onMissingTrace` decides what happens when no trace can be
117
+ resolved for a case. `"ignore"` (default, also when `strict` is
118
+ not set) drops the score silently. `"warn"` logs a
119
+ `console.warn`. `"throw"` rejects with an error. The legacy
120
+ `strict: true` option continues to work as an alias for
121
+ `"throw"`.
122
+
123
+ - `truncateInputAt` caps the byte size of `caseInputSummary` and
124
+ `caseExpectedSummary` metadata keys. Truncation appends
125
+ `<truncated>` to the cut value.
126
+
127
+ - `includeMessages` controls whether `output.messages` (if present)
128
+ is included in score metadata.
129
+
130
+ ### Trace resolution
131
+
132
+ The reporter resolves a trace ID for each case in three tiers:
133
+
134
+ 1. `output.trace` (most direct, set by an agent run).
135
+ 2. `case.input.trace` (useful when the case input bundles trace
136
+ info).
137
+ 3. `case.metadata.traceId` (and optional `observationId`).
138
+
139
+ ### Metric annotations
140
+
141
+ `EvalMetric` (in `@anvia/core`) accepts optional `dataType`,
142
+ `configId` / `scoreConfigId`, and `metadata` fields. The reporter
143
+ forwards them to Langfuse, so categorical or boolean metrics are
144
+ sent with the right shape:
145
+
146
+ ```ts
147
+ import { defineMetric, EvalOutcome } from "@anvia/core";
148
+
149
+ const judge = defineMetric({
150
+ name: "quality",
151
+ dataType: "CATEGORICAL",
152
+ configId: "quality-config",
153
+ metadata: { source: "judge-llm" },
154
+ evaluate: () => EvalOutcome.pass("good"),
155
+ });
156
+ ```
157
+
158
+ `defineMetric` is a small identity helper that signals intent and
159
+ preserves type inference; plain object literals continue to work.
160
+
161
+ ### Typed scores and overrides
162
+
163
+ `tracing.score()` accepts a `dataType` (`"NUMERIC" | "CATEGORICAL" | "BOOLEAN"`), a `configId` (or its `scoreConfigId` alias), a per-score `environment` override, and a `timestamp` (Date or ISO 8601 string). The adapter validates `value` against the dataType at the boundary.
164
+
165
+ ```ts
166
+ await tracing.score({
167
+ traceId: trace.traceId,
168
+ name: "verdict",
169
+ value: "pass", // string for CATEGORICAL
170
+ dataType: "CATEGORICAL",
171
+ configId: "cfg-1",
172
+ environment: "staging",
173
+ timestamp: new Date(),
174
+ });
175
+ ```
176
+
177
+ The score fetch has a default timeout of 30 s, overrideable via
178
+ `langfuse.create({ timeoutMs: ... })`.
179
+
180
+ ### Batching and retry (high-volume evals)
181
+
182
+ Enable the in-memory score queue by setting `scoreBatchSize` on
183
+ `langfuse.create()`. When enabled, `tracing.score()` enqueues the
184
+ score and returns immediately. The queue flushes when it reaches
185
+ `scoreBatchSize`, on a debounce timer (`scoreFlushIntervalMs`,
186
+ default 250 ms), and on `flushScores()`, `flush()`, or `shutdown()`.
187
+
188
+ ```ts
189
+ const tracing = langfuse.create({
190
+ publicKey,
191
+ secretKey,
192
+ scoreBatchSize: 20, // enable queue; flushes at 20 items
193
+ scoreFlushIntervalMs: 500, // or after 500ms
194
+ scoreMaxRetries: 3, // retry 429 / 5xx with backoff
195
+ });
196
+
197
+ await tracing.score({ traceId, name: "quality", value: 1 });
198
+ await tracing.score({ traceId, name: "latency", value: 0.4 });
199
+ await tracing.flushScores(); // drain the queue
200
+ console.log(tracing.scoreQueueDepth()); // 0
201
+ ```
202
+
203
+ `flush()` and `shutdown()` also drain the score queue. After all
204
+ retries are exhausted, the queue throws a `LangfuseScoreError` whose
205
+ `scores` property contains the failed payloads so you can inspect
206
+ what was lost.
207
+
208
+ ## Event observations & trace handle
209
+
210
+ After a run starts, you can record ad-hoc checkpoints and attach
211
+ extra attributes to the active trace without threading the run
212
+ observer through every function call. The tracing instance exposes
213
+ the most recent trace through `getCurrentTrace()`:
214
+
215
+ ```ts
216
+ import { langfuse } from "@anvia/langfuse";
217
+
218
+ const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
219
+
220
+ await tracing.startRun({
221
+ agentName: "support",
222
+ prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
223
+ history: [],
224
+ maxTurns: 3,
225
+ });
226
+
227
+ const trace = tracing.getCurrentTrace();
228
+
229
+ trace?.addEvent("retrieval.done", { docCount: 4 });
230
+ trace?.addEvent("validation.passed");
231
+ trace?.addAttributes({ quality: "high" });
232
+ ```
233
+
234
+ `addEvent` creates a Langfuse `event` observation under the active
235
+ root and ends it immediately. `addAttributes` updates the root
236
+ observation's metadata. Both calls bubble up to Langfuse via the
237
+ existing OpenTelemetry span processor.
238
+
239
+ If you have the run observer returned by `startRun`, you can also
240
+ use its `event?(...)` hook (added in `@anvia/core`) to record
241
+ checkpoints:
242
+
243
+ ```ts
244
+ const run = await tracing.startRun({
245
+ agentName: "support",
246
+ prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
247
+ history: [],
248
+ maxTurns: 3,
249
+ });
250
+
251
+ await run.event?.({
252
+ name: "retrieval.done",
253
+ attributes: { docCount: 4 },
254
+ });
255
+ ```
256
+
257
+ The trace handle is cleared when the run `end`s or `error`s. If you
258
+ call `addEvent` or `addAttributes` after that, the underlying
259
+ Langfuse SDK will reject the call because the root observation is
260
+ no longer accepting children. We let the error propagate so it is
261
+ visible.
262
+
263
+ `getCurrentTrace()` is per-tracing-instance and last-write-wins:
264
+ when multiple runs start on the same instance, the handle always
265
+ points at the most recent run. For true concurrency, manage your
266
+ own context (e.g. capture the run observer or build your own
267
+ mapping keyed on user/session).
268
+
269
+ ## Datasets & Experiment runs
270
+
271
+ Bridge `@anvia/core/evals` to Langfuse's dataset / experiment-run
272
+ workflow. The dataset client surfaces the four endpoints you need
273
+ to create a dataset, fetch its items, upsert items, and post a
274
+ batched dataset-run-items payload.
275
+
276
+ ```ts
277
+ import {
278
+ createLangfuseDatasetClient,
279
+ runEvalAsExperiment,
280
+ langfuse,
281
+ } from "@anvia/langfuse";
282
+
283
+ const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
284
+ const client = createLangfuseDatasetClient(tracing);
285
+
286
+ await client.createDataset({ name: "support-smoke" });
287
+ await client.upsertItems("support-smoke", [
288
+ { id: "c-1", input: { q: "hi" }, expected: "hello" },
289
+ { id: "c-2", input: { q: "bye" } },
290
+ ]);
291
+
292
+ const dataset = await client.getDataset("support-smoke");
293
+ console.log(dataset.items);
294
+
295
+ await client.runExperiment({
296
+ datasetName: "support-smoke",
297
+ runName: "smoke-2026-06",
298
+ run: (item) => ({
299
+ output: `answer-for-${item.id}`,
300
+ trace: { traceId: `trace-${item.id}` },
301
+ }),
302
+ });
303
+ ```
304
+
305
+ For eval suites, `runEvalAsExperiment` runs the suite and posts a
306
+ dataset run alongside the metric scores:
307
+
308
+ ```ts
309
+ import { runEvalAsExperiment } from "@anvia/langfuse";
310
+
311
+ const { suite, datasetRun } = await runEvalAsExperiment(
312
+ {
313
+ name: "smoke",
314
+ cases: [
315
+ { id: "c-1", input: "a", expected: "A" },
316
+ { id: "c-2", input: "b", expected: "B" },
317
+ ],
318
+ target: async (input) => input.toUpperCase(),
319
+ metrics: [/* ... */],
320
+ reporters: [/* existing reporters still score */],
321
+ },
322
+ {
323
+ tracing,
324
+ datasetName: "smoke-set",
325
+ runName: "smoke-run",
326
+ },
327
+ );
328
+
329
+ console.log(suite.passed, datasetRun.posted);
330
+ ```
331
+
332
+ ### Options
333
+
334
+ - `createLangfuseDatasetClient(tracing, options)`:
335
+ - `publicKey`, `secretKey`, `baseUrl`: optionally override the
336
+ tracing instance's resolved values. Falls back to env vars
337
+ (`LANGFUSE_PUBLIC_KEY`, `LANGFUSE_SECRET_KEY`, `LANGFUSE_BASE_URL`).
338
+ - `pageSize` (default `50`): dataset item pagination.
339
+ - `timeoutMs` (default `30_000`): per-request timeout via
340
+ `AbortSignal.timeout`.
341
+ - `client.runExperiment({ ... })`:
342
+ - `items` (optional): supply items directly to skip the
343
+ `getDataset` round trip.
344
+ - `run`: invoked per item, returns `{ output, trace? }`.
345
+ Per-item failures are caught and surfaced in the result's
346
+ `errors` array; only successful items reach the batched POST.
347
+
348
+ ### Failure handling
349
+
350
+ `runExperiment` continues on per-item errors. The `errors` array
351
+ contains `{ itemId, error }` entries for failed items; the POST
352
+ still goes out with the successful subset. Non-2xx on the
353
+ batched POST throws (and the items that already errored are
354
+ also included in the thrown error's context).
355
+
356
+ ## Prompt management
357
+
358
+ Pull prompts from Langfuse's prompt store and link generations to
359
+ the prompt version. The tracing instance attaches the prompt name
360
+ and version to the root trace and every generation in the run
361
+ when a prompt ref is configured.
362
+
363
+ ```ts
364
+ import { createLangfusePromptClient, langfuse } from "@anvia/langfuse";
365
+
366
+ const tracing = langfuse.create({ publicKey: "pk", secretKey: "sk" });
367
+ const prompts = createLangfusePromptClient(tracing);
368
+
369
+ const prompt = await prompts.getPrompt("support.system");
370
+ console.log(prompt.prompt, prompt.version);
371
+
372
+ await tracing.startRun({
373
+ agentName: "support",
374
+ prompt: { role: "user", content: [{ type: "text", text: "hi" }] },
375
+ history: [],
376
+ maxTurns: 3,
377
+ promptRef: { name: "support.system", version: prompt.version },
378
+ });
379
+ ```
380
+
381
+ `promptRef` is also accepted on `trace.metadata` (keys
382
+ `promptName` and `promptVersion`) for back-compat with users who
383
+ already attach metadata to `withTrace(...)`.
384
+
385
+ ### Prompt client options
386
+
387
+ - `cacheTtlMs` (default `60_000`): in-memory TTL per
388
+ `${name}::${version}::${label}` key.
389
+ - `timeoutMs` (default `30_000`): per-request timeout via
390
+ `AbortSignal.timeout`.
391
+ - `publicKey`, `secretKey`, `baseUrl`: override the tracing
392
+ instance's resolved values, falling back to env vars.
393
+
394
+ ### Per-call options
395
+
396
+ - `getPrompt(name, { version?, label?, cacheTtlMs?, refresh? })`:
397
+ - `version` and `label` filter the upstream request.
398
+ - `cacheTtlMs` overrides the client default for this call.
399
+ - `refresh: true` skips the cache and re-fetches.
400
+
401
+ ### Helpers
402
+
403
+ - `getPromptText(name, options?)`: returns the prompt string;
404
+ throws if the prompt is a chat prompt.
405
+ - `getPromptChat(name, options?)`: returns the chat message
406
+ array; throws if the prompt is a text prompt.
407
+ - `refresh()`: clears the cache.
408
+
409
+ ## PII redaction
410
+
411
+ Mask personally identifiable information in observations before it
412
+ leaves the process. The default pattern set catches emails,
413
+ credit cards (with Luhn validation), phone numbers, IPv4
414
+ addresses, JWTs, and common API key shapes.
415
+
416
+ ```ts
417
+ import { langfuse } from "@anvia/langfuse";
418
+
419
+ const tracing = langfuse.create({
420
+ publicKey: "pk",
421
+ secretKey: "sk",
422
+ redactInputs: true,
423
+ redactOutputs: "deep",
424
+ });
425
+ ```
426
+
427
+ - `redactInputs` redacts text on root inputs, chat history, and tool
428
+ arguments.
429
+ - `redactOutputs` redacts generation output text and tool results.
430
+ - `"deep"` recurses into nested objects and arrays (in addition to
431
+ top-level strings).
432
+
433
+ ### Customizing
434
+
435
+ ```ts
436
+ const tracing = langfuse.create({
437
+ publicKey: "pk",
438
+ secretKey: "sk",
439
+ redactOutputs: true,
440
+ redaction: {
441
+ replacement: "[HIDDEN]",
442
+ patterns: [
443
+ { name: "ssn", regex: /\b\d{3}-\d{2}-\d{4}\b/g },
444
+ ],
445
+ },
446
+ });
447
+ ```
448
+
449
+ `createPiiRedactor(options)` is also exported for ad-hoc use:
450
+
451
+ ```ts
452
+ import { createPiiRedactor } from "@anvia/langfuse";
453
+
454
+ const redactor = createPiiRedactor();
455
+ const safe = redactor.redactString("email alice@example.com today");
456
+ const safeMessages = redactor.redactMessages(messages);
457
+ const safeObject = redactor.redactObject(spanInput);
458
+ ```
459
+
62
460
  ## Exports
63
461
 
64
462
  - `langfuse`
463
+ - `createLangfuseDatasetClient`
65
464
  - `createLangfuseEvalReporter`
465
+ - `createLangfusePromptClient`
466
+ - `runEvalAsExperiment`
467
+ - `LangfuseScoreError`
66
468
  - `LangfuseTracing`
469
+ - `LangfuseTraceHandle`
67
470
  - `LangfuseTracingOptions`
68
471
  - `LangfuseScoreArgs`
472
+ - `LangfuseScoreDataType`
69
473
  - `LangfuseEvalReporterOptions`
474
+ - `LangfuseDatasetClient`
475
+ - `LangfuseDatasetClientOptions`
476
+ - `LangfuseDataset`
477
+ - `LangfuseDatasetItem`
478
+ - `LangfuseRunExperimentOptions`
479
+ - `LangfuseRunExperimentResult`
480
+ - `LangfuseRunItemResult`
481
+ - `LangfuseRunItemError`
482
+ - `LangfusePromptClient`
483
+ - `LangfusePromptClientOptions`
484
+ - `LangfusePromptGetOptions`
485
+ - `LangfusePrompt`
486
+ - `LangfuseChatMessage`
487
+ - `RunEvalAsExperimentOptions`
488
+ - `RunEvalAsExperimentResult`
489
+ - `createPiiRedactor`
490
+ - `DEFAULT_PATTERNS`
491
+ - `PiiRedactor`
492
+ - `RedactorPattern`
493
+ - `LangfuseRedactionOptions`
494
+ - `LangfuseRedactionMode`
70
495
 
71
496
  ## Development
72
497
 
package/dist/index.d.ts CHANGED
@@ -1,36 +1,196 @@
1
- import { EvalReporter } from '@anvia/core/evals';
2
- import { JsonValue } from '@anvia/core/completion';
1
+ import { JsonValue, Message } from '@anvia/core/completion';
3
2
  import { AgentObserver } from '@anvia/core/observability';
3
+ import { EvalReporter, EvalSuiteResult, RunEvalSuiteOptions } from '@anvia/core/evals';
4
4
 
5
+ type LangfuseRedactionMode = boolean | "deep";
6
+ type LangfuseRedactionOptions$1 = {
7
+ patterns?: RedactorPattern$1[];
8
+ replacement?: string;
9
+ };
10
+ type RedactorPattern$1 = {
11
+ name: string;
12
+ regex: RegExp;
13
+ };
5
14
  type LangfuseTracingOptions = {
6
15
  publicKey?: string | undefined;
7
16
  secretKey?: string | undefined;
8
17
  baseUrl?: string | undefined;
9
18
  environment?: string | undefined;
10
19
  release?: string | undefined;
20
+ serviceName?: string | undefined;
21
+ timeoutMs?: number | undefined;
22
+ scoreBatchSize?: number | undefined;
23
+ scoreFlushIntervalMs?: number | undefined;
24
+ scoreMaxRetries?: number | undefined;
25
+ redactInputs?: LangfuseRedactionMode | undefined;
26
+ redactOutputs?: LangfuseRedactionMode | undefined;
27
+ redaction?: LangfuseRedactionOptions$1 | undefined;
11
28
  };
29
+ type LangfuseScoreDataType = "NUMERIC" | "CATEGORICAL" | "BOOLEAN";
12
30
  type LangfuseScoreArgs = {
13
31
  traceId?: string | undefined;
14
32
  observationId?: string | undefined;
15
33
  name: string;
16
- value: number;
34
+ value: number | string;
35
+ dataType?: LangfuseScoreDataType | undefined;
17
36
  comment?: string | undefined;
18
37
  metadata?: Record<string, JsonValue | undefined> | undefined;
38
+ configId?: string | undefined;
39
+ scoreConfigId?: string | undefined;
40
+ environment?: string | undefined;
41
+ timestamp?: Date | string | undefined;
42
+ };
43
+ type LangfuseTraceHandle = {
44
+ readonly traceId: string;
45
+ readonly observationId: string;
46
+ addAttributes(attributes: Record<string, JsonValue | undefined>): void;
47
+ addEvent(name: string, attributes?: Record<string, JsonValue | undefined>): void;
19
48
  };
20
49
  type LangfuseTracing = AgentObserver & {
21
50
  flush(): Promise<void>;
22
51
  shutdown(): Promise<void>;
23
52
  score(args: LangfuseScoreArgs): Promise<void>;
53
+ flushScores(): Promise<void>;
54
+ scoreQueueDepth(): number;
55
+ getCurrentTrace(): LangfuseTraceHandle | undefined;
24
56
  };
25
57
  type LangfuseEvalReporterOptions = {
26
58
  publishInvalid?: boolean | undefined;
27
59
  strict?: boolean | undefined;
60
+ onMissingTrace?: "ignore" | "warn" | "throw" | undefined;
61
+ truncateInputAt?: number | undefined;
62
+ includeMessages?: boolean | undefined;
63
+ };
64
+ type LangfuseDatasetClientOptions = {
65
+ baseUrl?: string | undefined;
66
+ publicKey?: string | undefined;
67
+ secretKey?: string | undefined;
68
+ pageSize?: number | undefined;
69
+ timeoutMs?: number | undefined;
70
+ };
71
+ type LangfuseDatasetItem<Input = unknown, Expected = unknown> = {
72
+ id: string;
73
+ input: Input;
74
+ expected?: Expected | undefined;
75
+ metadata?: Record<string, JsonValue | undefined> | undefined;
76
+ };
77
+ type LangfuseDataset<Input = unknown, Expected = unknown> = {
78
+ name: string;
79
+ description?: string | undefined;
80
+ metadata?: Record<string, JsonValue | undefined> | undefined;
81
+ items: LangfuseDatasetItem<Input, Expected>[];
82
+ };
83
+ type LangfuseRunItemResult<Output = unknown> = {
84
+ output: Output;
85
+ trace?: {
86
+ traceId: string;
87
+ observationId?: string | undefined;
88
+ } | undefined;
89
+ };
90
+ type LangfuseRunItemError = {
91
+ itemId: string;
92
+ error: unknown;
93
+ };
94
+ type LangfuseRunExperimentOptions<Input = unknown, Output = unknown, Expected = unknown> = {
95
+ datasetName: string;
96
+ runName: string;
97
+ description?: string | undefined;
98
+ metadata?: Record<string, JsonValue | undefined> | undefined;
99
+ items?: LangfuseDatasetItem<Input, Expected>[] | undefined;
100
+ run: (item: LangfuseDatasetItem<Input, Expected>) => LangfuseRunItemResult<Output> | Promise<LangfuseRunItemResult<Output>>;
101
+ };
102
+ type LangfuseRunExperimentResult = {
103
+ runName: string;
104
+ datasetName: string;
105
+ posted: number;
106
+ errors: LangfuseRunItemError[];
107
+ };
108
+ type LangfuseDatasetClient = {
109
+ createDataset(dataset: {
110
+ name: string;
111
+ description?: string | undefined;
112
+ metadata?: Record<string, JsonValue | undefined> | undefined;
113
+ }): Promise<LangfuseDataset<unknown, unknown>>;
114
+ getDataset<Input, Expected>(name: string): Promise<LangfuseDataset<Input, Expected>>;
115
+ upsertItems<Input, Expected>(name: string, items: LangfuseDatasetItem<Input, Expected>[]): Promise<void>;
116
+ runExperiment<Input, Output, Expected>(options: LangfuseRunExperimentOptions<Input, Output, Expected>): Promise<LangfuseRunExperimentResult>;
28
117
  };
118
+ type LangfusePromptClientOptions = {
119
+ baseUrl?: string | undefined;
120
+ publicKey?: string | undefined;
121
+ secretKey?: string | undefined;
122
+ cacheTtlMs?: number | undefined;
123
+ timeoutMs?: number | undefined;
124
+ };
125
+ type LangfusePromptGetOptions = {
126
+ version?: number | undefined;
127
+ label?: string | undefined;
128
+ cacheTtlMs?: number | undefined;
129
+ refresh?: boolean | undefined;
130
+ };
131
+ type LangfuseChatMessage = {
132
+ role: "system" | "user" | "assistant" | "tool";
133
+ content: string;
134
+ };
135
+ type LangfusePrompt = {
136
+ name: string;
137
+ version: number;
138
+ labels: string[];
139
+ prompt: string | LangfuseChatMessage[];
140
+ type: "text" | "chat";
141
+ tags?: string[];
142
+ resolvedAt: Date;
143
+ };
144
+ type LangfusePromptClient = {
145
+ getPrompt(name: string, options?: LangfusePromptGetOptions): Promise<LangfusePrompt>;
146
+ getPromptText(name: string, options?: LangfusePromptGetOptions): Promise<string>;
147
+ getPromptChat(name: string, options?: LangfusePromptGetOptions): Promise<LangfuseChatMessage[]>;
148
+ refresh(): void;
149
+ };
150
+
151
+ declare function createLangfuseDatasetClient(tracing: Pick<LangfuseTracing, "score">, options?: LangfuseDatasetClientOptions): LangfuseDatasetClient;
29
152
 
30
153
  declare function createLangfuseEvalReporter<Input = unknown, Output = unknown, Expected = unknown>(tracing: Pick<LangfuseTracing, "score">, options?: LangfuseEvalReporterOptions): EvalReporter<Input, Output, Expected>;
31
154
 
155
+ type RunEvalAsExperimentOptions<Input, Output, Expected = unknown> = Omit<LangfuseRunExperimentOptions<Input, Output, Expected>, "items" | "run"> & {
156
+ tracing: Pick<LangfuseTracing, "score">;
157
+ client?: LangfuseDatasetClient;
158
+ pageSize?: number | undefined;
159
+ timeoutMs?: number | undefined;
160
+ };
161
+ type RunEvalAsExperimentResult<Input, Output, Expected = unknown> = {
162
+ suite: EvalSuiteResult<Input, Output, Expected>;
163
+ datasetRun: LangfuseRunExperimentResult;
164
+ };
165
+ declare function runEvalAsExperiment<Input, Output, Expected = unknown>(evalOptions: RunEvalSuiteOptions<Input, Output, Expected>, experimentOptions: RunEvalAsExperimentOptions<Input, Output, Expected>): Promise<RunEvalAsExperimentResult<Input, Output, Expected>>;
166
+
167
+ declare function createLangfusePromptClient(tracing: Pick<LangfuseTracing, "score">, options?: LangfusePromptClientOptions): LangfusePromptClient;
168
+
169
+ type RedactorPattern = {
170
+ name: string;
171
+ regex: RegExp;
172
+ };
173
+ type LangfuseRedactionOptions = {
174
+ patterns?: RedactorPattern[];
175
+ replacement?: string;
176
+ };
177
+ type PiiRedactor = {
178
+ redactString(input: string): string;
179
+ redactObject<T>(input: T): T;
180
+ redactMessages(input: Message[]): Message[];
181
+ patternNames(): string[];
182
+ };
183
+ declare function createPiiRedactor(options?: LangfuseRedactionOptions): PiiRedactor;
184
+ declare const DEFAULT_PATTERNS: RedactorPattern[];
185
+
186
+ declare class LangfuseScoreError extends Error {
187
+ readonly scores: LangfuseScoreArgs[];
188
+ readonly cause?: unknown;
189
+ constructor(message: string, scores: LangfuseScoreArgs[], cause?: unknown);
190
+ }
191
+
32
192
  declare const langfuse: {
33
193
  create(options?: LangfuseTracingOptions): LangfuseTracing;
34
194
  };
35
195
 
36
- export { type LangfuseEvalReporterOptions, type LangfuseScoreArgs, type LangfuseTracing, type LangfuseTracingOptions, createLangfuseEvalReporter, langfuse };
196
+ export { DEFAULT_PATTERNS, type LangfuseChatMessage, type LangfuseDataset, type LangfuseDatasetClient, type LangfuseDatasetClientOptions, type LangfuseDatasetItem, type LangfuseEvalReporterOptions, type LangfusePrompt, type LangfusePromptClient, type LangfusePromptClientOptions, type LangfusePromptGetOptions, type LangfuseRedactionMode, type LangfuseRedactionOptions, type LangfuseRunExperimentOptions, type LangfuseRunExperimentResult, type LangfuseRunItemError, type LangfuseRunItemResult, type LangfuseScoreArgs, type LangfuseScoreDataType, LangfuseScoreError, type LangfuseTraceHandle, type LangfuseTracing, type LangfuseTracingOptions, type PiiRedactor, type RedactorPattern, type RunEvalAsExperimentOptions, type RunEvalAsExperimentResult, createLangfuseDatasetClient, createLangfuseEvalReporter, createLangfusePromptClient, createPiiRedactor, langfuse, runEvalAsExperiment };