@langfuse/core 5.4.1 → 5.5.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +302 -14
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +900 -335
- package/dist/index.d.ts +900 -335
- package/dist/index.mjs +301 -14
- package/dist/index.mjs.map +1 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -105,6 +105,7 @@ __export(index_exports, {
|
|
|
105
105
|
scim: () => scim_exports,
|
|
106
106
|
scoreConfigs: () => scoreConfigs_exports,
|
|
107
107
|
scores: () => scores_exports,
|
|
108
|
+
scoresV3: () => scoresV3_exports,
|
|
108
109
|
serializeValue: () => serializeValue,
|
|
109
110
|
sessions: () => sessions_exports,
|
|
110
111
|
setLangfuseTraceIdInBaggage: () => setLangfuseTraceIdInBaggage,
|
|
@@ -386,7 +387,7 @@ var resetGlobalLogger = () => {
|
|
|
386
387
|
// package.json
|
|
387
388
|
var package_default = {
|
|
388
389
|
name: "@langfuse/core",
|
|
389
|
-
version: "5.
|
|
390
|
+
version: "5.5.0-beta.0",
|
|
390
391
|
description: "Core functions and utilities for Langfuse packages",
|
|
391
392
|
type: "module",
|
|
392
393
|
sideEffects: false,
|
|
@@ -988,6 +989,9 @@ var scim_exports = {};
|
|
|
988
989
|
// src/api/api/resources/scoreConfigs/index.ts
|
|
989
990
|
var scoreConfigs_exports = {};
|
|
990
991
|
|
|
992
|
+
// src/api/api/resources/scoresV3/index.ts
|
|
993
|
+
var scoresV3_exports = {};
|
|
994
|
+
|
|
991
995
|
// src/api/api/resources/scores/index.ts
|
|
992
996
|
var scores_exports = {};
|
|
993
997
|
|
|
@@ -1002,6 +1006,7 @@ var unstable_exports = {};
|
|
|
1002
1006
|
__export(unstable_exports, {
|
|
1003
1007
|
AccessDeniedError: () => AccessDeniedError2,
|
|
1004
1008
|
BadRequestError: () => BadRequestError,
|
|
1009
|
+
CodeEvaluatorSourceCodeLanguage: () => CodeEvaluatorSourceCodeLanguage,
|
|
1005
1010
|
ConflictError: () => ConflictError,
|
|
1006
1011
|
EvaluationRuleArrayOptionsFilterOperator: () => EvaluationRuleArrayOptionsFilterOperator,
|
|
1007
1012
|
EvaluationRuleBooleanFilterOperator: () => EvaluationRuleBooleanFilterOperator,
|
|
@@ -1016,6 +1021,7 @@ __export(unstable_exports, {
|
|
|
1016
1021
|
EvaluatorScope: () => EvaluatorScope,
|
|
1017
1022
|
EvaluatorType: () => EvaluatorType,
|
|
1018
1023
|
InternalServerError: () => InternalServerError,
|
|
1024
|
+
LlmAsJudgeEvaluatorType: () => LlmAsJudgeEvaluatorType,
|
|
1019
1025
|
MethodNotAllowedError: () => MethodNotAllowedError2,
|
|
1020
1026
|
NotFoundError: () => NotFoundError2,
|
|
1021
1027
|
PublicApiErrorCode: () => PublicApiErrorCode,
|
|
@@ -1031,6 +1037,7 @@ __export(unstable_exports, {
|
|
|
1031
1037
|
// src/api/api/resources/unstable/resources/commons/index.ts
|
|
1032
1038
|
var commons_exports2 = {};
|
|
1033
1039
|
__export(commons_exports2, {
|
|
1040
|
+
CodeEvaluatorSourceCodeLanguage: () => CodeEvaluatorSourceCodeLanguage,
|
|
1034
1041
|
EvaluationRuleArrayOptionsFilterOperator: () => EvaluationRuleArrayOptionsFilterOperator,
|
|
1035
1042
|
EvaluationRuleBooleanFilterOperator: () => EvaluationRuleBooleanFilterOperator,
|
|
1036
1043
|
EvaluationRuleMappingSource: () => EvaluationRuleMappingSource,
|
|
@@ -1047,7 +1054,14 @@ __export(commons_exports2, {
|
|
|
1047
1054
|
|
|
1048
1055
|
// src/api/api/resources/unstable/resources/commons/types/EvaluatorType.ts
|
|
1049
1056
|
var EvaluatorType = {
|
|
1050
|
-
LlmAsJudge: "llm_as_judge"
|
|
1057
|
+
LlmAsJudge: "llm_as_judge",
|
|
1058
|
+
Code: "code"
|
|
1059
|
+
};
|
|
1060
|
+
|
|
1061
|
+
// src/api/api/resources/unstable/resources/commons/types/CodeEvaluatorSourceCodeLanguage.ts
|
|
1062
|
+
var CodeEvaluatorSourceCodeLanguage = {
|
|
1063
|
+
Python: "PYTHON",
|
|
1064
|
+
Typescript: "TYPESCRIPT"
|
|
1051
1065
|
};
|
|
1052
1066
|
|
|
1053
1067
|
// src/api/api/resources/unstable/resources/commons/types/EvaluatorScope.ts
|
|
@@ -1284,6 +1298,14 @@ var InternalServerError = class _InternalServerError extends LangfuseAPIError {
|
|
|
1284
1298
|
|
|
1285
1299
|
// src/api/api/resources/unstable/resources/evaluationRules/index.ts
|
|
1286
1300
|
var evaluationRules_exports = {};
|
|
1301
|
+
__export(evaluationRules_exports, {
|
|
1302
|
+
LlmAsJudgeEvaluatorType: () => LlmAsJudgeEvaluatorType
|
|
1303
|
+
});
|
|
1304
|
+
|
|
1305
|
+
// src/api/api/resources/unstable/resources/evaluationRules/types/LlmAsJudgeEvaluatorType.ts
|
|
1306
|
+
var LlmAsJudgeEvaluatorType = {
|
|
1307
|
+
LlmAsJudge: "llm_as_judge"
|
|
1308
|
+
};
|
|
1287
1309
|
|
|
1288
1310
|
// src/api/api/resources/unstable/resources/evaluators/index.ts
|
|
1289
1311
|
var evaluators_exports = {};
|
|
@@ -11527,6 +11549,220 @@ var ScoreConfigs = class {
|
|
|
11527
11549
|
}
|
|
11528
11550
|
};
|
|
11529
11551
|
|
|
11552
|
+
// src/api/api/resources/scoresV3/client/Client.ts
|
|
11553
|
+
var ScoresV3 = class {
|
|
11554
|
+
constructor(_options) {
|
|
11555
|
+
this._options = _options;
|
|
11556
|
+
}
|
|
11557
|
+
/**
|
|
11558
|
+
* Get a list of scores with a polymorphic `value` field (v3).
|
|
11559
|
+
*
|
|
11560
|
+
* This endpoint requires Langfuse v4 or later.
|
|
11561
|
+
*
|
|
11562
|
+
* The `value` field type depends on `dataType`:
|
|
11563
|
+
* - `NUMERIC` → number
|
|
11564
|
+
* - `BOOLEAN` → boolean
|
|
11565
|
+
* - `CATEGORICAL`, `TEXT`, `CORRECTION` → string
|
|
11566
|
+
*
|
|
11567
|
+
* Use the `fields` parameter to include optional field groups beyond the
|
|
11568
|
+
* default `core`. Unknown group names return HTTP 400.
|
|
11569
|
+
*
|
|
11570
|
+
* @param {LangfuseAPI.GetScoresV3Request} request
|
|
11571
|
+
* @param {ScoresV3.RequestOptions} requestOptions - Request-specific configuration.
|
|
11572
|
+
*
|
|
11573
|
+
* @throws {@link LangfuseAPI.Error}
|
|
11574
|
+
* @throws {@link LangfuseAPI.UnauthorizedError}
|
|
11575
|
+
* @throws {@link LangfuseAPI.AccessDeniedError}
|
|
11576
|
+
* @throws {@link LangfuseAPI.MethodNotAllowedError}
|
|
11577
|
+
* @throws {@link LangfuseAPI.NotFoundError}
|
|
11578
|
+
*
|
|
11579
|
+
* @example
|
|
11580
|
+
* await client.scoresV3.getManyV3()
|
|
11581
|
+
*/
|
|
11582
|
+
getManyV3(request = {}, requestOptions) {
|
|
11583
|
+
return HttpResponsePromise.fromPromise(
|
|
11584
|
+
this.__getManyV3(request, requestOptions)
|
|
11585
|
+
);
|
|
11586
|
+
}
|
|
11587
|
+
async __getManyV3(request = {}, requestOptions) {
|
|
11588
|
+
var _a2, _b, _c, _d, _e, _f, _g, _h;
|
|
11589
|
+
const {
|
|
11590
|
+
limit,
|
|
11591
|
+
cursor,
|
|
11592
|
+
fields,
|
|
11593
|
+
id,
|
|
11594
|
+
name,
|
|
11595
|
+
source,
|
|
11596
|
+
dataType,
|
|
11597
|
+
environment,
|
|
11598
|
+
configId,
|
|
11599
|
+
queueId,
|
|
11600
|
+
authorUserId,
|
|
11601
|
+
value,
|
|
11602
|
+
valueMin,
|
|
11603
|
+
valueMax,
|
|
11604
|
+
traceId,
|
|
11605
|
+
sessionId,
|
|
11606
|
+
observationId,
|
|
11607
|
+
experimentId,
|
|
11608
|
+
fromTimestamp,
|
|
11609
|
+
toTimestamp
|
|
11610
|
+
} = request;
|
|
11611
|
+
const _queryParams = {};
|
|
11612
|
+
if (limit != null) {
|
|
11613
|
+
_queryParams["limit"] = limit.toString();
|
|
11614
|
+
}
|
|
11615
|
+
if (cursor != null) {
|
|
11616
|
+
_queryParams["cursor"] = cursor;
|
|
11617
|
+
}
|
|
11618
|
+
if (fields != null) {
|
|
11619
|
+
_queryParams["fields"] = fields;
|
|
11620
|
+
}
|
|
11621
|
+
if (id != null) {
|
|
11622
|
+
_queryParams["id"] = id;
|
|
11623
|
+
}
|
|
11624
|
+
if (name != null) {
|
|
11625
|
+
_queryParams["name"] = name;
|
|
11626
|
+
}
|
|
11627
|
+
if (source != null) {
|
|
11628
|
+
_queryParams["source"] = source;
|
|
11629
|
+
}
|
|
11630
|
+
if (dataType != null) {
|
|
11631
|
+
_queryParams["dataType"] = dataType;
|
|
11632
|
+
}
|
|
11633
|
+
if (environment != null) {
|
|
11634
|
+
_queryParams["environment"] = environment;
|
|
11635
|
+
}
|
|
11636
|
+
if (configId != null) {
|
|
11637
|
+
_queryParams["configId"] = configId;
|
|
11638
|
+
}
|
|
11639
|
+
if (queueId != null) {
|
|
11640
|
+
_queryParams["queueId"] = queueId;
|
|
11641
|
+
}
|
|
11642
|
+
if (authorUserId != null) {
|
|
11643
|
+
_queryParams["authorUserId"] = authorUserId;
|
|
11644
|
+
}
|
|
11645
|
+
if (value != null) {
|
|
11646
|
+
_queryParams["value"] = value;
|
|
11647
|
+
}
|
|
11648
|
+
if (valueMin != null) {
|
|
11649
|
+
_queryParams["valueMin"] = valueMin.toString();
|
|
11650
|
+
}
|
|
11651
|
+
if (valueMax != null) {
|
|
11652
|
+
_queryParams["valueMax"] = valueMax.toString();
|
|
11653
|
+
}
|
|
11654
|
+
if (traceId != null) {
|
|
11655
|
+
_queryParams["traceId"] = traceId;
|
|
11656
|
+
}
|
|
11657
|
+
if (sessionId != null) {
|
|
11658
|
+
_queryParams["sessionId"] = sessionId;
|
|
11659
|
+
}
|
|
11660
|
+
if (observationId != null) {
|
|
11661
|
+
_queryParams["observationId"] = observationId;
|
|
11662
|
+
}
|
|
11663
|
+
if (experimentId != null) {
|
|
11664
|
+
_queryParams["experimentId"] = experimentId;
|
|
11665
|
+
}
|
|
11666
|
+
if (fromTimestamp != null) {
|
|
11667
|
+
_queryParams["fromTimestamp"] = fromTimestamp;
|
|
11668
|
+
}
|
|
11669
|
+
if (toTimestamp != null) {
|
|
11670
|
+
_queryParams["toTimestamp"] = toTimestamp;
|
|
11671
|
+
}
|
|
11672
|
+
let _headers = mergeHeaders(
|
|
11673
|
+
(_a2 = this._options) == null ? void 0 : _a2.headers,
|
|
11674
|
+
mergeOnlyDefinedHeaders({
|
|
11675
|
+
Authorization: await this._getAuthorizationHeader(),
|
|
11676
|
+
"X-Langfuse-Sdk-Name": (_c = requestOptions == null ? void 0 : requestOptions.xLangfuseSdkName) != null ? _c : (_b = this._options) == null ? void 0 : _b.xLangfuseSdkName,
|
|
11677
|
+
"X-Langfuse-Sdk-Version": (_e = requestOptions == null ? void 0 : requestOptions.xLangfuseSdkVersion) != null ? _e : (_d = this._options) == null ? void 0 : _d.xLangfuseSdkVersion,
|
|
11678
|
+
"X-Langfuse-Public-Key": (_g = requestOptions == null ? void 0 : requestOptions.xLangfusePublicKey) != null ? _g : (_f = this._options) == null ? void 0 : _f.xLangfusePublicKey
|
|
11679
|
+
}),
|
|
11680
|
+
requestOptions == null ? void 0 : requestOptions.headers
|
|
11681
|
+
);
|
|
11682
|
+
const _response = await fetcher({
|
|
11683
|
+
url: url_exports.join(
|
|
11684
|
+
(_h = await Supplier.get(this._options.baseUrl)) != null ? _h : await Supplier.get(this._options.environment),
|
|
11685
|
+
"/api/public/v3/scores"
|
|
11686
|
+
),
|
|
11687
|
+
method: "GET",
|
|
11688
|
+
headers: _headers,
|
|
11689
|
+
queryParameters: { ..._queryParams, ...requestOptions == null ? void 0 : requestOptions.queryParams },
|
|
11690
|
+
timeoutMs: (requestOptions == null ? void 0 : requestOptions.timeoutInSeconds) != null ? requestOptions.timeoutInSeconds * 1e3 : 6e4,
|
|
11691
|
+
maxRetries: requestOptions == null ? void 0 : requestOptions.maxRetries,
|
|
11692
|
+
abortSignal: requestOptions == null ? void 0 : requestOptions.abortSignal
|
|
11693
|
+
});
|
|
11694
|
+
if (_response.ok) {
|
|
11695
|
+
return {
|
|
11696
|
+
data: _response.body,
|
|
11697
|
+
rawResponse: _response.rawResponse
|
|
11698
|
+
};
|
|
11699
|
+
}
|
|
11700
|
+
if (_response.error.reason === "status-code") {
|
|
11701
|
+
switch (_response.error.statusCode) {
|
|
11702
|
+
case 400:
|
|
11703
|
+
throw new Error2(
|
|
11704
|
+
_response.error.body,
|
|
11705
|
+
_response.rawResponse
|
|
11706
|
+
);
|
|
11707
|
+
case 401:
|
|
11708
|
+
throw new UnauthorizedError(
|
|
11709
|
+
_response.error.body,
|
|
11710
|
+
_response.rawResponse
|
|
11711
|
+
);
|
|
11712
|
+
case 403:
|
|
11713
|
+
throw new AccessDeniedError(
|
|
11714
|
+
_response.error.body,
|
|
11715
|
+
_response.rawResponse
|
|
11716
|
+
);
|
|
11717
|
+
case 405:
|
|
11718
|
+
throw new MethodNotAllowedError(
|
|
11719
|
+
_response.error.body,
|
|
11720
|
+
_response.rawResponse
|
|
11721
|
+
);
|
|
11722
|
+
case 404:
|
|
11723
|
+
throw new NotFoundError(
|
|
11724
|
+
_response.error.body,
|
|
11725
|
+
_response.rawResponse
|
|
11726
|
+
);
|
|
11727
|
+
default:
|
|
11728
|
+
throw new LangfuseAPIError({
|
|
11729
|
+
statusCode: _response.error.statusCode,
|
|
11730
|
+
body: _response.error.body,
|
|
11731
|
+
rawResponse: _response.rawResponse
|
|
11732
|
+
});
|
|
11733
|
+
}
|
|
11734
|
+
}
|
|
11735
|
+
switch (_response.error.reason) {
|
|
11736
|
+
case "non-json":
|
|
11737
|
+
throw new LangfuseAPIError({
|
|
11738
|
+
statusCode: _response.error.statusCode,
|
|
11739
|
+
body: _response.error.rawBody,
|
|
11740
|
+
rawResponse: _response.rawResponse
|
|
11741
|
+
});
|
|
11742
|
+
case "timeout":
|
|
11743
|
+
throw new LangfuseAPITimeoutError(
|
|
11744
|
+
"Timeout exceeded when calling GET /api/public/v3/scores."
|
|
11745
|
+
);
|
|
11746
|
+
case "unknown":
|
|
11747
|
+
throw new LangfuseAPIError({
|
|
11748
|
+
message: _response.error.errorMessage,
|
|
11749
|
+
rawResponse: _response.rawResponse
|
|
11750
|
+
});
|
|
11751
|
+
}
|
|
11752
|
+
}
|
|
11753
|
+
async _getAuthorizationHeader() {
|
|
11754
|
+
const username = await Supplier.get(this._options.username);
|
|
11755
|
+
const password = await Supplier.get(this._options.password);
|
|
11756
|
+
if (username != null && password != null) {
|
|
11757
|
+
return BasicAuth.toAuthorizationHeader({
|
|
11758
|
+
username,
|
|
11759
|
+
password
|
|
11760
|
+
});
|
|
11761
|
+
}
|
|
11762
|
+
return void 0;
|
|
11763
|
+
}
|
|
11764
|
+
};
|
|
11765
|
+
|
|
11530
11766
|
// src/api/api/resources/scores/client/Client.ts
|
|
11531
11767
|
var Scores = class {
|
|
11532
11768
|
constructor(_options) {
|
|
@@ -12617,8 +12853,9 @@ var EvaluationRules = class {
|
|
|
12617
12853
|
* - `evaluator.name` + `evaluator.scope` must identify an existing evaluator family returned by the evaluator endpoints
|
|
12618
12854
|
* - Langfuse resolves that family to its latest version before saving the evaluation rule
|
|
12619
12855
|
* - for `target=experiment`, use dataset `id` values from `GET /api/public/v2/datasets` when filtering by `datasetId`
|
|
12620
|
-
* - every evaluator prompt variable must be mapped exactly once
|
|
12621
|
-
* - `
|
|
12856
|
+
* - for `llm_as_judge` evaluators, every evaluator prompt variable must be mapped exactly once
|
|
12857
|
+
* - for `code` evaluators, Langfuse uses the fixed code runtime mapping; omit `mapping` in create and update requests
|
|
12858
|
+
* - for user-provided `llm_as_judge` mappings, `expected_output` and `experiment_item_metadata` are only valid for `target=experiment`
|
|
12622
12859
|
* - if `enabled=true`, Langfuse validates that the referenced evaluator can currently run
|
|
12623
12860
|
* - at most 50 evaluation rules can be effectively active in one project at the same time
|
|
12624
12861
|
*
|
|
@@ -12635,9 +12872,9 @@ var EvaluationRules = class {
|
|
|
12635
12872
|
* Recovery guidance:
|
|
12636
12873
|
* - `400 invalid_filter_value`: fix the filter `column` or `value` using `details.column`, `details.invalidValues`, and `details.allowedValues`
|
|
12637
12874
|
* - `400 invalid_filter_value` with `details.column=datasetId`: call `GET /api/public/v2/datasets`, then retry with dataset `id` values from that response
|
|
12638
|
-
* - `400 missing_variable_mapping`: fetch the evaluator again and make sure every variable in `variables` appears exactly once in `mapping`
|
|
12875
|
+
* - `400 missing_variable_mapping`: for `llm_as_judge` evaluators, fetch the evaluator again and make sure every variable in `variables` appears exactly once in `mapping`
|
|
12639
12876
|
* - `400 duplicate_variable_mapping`: remove repeated mappings for the same variable
|
|
12640
|
-
* - `400 invalid_variable_mapping`: switch to a valid `source` for the selected `target`, or fix the variable name
|
|
12877
|
+
* - `400 invalid_variable_mapping`: for `llm_as_judge`, switch to a valid `source` for the selected `target`, or fix the variable name
|
|
12641
12878
|
* - `400 invalid_json_path`: remove or correct the `jsonPath`
|
|
12642
12879
|
* - `422 evaluator_preflight_failed`: the selected evaluator cannot run with the resolved model configuration. Fix the evaluator/default model setup, then retry the create request.
|
|
12643
12880
|
*
|
|
@@ -12664,7 +12901,8 @@ var EvaluationRules = class {
|
|
|
12664
12901
|
* name: "answer-correctness-live",
|
|
12665
12902
|
* evaluator: {
|
|
12666
12903
|
* name: "answer-correctness",
|
|
12667
|
-
* scope: "project"
|
|
12904
|
+
* scope: "project",
|
|
12905
|
+
* type: "llm_as_judge"
|
|
12668
12906
|
* },
|
|
12669
12907
|
* target: "observation",
|
|
12670
12908
|
* enabled: true,
|
|
@@ -12686,10 +12924,30 @@ var EvaluationRules = class {
|
|
|
12686
12924
|
*
|
|
12687
12925
|
* @example
|
|
12688
12926
|
* await client.unstable.evaluationRules.create({
|
|
12927
|
+
* name: "toxicity-code-live",
|
|
12928
|
+
* evaluator: {
|
|
12929
|
+
* name: "toxicity-detector",
|
|
12930
|
+
* scope: "project",
|
|
12931
|
+
* type: "code"
|
|
12932
|
+
* },
|
|
12933
|
+
* target: "observation",
|
|
12934
|
+
* enabled: true,
|
|
12935
|
+
* sampling: 1,
|
|
12936
|
+
* filter: [{
|
|
12937
|
+
* type: "stringOptions",
|
|
12938
|
+
* column: "type",
|
|
12939
|
+
* operator: "any of",
|
|
12940
|
+
* value: ["GENERATION"]
|
|
12941
|
+
* }]
|
|
12942
|
+
* })
|
|
12943
|
+
*
|
|
12944
|
+
* @example
|
|
12945
|
+
* await client.unstable.evaluationRules.create({
|
|
12689
12946
|
* name: "experiment-expected-output-match",
|
|
12690
12947
|
* evaluator: {
|
|
12691
12948
|
* name: "expected-output-match",
|
|
12692
|
-
* scope: "project"
|
|
12949
|
+
* scope: "project",
|
|
12950
|
+
* type: "llm_as_judge"
|
|
12693
12951
|
* },
|
|
12694
12952
|
* target: "experiment",
|
|
12695
12953
|
* enabled: true,
|
|
@@ -13149,18 +13407,19 @@ var EvaluationRules = class {
|
|
|
13149
13407
|
* - switch to another evaluator
|
|
13150
13408
|
* - adjust sampling
|
|
13151
13409
|
* - change filters
|
|
13152
|
-
* - update variable mappings
|
|
13410
|
+
* - update LLM-as-judge variable mappings
|
|
13153
13411
|
*
|
|
13154
13412
|
* Important behavior:
|
|
13155
13413
|
* - provide only the fields you want to change
|
|
13156
13414
|
* - if you provide `evaluator`, Langfuse resolves that evaluator family to its latest version before saving
|
|
13157
|
-
* - changing `target`, `filter`, or `mapping` must still produce a valid target-specific configuration
|
|
13158
|
-
* - if you change `target
|
|
13415
|
+
* - changing `target`, `filter`, or an LLM-as-judge `mapping` must still produce a valid target-specific configuration
|
|
13416
|
+
* - if you change `target` for an LLM-as-judge rule, also send a compatible `filter` and `mapping` in the same request unless the existing ones are still valid for the new target
|
|
13417
|
+
* - for `code` evaluator rules, omit `mapping`; Langfuse stores the fixed code runtime mapping automatically
|
|
13159
13418
|
* - if the resulting config is enabled, Langfuse re-validates that the selected evaluator can run
|
|
13160
13419
|
* - if the update would move a non-active evaluation rule into the active state and the project already has 50 active evaluation rules, the API returns `409`
|
|
13161
13420
|
*
|
|
13162
13421
|
* Recovery guidance:
|
|
13163
|
-
* - if
|
|
13422
|
+
* - if an LLM-as-judge update fails with `missing_variable_mapping` or `invalid_variable_mapping` after changing `evaluator` or `target`, resend the request with a complete new `mapping`
|
|
13164
13423
|
* - if the update fails with `invalid_filter_value` after changing `target`, resend the request with a target-compatible `filter`
|
|
13165
13424
|
*
|
|
13166
13425
|
* @param {string} evaluationRuleId - Evaluation rule identifier.
|
|
@@ -13491,7 +13750,9 @@ var Evaluators = class {
|
|
|
13491
13750
|
/**
|
|
13492
13751
|
* Create an evaluator in the authenticated project.
|
|
13493
13752
|
*
|
|
13494
|
-
* Use evaluators to define **how** Langfuse should score data
|
|
13753
|
+
* Use evaluators to define **how** Langfuse should score data.
|
|
13754
|
+
* LLM-as-a-judge evaluators define a prompt, expected structured output, and optional model configuration.
|
|
13755
|
+
* Code evaluators define source code and a runtime language.
|
|
13495
13756
|
*
|
|
13496
13757
|
* Naming behavior:
|
|
13497
13758
|
* - If this is a new evaluator name in your project, Langfuse creates version `1`.
|
|
@@ -13504,10 +13765,15 @@ var Evaluators = class {
|
|
|
13504
13765
|
* 3. Read the returned `outputDefinition.dataType` so the client knows whether future scores will be numeric, boolean, or categorical.
|
|
13505
13766
|
* 4. Create one or more evaluation rules that reference the returned evaluator family using `name` and `scope`.
|
|
13506
13767
|
*
|
|
13768
|
+
* Code evaluator validation:
|
|
13769
|
+
* - At creation, Langfuse only validates the request shape
|
|
13770
|
+
* - The `sourceCode` itself is not executed here. It is first run (preflight-tested against a sample observation) when you link the evaluator to an evaluation rule, so runtime errors in the code surface at evaluation-rule creation, not at evaluator creation.
|
|
13771
|
+
*
|
|
13507
13772
|
* Recovery guidance:
|
|
13508
13773
|
* - `422` with `code=evaluator_preflight_failed`: the evaluator cannot run with the resolved model configuration. Add a valid explicit `modelConfig`, or configure the project's default evaluation model, then retry the same request.
|
|
13509
13774
|
* - `400` with `code=invalid_body`: the request shape is malformed. Use the structured `details.issues` array to fix the specific fields and retry.
|
|
13510
|
-
* - `400` with `code=invalid_body` on `outputDefinition`: send `dataType`, `reasoning.description`, and `score.description`. Do not send `version`; it is not part of the public request shape.
|
|
13775
|
+
* - `400` with `code=invalid_body` on `outputDefinition`: for `type=llm_as_judge`, send `dataType`, `reasoning.description`, and `score.description`. Do not send `version`; it is not part of the public request shape.
|
|
13776
|
+
* - If `type` is omitted, Langfuse treats the request as `type=llm_as_judge` for backwards compatibility. New clients should send `type` explicitly.
|
|
13511
13777
|
*
|
|
13512
13778
|
* Unstable API note:
|
|
13513
13779
|
* - This surface may evolve while the underlying evaluation data model is being redesigned.
|
|
@@ -13531,6 +13797,7 @@ var Evaluators = class {
|
|
|
13531
13797
|
*
|
|
13532
13798
|
* @example
|
|
13533
13799
|
* await client.unstable.evaluators.create({
|
|
13800
|
+
* type: "llm_as_judge",
|
|
13534
13801
|
* name: "answer-correctness",
|
|
13535
13802
|
* prompt: "You are grading an answer.\n\nInput:\n{{input}}\n\nOutput:\n{{output}}\n\nReturn a score between 0 and 1.\n",
|
|
13536
13803
|
* outputDefinition: {
|
|
@@ -13548,6 +13815,22 @@ var Evaluators = class {
|
|
|
13548
13815
|
* model: "gpt-4.1-mini"
|
|
13549
13816
|
* }
|
|
13550
13817
|
* })
|
|
13818
|
+
*
|
|
13819
|
+
* @example
|
|
13820
|
+
* await client.unstable.evaluators.create({
|
|
13821
|
+
* type: "code",
|
|
13822
|
+
* name: "exact-match",
|
|
13823
|
+
* sourceCode: "function evaluate(ctx: EvaluationContext): EvaluationResult {\n const input = ctx.observation.input;\n const matchesOutput =\n input !== undefined && ctx.observation.output === input;\n\n return {\n scores: [\n {\n name: \"Exact match\",\n value: matchesOutput,\n dataType: \"BOOLEAN\",\n comment: matchesOutput\n ? \"Output exactly matches the input.\"\n : \"Output does not match the input.\",\n },\n ],\n };\n}\n",
|
|
13824
|
+
* sourceCodeLanguage: "TYPESCRIPT"
|
|
13825
|
+
* })
|
|
13826
|
+
*
|
|
13827
|
+
* @example
|
|
13828
|
+
* await client.unstable.evaluators.create({
|
|
13829
|
+
* type: "code",
|
|
13830
|
+
* name: "exact-match",
|
|
13831
|
+
* sourceCode: "def evaluate(ctx: EvaluationContext) -> EvaluationResult:\n \"\"\"Evaluates one observation and returns one or more Langfuse scores.\"\"\"\n input = ctx.observation.input\n matches_output = input is not None and ctx.observation.output == input\n\n return EvaluationResult(\n scores=[\n Score(\n name=\"Exact match\",\n value=matches_output,\n data_type=\"BOOLEAN\",\n comment=(\n \"Output exactly matches the input.\"\n if matches_output\n else \"Output does not match the input.\"\n ),\n )\n ]\n )\n",
|
|
13832
|
+
* sourceCodeLanguage: "PYTHON"
|
|
13833
|
+
* })
|
|
13551
13834
|
*/
|
|
13552
13835
|
create(request, requestOptions) {
|
|
13553
13836
|
return HttpResponsePromise.fromPromise(
|
|
@@ -14108,6 +14391,10 @@ var LangfuseAPIClient = class {
|
|
|
14108
14391
|
var _a2;
|
|
14109
14392
|
return (_a2 = this._scoreConfigs) != null ? _a2 : this._scoreConfigs = new ScoreConfigs(this._options);
|
|
14110
14393
|
}
|
|
14394
|
+
get scoresV3() {
|
|
14395
|
+
var _a2;
|
|
14396
|
+
return (_a2 = this._scoresV3) != null ? _a2 : this._scoresV3 = new ScoresV3(this._options);
|
|
14397
|
+
}
|
|
14111
14398
|
get scores() {
|
|
14112
14399
|
var _a2;
|
|
14113
14400
|
return (_a2 = this._scores) != null ? _a2 : this._scores = new Scores(this._options);
|
|
@@ -14789,6 +15076,7 @@ function getSpanKeyFromBaggageKey(baggageKey) {
|
|
|
14789
15076
|
scim,
|
|
14790
15077
|
scoreConfigs,
|
|
14791
15078
|
scores,
|
|
15079
|
+
scoresV3,
|
|
14792
15080
|
serializeValue,
|
|
14793
15081
|
sessions,
|
|
14794
15082
|
setLangfuseTraceIdInBaggage,
|