@huggingface/tasks 0.21.2 → 0.21.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commonjs/eval.d.ts +5 -0
- package/dist/commonjs/eval.d.ts.map +1 -1
- package/dist/commonjs/eval.js +5 -0
- package/dist/esm/eval.d.ts +5 -0
- package/dist/esm/eval.d.ts.map +1 -1
- package/dist/esm/eval.js +5 -0
- package/package.json +1 -1
- package/src/eval.ts +6 -0
package/dist/commonjs/eval.d.ts
CHANGED
|
@@ -107,5 +107,10 @@ export declare const EVALUATION_FRAMEWORKS: {
|
|
|
107
107
|
readonly description: "WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.";
|
|
108
108
|
readonly url: "https://github.com/InternLM/WildClawBench";
|
|
109
109
|
};
|
|
110
|
+
readonly wbench: {
|
|
111
|
+
readonly name: "wbench";
|
|
112
|
+
readonly description: "WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.";
|
|
113
|
+
readonly url: "https://github.com/meituan-longcat/WBench";
|
|
114
|
+
};
|
|
110
115
|
};
|
|
111
116
|
//# sourceMappingURL=eval.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval.d.ts","sourceRoot":"","sources":["../../src/eval.ts"],"names":[],"mappings":"AAAA;;GAEG;AACH,eAAO,MAAM,qBAAqB
|
|
1
|
+
{"version":3,"file":"eval.d.ts","sourceRoot":"","sources":["../../src/eval.ts"],"names":[],"mappings":"AAAA;;GAEG;AACH,eAAO,MAAM,qBAAqB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA4HxB,CAAC"}
|
package/dist/commonjs/eval.js
CHANGED
|
@@ -110,4 +110,9 @@ exports.EVALUATION_FRAMEWORKS = {
|
|
|
110
110
|
description: "WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.",
|
|
111
111
|
url: "https://github.com/InternLM/WildClawBench",
|
|
112
112
|
},
|
|
113
|
+
wbench: {
|
|
114
|
+
name: "wbench",
|
|
115
|
+
description: "WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.",
|
|
116
|
+
url: "https://github.com/meituan-longcat/WBench",
|
|
117
|
+
},
|
|
113
118
|
};
|
package/dist/esm/eval.d.ts
CHANGED
|
@@ -107,5 +107,10 @@ export declare const EVALUATION_FRAMEWORKS: {
|
|
|
107
107
|
readonly description: "WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.";
|
|
108
108
|
readonly url: "https://github.com/InternLM/WildClawBench";
|
|
109
109
|
};
|
|
110
|
+
readonly wbench: {
|
|
111
|
+
readonly name: "wbench";
|
|
112
|
+
readonly description: "WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.";
|
|
113
|
+
readonly url: "https://github.com/meituan-longcat/WBench";
|
|
114
|
+
};
|
|
110
115
|
};
|
|
111
116
|
//# sourceMappingURL=eval.d.ts.map
|
package/dist/esm/eval.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval.d.ts","sourceRoot":"","sources":["../../src/eval.ts"],"names":[],"mappings":"AAAA;;GAEG;AACH,eAAO,MAAM,qBAAqB
|
|
1
|
+
{"version":3,"file":"eval.d.ts","sourceRoot":"","sources":["../../src/eval.ts"],"names":[],"mappings":"AAAA;;GAEG;AACH,eAAO,MAAM,qBAAqB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA4HxB,CAAC"}
|
package/dist/esm/eval.js
CHANGED
|
@@ -107,4 +107,9 @@ export const EVALUATION_FRAMEWORKS = {
|
|
|
107
107
|
description: "WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.",
|
|
108
108
|
url: "https://github.com/InternLM/WildClawBench",
|
|
109
109
|
},
|
|
110
|
+
wbench: {
|
|
111
|
+
name: "wbench",
|
|
112
|
+
description: "WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.",
|
|
113
|
+
url: "https://github.com/meituan-longcat/WBench",
|
|
114
|
+
},
|
|
110
115
|
};
|
package/package.json
CHANGED
package/src/eval.ts
CHANGED
|
@@ -119,4 +119,10 @@ export const EVALUATION_FRAMEWORKS = {
|
|
|
119
119
|
"WildClawBench is an in-the-wild benchmark for evaluating AI agents in the OpenClaw environment across 60 hand-built, end-to-end tasks spanning productivity, code intelligence, social interaction, search, creative synthesis, and safety domains.",
|
|
120
120
|
url: "https://github.com/InternLM/WildClawBench",
|
|
121
121
|
},
|
|
122
|
+
wbench: {
|
|
123
|
+
name: "wbench",
|
|
124
|
+
description:
|
|
125
|
+
"WBench is a comprehensive multi-turn benchmark for interactive video world model evaluation, assessing models across 5 dimensions (video quality, setting adherence, interaction adherence, consistency, physics compliance) and 22 metrics over 289 multi-turn interaction cases.",
|
|
126
|
+
url: "https://github.com/meituan-longcat/WBench",
|
|
127
|
+
},
|
|
122
128
|
} as const;
|