oh-my-knowledge 1.0.0-beta.10 → 1.0.0-beta.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/agent-skills/omk/SKILL.md +1 -1
- package/dist/cli/lib/codex-model-hint.js +5 -14
- package/dist/cli/lib/runtime-defaults.js +2 -2
- package/dist/diagnosis/lifecycle.js +2 -2
- package/dist/eval-core/compiler/validation.js +5 -2
- package/dist/eval-runtime/builders/custom-evaluator.d.ts +49 -0
- package/dist/eval-runtime/builders/custom-evaluator.js +67 -0
- package/dist/eval-runtime/builders/rubric-evaluator.d.ts +7 -0
- package/dist/eval-runtime/builders/rubric-evaluator.js +46 -0
- package/dist/eval-runtime/conformance/cache.js +15 -8
- package/dist/eval-runtime/conformance/evaluator.d.ts +3 -3
- package/dist/eval-runtime/conformance/evaluator.js +28 -17
- package/dist/eval-runtime/conformance/judge.js +5 -6
- package/dist/eval-runtime/conformance/runtime.d.ts +1 -1
- package/dist/eval-runtime/conformance/runtime.js +1 -1
- package/dist/eval-runtime/custom-evaluator.d.ts +14 -7
- package/dist/eval-runtime/custom-evaluator.js +172 -96
- package/dist/eval-runtime/debug-evaluator.d.ts +26 -0
- package/dist/eval-runtime/debug-evaluator.js +65 -0
- package/dist/eval-runtime/debug-rubric.d.ts +29 -0
- package/dist/eval-runtime/debug-rubric.js +92 -0
- package/dist/eval-runtime/evaluate.d.ts +1 -1
- package/dist/eval-runtime/evaluation/capture-evaluators.js +26 -116
- package/dist/eval-runtime/evaluation/capture-judge.d.ts +39 -0
- package/dist/eval-runtime/evaluation/capture-judge.js +45 -0
- package/dist/eval-runtime/evaluation/contracts.d.ts +4 -2
- package/dist/eval-runtime/evaluation/errors.d.ts +7 -1
- package/dist/eval-runtime/evaluation/errors.js +5 -1
- package/dist/eval-runtime/evaluation/rubric-declaration.d.ts +17 -0
- package/dist/eval-runtime/evaluation/rubric-declaration.js +118 -0
- package/dist/eval-runtime/evaluators/formula.js +18 -14
- package/dist/eval-runtime/index.d.ts +9 -2
- package/dist/eval-runtime/index.js +3 -0
- package/dist/eval-runtime/judges/rubric-contracts.d.ts +6 -5
- package/dist/eval-runtime/judges/rubric-contracts.js +25 -19
- package/dist/eval-runtime/judges/rubric-judge.d.ts +5 -4
- package/dist/eval-runtime/judges/rubric-judge.js +46 -62
- package/dist/eval-runtime/judges/rubric-kit.d.ts +5 -4
- package/dist/eval-runtime/judges/rubric-kit.js +8 -15
- package/dist/eval-runtime/judges/rubric-prompt.d.ts +2 -1
- package/dist/eval-runtime/judges/rubric-prompt.js +10 -12
- package/dist/eval-runtime/judges/rubric-readings.d.ts +13 -0
- package/dist/eval-runtime/judges/rubric-readings.js +44 -0
- package/dist/eval-workflows/hosts/adapters/codex/reference-evaluator.js +3 -4
- package/dist/eval-workflows/hosts/adapters/codex/reference-model.js +9 -20
- package/dist/eval-workflows/input-compilation/compile.js +1 -1
- package/dist/eval-workflows/instruments/prompts/judge-prompts.d.ts +1 -1
- package/dist/eval-workflows/instruments/prompts/judge-prompts.js +3 -1
- package/dist/eval-workflows/orchestration/measurement-design.js +31 -15
- package/dist/executors/openai/codex/config.d.ts +16 -0
- package/dist/executors/openai/codex/config.js +37 -0
- package/dist/observability/conversation/catalog.d.ts +6 -1
- package/dist/observability/inbox/view-model.d.ts +0 -1
- package/dist/observability/inbox/view-model.js +0 -6
- package/dist/studio/application/conversations/replay/format.d.ts +4 -4
- package/dist/studio/application/conversations/replay/format.js +6 -19
- package/dist/studio/application/conversations/replay/layout.js +3 -3
- package/dist/studio/application/conversations/replay/projection.js +6 -5
- package/dist/studio/application/display/format.d.ts +35 -0
- package/dist/studio/application/display/format.js +79 -0
- package/dist/studio/application/display/tone.d.ts +10 -0
- package/dist/studio/application/display/tone.js +10 -0
- package/dist/studio/application/knowledge/skill-health.d.ts +7 -2
- package/dist/studio/application/knowledge/skill-health.js +28 -19
- package/dist/studio/application/knowledge/skill-index.js +2 -2
- package/dist/studio/application/knowledge/skill-insights.js +3 -2
- package/dist/studio/application/measure/core-run-format.d.ts +3 -10
- package/dist/studio/application/measure/core-run-format.js +8 -17
- package/dist/studio/application/observe/health-format.d.ts +23 -19
- package/dist/studio/application/observe/health-format.js +44 -36
- package/dist/studio/http/errors.d.ts +1 -1
- package/dist/studio/http/next-context.d.ts +3 -2
- package/dist/studio/http/next-context.js +2 -1
- package/dist/studio/http/next-server.js +16 -26
- package/dist/studio/http/pages/measure-page.d.ts +15 -0
- package/dist/studio/http/pages/measure-page.js +25 -0
- package/dist/studio/http/pages/observe-page.js +2 -0
- package/dist/studio/http/request-handler.js +1 -3
- package/dist/studio/http/routes/knowledge.js +2 -57
- package/dist/studio/http/routes/observations.d.ts +2 -8
- package/dist/studio/http/routes/observations.js +3 -36
- package/dist/studio/view-models/display/tone.d.ts +8 -0
- package/dist/studio/view-models/display/tone.js +1 -0
- package/dist/studio/view-models/knowledge/health-assessment.d.ts +12 -1
- package/dist/studio/view-models/knowledge/skill-index.d.ts +8 -2
- package/dist/studio/web/.next/BUILD_ID +1 -1
- package/dist/studio/web/.next/app-path-routes-manifest.json +5 -5
- package/dist/studio/web/.next/build-manifest.json +2 -2
- package/dist/studio/web/.next/prerender-manifest.json +3 -3
- package/dist/studio/web/.next/server/app/_global-error/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/_global-error.html +1 -1
- package/dist/studio/web/.next/server/app/_global-error.rsc +4 -4
- package/dist/studio/web/.next/server/app/_global-error.segments/_full.segment.rsc +4 -4
- package/dist/studio/web/.next/server/app/_global-error.segments/_global-error/__PAGE__.segment.rsc +4 -4
- package/dist/studio/web/.next/server/app/_global-error.segments/_tree.segment.rsc +1 -1
- package/dist/studio/web/.next/server/app/_not-found/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/candidates/page.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/candidates/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/[managedId]/page.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/[managedId]/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/[managedId]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/page.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/knowledge/managed/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/page.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/knowledge/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/skills/[skillName]/page.js +1 -1
- package/dist/studio/web/.next/server/app/knowledge/skills/[skillName]/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/knowledge/skills/[skillName]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/measure/[runId]/page.js +1 -1
- package/dist/studio/web/.next/server/app/measure/[runId]/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/measure/[runId]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/measure/page.js +1 -1
- package/dist/studio/web/.next/server/app/measure/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/measure/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/conversations/[threadId]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/conversations/[threadId]/tasks/[turnId]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health/[analysisId]/page.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health/[analysisId]/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/observe/health/[analysisId]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health/page.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/observe/health/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health-diff/page.js +1 -1
- package/dist/studio/web/.next/server/app/observe/health-diff/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/observe/health-diff/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/inbox/page.js +8 -8
- package/dist/studio/web/.next/server/app/observe/inbox/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/observe/inbox/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app/observe/skill-trend/[skillName]/page.js +1 -1
- package/dist/studio/web/.next/server/app/observe/skill-trend/[skillName]/page.js.nft.json +1 -1
- package/dist/studio/web/.next/server/app/observe/skill-trend/[skillName]/page_client-reference-manifest.js +1 -1
- package/dist/studio/web/.next/server/app-paths-manifest.json +5 -5
- package/dist/studio/web/.next/server/chunks/121.js +1 -0
- package/dist/studio/web/.next/server/chunks/286.js +1 -1
- package/dist/studio/web/.next/server/chunks/3.js +1 -0
- package/dist/studio/web/.next/server/chunks/475.js +1 -0
- package/dist/studio/web/.next/server/chunks/499.js +1 -1
- package/dist/studio/web/.next/server/chunks/955.js +1 -0
- package/dist/studio/web/.next/server/chunks/996.js +1 -0
- package/dist/studio/web/.next/server/middleware-build-manifest.js +1 -1
- package/dist/studio/web/.next/server/pages/500.html +1 -1
- package/dist/studio/web/.next/server/server-reference-manifest.json +1 -1
- package/dist/studio/web/.next/static/chunks/2038-ed7f68a7ae5147e7.js +1 -0
- package/dist/studio/web/.next/static/chunks/4684-a0ed05ac9d8ab616.js +1 -0
- package/dist/studio/web/.next/static/chunks/5834-509bc646670d19f0.js +1 -0
- package/dist/studio/web/.next/static/chunks/6250-52fb4cf774039f4d.js +1 -0
- package/dist/studio/web/.next/static/chunks/8284-56793a0da5776212.js +1 -0
- package/dist/studio/web/.next/static/chunks/8448-2a7aaa0bc1e3af89.js +1 -0
- package/dist/studio/web/.next/static/chunks/app/knowledge/candidates/page-fe0361724325f726.js +1 -0
- package/dist/studio/web/.next/static/chunks/app/not-found-23ff9fbd04f66261.js +1 -0
- package/dist/studio/web/.next/static/chunks/app/observe/inbox/page-48d0fb8be0af3319.js +1 -0
- package/dist/studio/web/.next/static/css/4a9c7da63c369015.css +1 -0
- package/dist/studio/web/.next/trace +41 -41
- package/dist/studio/web/.next/trace-build +1 -1
- package/package.json +2 -2
- package/dist/studio/web/.next/server/chunks/166.js +0 -1
- package/dist/studio/web/.next/server/chunks/208.js +0 -1
- package/dist/studio/web/.next/server/chunks/342.js +0 -1
- package/dist/studio/web/.next/server/chunks/670.js +0 -1
- package/dist/studio/web/.next/static/chunks/2038-80da4319005c0706.js +0 -1
- package/dist/studio/web/.next/static/chunks/4684-ad2bc276b9605508.js +0 -1
- package/dist/studio/web/.next/static/chunks/5834-fd19cbd0c65fcabb.js +0 -1
- package/dist/studio/web/.next/static/chunks/6250-7ac45071499be5a5.js +0 -1
- package/dist/studio/web/.next/static/chunks/8284-1a543f77443488b5.js +0 -1
- package/dist/studio/web/.next/static/chunks/8448-d4140d31b4fd5886.js +0 -1
- package/dist/studio/web/.next/static/chunks/app/knowledge/candidates/page-8d7ffd0346a5e4c3.js +0 -1
- package/dist/studio/web/.next/static/chunks/app/not-found-a0e503a733ff0da1.js +0 -1
- package/dist/studio/web/.next/static/chunks/app/observe/inbox/page-0f824de8119624e4.js +0 -1
- package/dist/studio/web/.next/static/css/f2bdbbde745c8ff8.css +0 -1
- /package/dist/studio/web/.next/static/{2l1orR2TyYw7vnmDqxoyT → hccv6HVNJDzgZgMSLell8}/_buildManifest.js +0 -0
- /package/dist/studio/web/.next/static/{2l1orR2TyYw7vnmDqxoyT → hccv6HVNJDzgZgMSLell8}/_ssgManifest.js +0 -0
|
@@ -33,7 +33,7 @@ omk CLI 顶层命令包括:`init` / `install` / `list` / `promote` / `rollback
|
|
|
33
33
|
|
|
34
34
|
### 在 Codex / 支持 MCP 的客户端中
|
|
35
35
|
|
|
36
|
-
Codex 是 omk 的一等 runtime。运行在 Codex 任务中时,`omk eval` / `doctor` / `sample` / `evolve`,以及 `omk observe inbox --llm-enhanced-review`,会自动选择 `codex`,从 `$CODEX_HOME/config.toml` 或 `~/.codex/config.toml`
|
|
36
|
+
Codex 是 omk 的一等 runtime。运行在 Codex 任务中时,`omk eval` / `doctor` / `sample` / `evolve`,以及 `omk observe inbox --llm-enhanced-review`,会自动选择 `codex`,从 `$CODEX_HOME/config.toml` 或 `~/.codex/config.toml` 解析模型(顶层 `profile` 指向某个 profile 时取该 profile 的 `model`,否则取顶层 `model`),默认评委沿用同一个 Codex 模型;不要额外回落到 Claude。
|
|
37
37
|
|
|
38
38
|
普通终端想固定走 Codex 时,可以设置 `OMK_EXECUTOR=codex`;`OMK_MODEL` 可覆盖本机 Codex 配置,`OMK_JUDGE_MODELS` 可覆盖默认评委。逐次覆盖仍可使用 `--executor` / `--model` / `--judge-models`。Codex 不需要 Claude Code 风格的 `/omk` slash command,直接执行 CLI。
|
|
39
39
|
|
|
@@ -1,26 +1,17 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from 'node:fs';
|
|
2
2
|
import { homedir } from 'node:os';
|
|
3
3
|
import { join } from 'node:path';
|
|
4
|
+
import { resolveCodexConfigModel } from '../../executors/openai/codex/config.js';
|
|
4
5
|
const CODEX_MODEL_PLACEHOLDER = '<codex-model>';
|
|
5
|
-
function parseTopLevelCodexModel(configText) {
|
|
6
|
-
for (const line of configText.split(/\r?\n/)) {
|
|
7
|
-
if (/^\s*\[/.test(line))
|
|
8
|
-
return null;
|
|
9
|
-
const match = line.match(/^\s*model\s*=\s*(?:"([^"]+)"|'([^']+)')\s*(?:#.*)?$/);
|
|
10
|
-
const model = match?.[1] ?? match?.[2];
|
|
11
|
-
if (model)
|
|
12
|
-
return model;
|
|
13
|
-
}
|
|
14
|
-
return null;
|
|
15
|
-
}
|
|
16
6
|
export function getCodexModelSuggestion(env = process.env) {
|
|
17
7
|
const codexHome = env.CODEX_HOME || join(homedir(), '.codex');
|
|
18
8
|
const configPath = join(codexHome, 'config.toml');
|
|
19
9
|
if (existsSync(configPath)) {
|
|
20
10
|
try {
|
|
21
|
-
const
|
|
22
|
-
if (
|
|
23
|
-
return { model, fromConfig: true, configPath };
|
|
11
|
+
const resolution = resolveCodexConfigModel(readFileSync(configPath, 'utf-8'));
|
|
12
|
+
if (resolution.status === 'resolved') {
|
|
13
|
+
return { model: resolution.model, fromConfig: true, configPath };
|
|
14
|
+
}
|
|
24
15
|
}
|
|
25
16
|
catch { /* best-effort hint only */ }
|
|
26
17
|
}
|
|
@@ -68,8 +68,8 @@ export function resolveCliModel(executor, explicitModel, options = {}) {
|
|
|
68
68
|
return suggestion.model;
|
|
69
69
|
const lang = options.lang ?? 'zh';
|
|
70
70
|
throw new Errors.CLIError(lang === 'zh'
|
|
71
|
-
? `Codex 执行器需要明确模型。请用 --model <model>、设置 OMK_MODEL,或在 ${suggestion.configPath}
|
|
72
|
-
: `The Codex executor needs an explicit model. Pass --model <model>, set OMK_MODEL, or configure
|
|
71
|
+
? `Codex 执行器需要明确模型。请用 --model <model>、设置 OMK_MODEL,或在 ${suggestion.configPath} 配置 model 与 profile。`
|
|
72
|
+
: `The Codex executor needs an explicit model. Pass --model <model>, set OMK_MODEL, or configure model and profile in ${suggestion.configPath}.`, { exit: 2 });
|
|
73
73
|
}
|
|
74
74
|
export function defaultJudgeModel(executor, taskModel) {
|
|
75
75
|
return executorFamily(executor) === 'claude' ? DEFAULT_CLAUDE_JUDGE_MODEL : taskModel;
|
|
@@ -6,8 +6,8 @@
|
|
|
6
6
|
* confirmed:目前 mapper 不产出,如果将来 producer / review-state 写出,会被一并算 inactive
|
|
7
7
|
* —— 跟 confirmed soft standard 的「已被认知、进入处理流程」语义一致。
|
|
8
8
|
*
|
|
9
|
-
* 抽这个 helper
|
|
10
|
-
*
|
|
9
|
+
* 抽这个 helper 是为了让 Insight 投影(影响 skill 健康 / 待优化数)与 Studio 的 active 诊断列表
|
|
10
|
+
* 共用同一份口径,避免「Insight 把 confirmed 算 active 但页面不算」的口径分叉。
|
|
11
11
|
*/
|
|
12
12
|
const ACTIVE_DIAGNOSIS_LIFECYCLES = new Set([
|
|
13
13
|
'detected',
|
|
@@ -632,12 +632,15 @@ export function validateDefinitionSemantics(definition, policy) {
|
|
|
632
632
|
}
|
|
633
633
|
}
|
|
634
634
|
}
|
|
635
|
-
|
|
635
|
+
// Disjoint sample scopes may share an instrument/member/replicate coordinate.
|
|
636
|
+
// Only coordinates that can produce two readings on the same sample conflict.
|
|
637
|
+
const measurementCoordinates = definition.evaluators.flatMap((evaluator) => (evaluator.applicableSampleIds ?? definition.dataset.samples.map((sample) => sample.sampleId)).map((sampleId) => canonicalizeJson({
|
|
638
|
+
sampleId,
|
|
636
639
|
instrumentId: evaluator.measurement.instrumentId,
|
|
637
640
|
ensembleMemberId: evaluator.measurement.ensembleMemberId,
|
|
638
641
|
replicateGroupId: evaluator.measurement.replicateGroupId,
|
|
639
642
|
replicateIndex: evaluator.measurement.replicateIndex,
|
|
640
|
-
}));
|
|
643
|
+
})));
|
|
641
644
|
assertUnique(measurementCoordinates, 'evaluator-measurement-coordinate');
|
|
642
645
|
for (const comparison of definition.comparisons) {
|
|
643
646
|
assertReference(targetIds, comparison.controlTargetId, `comparisons.${comparison.comparisonId}.controlTargetId`, 'Target');
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import type { JsonValue } from '../../eval-core/contracts/index.js';
|
|
2
|
+
import type { RuntimeValueParser } from '../adapters/json-executor.js';
|
|
3
|
+
import { type CustomEvaluator, type CustomEvaluatorInvocation, type CustomEvaluatorResult, type CustomMetricResult, type Metric } from '../custom-evaluator.js';
|
|
4
|
+
type MetricValue = {
|
|
5
|
+
numeric: number;
|
|
6
|
+
boolean: boolean;
|
|
7
|
+
categorical: string;
|
|
8
|
+
text: string;
|
|
9
|
+
ranking: string[];
|
|
10
|
+
};
|
|
11
|
+
type MetricDeclaration<M = Metric> = M extends Metric ? {
|
|
12
|
+
[ValueType in M['valueType']]: Omit<M, 'metricId' | 'missingPolicyId' | 'valueType'> & Readonly<{
|
|
13
|
+
valueType: ValueType;
|
|
14
|
+
missingPolicyId?: 'exclude/v1';
|
|
15
|
+
schema: RuntimeValueParser<MetricValue[ValueType]>;
|
|
16
|
+
}>;
|
|
17
|
+
}[M['valueType']] : never;
|
|
18
|
+
/** The map key is the metricId; the parser sits beside its measurement definition. */
|
|
19
|
+
export type CustomEvaluatorMetric = MetricDeclaration;
|
|
20
|
+
type KeyedMetricResult<Value extends JsonValue> = (Omit<Extract<CustomMetricResult, {
|
|
21
|
+
resultKind: 'score';
|
|
22
|
+
}>, 'metricId' | 'value'> & {
|
|
23
|
+
readonly value: Value;
|
|
24
|
+
}) | Omit<Extract<CustomMetricResult, {
|
|
25
|
+
resultKind: 'missing';
|
|
26
|
+
}>, 'metricId'> | Omit<Extract<CustomMetricResult, {
|
|
27
|
+
resultKind: 'invalid';
|
|
28
|
+
}>, 'metricId'>;
|
|
29
|
+
export type CustomEvaluatorScores<Metrics extends Record<string, CustomEvaluatorMetric>> = Readonly<{
|
|
30
|
+
resultKind: 'completed';
|
|
31
|
+
results: {
|
|
32
|
+
readonly [Id in keyof Metrics]: KeyedMetricResult<ReturnType<Metrics[Id]['schema']['parse']>>;
|
|
33
|
+
};
|
|
34
|
+
usage?: Extract<CustomEvaluatorResult, {
|
|
35
|
+
resultKind: 'completed';
|
|
36
|
+
}>['usage'];
|
|
37
|
+
}> | Extract<CustomEvaluatorResult, {
|
|
38
|
+
resultKind: 'failed';
|
|
39
|
+
}>;
|
|
40
|
+
export type CreateCustomEvaluatorInput<Metrics extends Record<string, CustomEvaluatorMetric>, Bindings extends Record<string, JsonValue> = Record<string, JsonValue>, Parameters extends JsonValue | undefined = JsonValue | undefined> = Omit<CustomEvaluator<Bindings, Parameters>, 'evaluatorKind' | 'metrics' | 'implementation'> & Readonly<{
|
|
41
|
+
metrics: Metrics;
|
|
42
|
+
implementation: Omit<CustomEvaluator<Bindings, Parameters>['implementation'], 'schemas' | 'evaluate'> & Readonly<{
|
|
43
|
+
schemas: Omit<CustomEvaluator<Bindings, Parameters>['implementation']['schemas'], 'values'>;
|
|
44
|
+
evaluate: (invocation: Readonly<CustomEvaluatorInvocation<Bindings, Parameters>>) => CustomEvaluatorScores<NoInfer<Metrics>> | Promise<CustomEvaluatorScores<NoInfer<Metrics>>>;
|
|
45
|
+
}>;
|
|
46
|
+
}>;
|
|
47
|
+
/** Expands an ergonomic declaration into the canonical v2 Custom Evaluator contract. */
|
|
48
|
+
export declare function createCustomEvaluator<const Metrics extends Record<string, CustomEvaluatorMetric>, Bindings extends Record<string, JsonValue>, Parameters extends JsonValue | undefined = JsonValue | undefined>(input: Readonly<CreateCustomEvaluatorInput<Metrics, Bindings, Parameters>>): CustomEvaluator<Bindings, Parameters>;
|
|
49
|
+
export {};
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { captureCustomEvaluator, CustomEvaluatorDeclarationError, } from '../custom-evaluator.js';
|
|
2
|
+
import { EvaluationConfigurationError } from '../evaluation/errors.js';
|
|
3
|
+
/** Expands an ergonomic declaration into the canonical v2 Custom Evaluator contract. */
|
|
4
|
+
export function createCustomEvaluator(input) {
|
|
5
|
+
let metricIds = [];
|
|
6
|
+
try {
|
|
7
|
+
if (input.metrics == null || typeof input.metrics !== 'object' || Array.isArray(input.metrics)) {
|
|
8
|
+
throw new CustomEvaluatorDeclarationError([{ path: ['metrics'], reasonCode: 'invalid-value' }]);
|
|
9
|
+
}
|
|
10
|
+
const entries = Object.entries(input.metrics);
|
|
11
|
+
metricIds = entries.map(([id]) => id);
|
|
12
|
+
if (input.implementation == null) {
|
|
13
|
+
throw new CustomEvaluatorDeclarationError([{ path: ['implementation'], reasonCode: 'invalid-value' }]);
|
|
14
|
+
}
|
|
15
|
+
for (const [id, metric] of entries) {
|
|
16
|
+
if (metric == null || typeof metric !== 'object' || Array.isArray(metric)) {
|
|
17
|
+
throw new CustomEvaluatorDeclarationError([{ path: ['metrics', id], reasonCode: 'invalid-value' }]);
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
const callback = input.implementation.evaluate;
|
|
21
|
+
const evaluator = {
|
|
22
|
+
...input,
|
|
23
|
+
evaluatorKind: 'custom',
|
|
24
|
+
metrics: entries.map(([metricId, { schema: _schema, ...metric }]) => ({
|
|
25
|
+
missingPolicyId: 'exclude/v1', ...metric, metricId,
|
|
26
|
+
})),
|
|
27
|
+
implementation: {
|
|
28
|
+
...input.implementation,
|
|
29
|
+
schemas: {
|
|
30
|
+
...input.implementation.schemas,
|
|
31
|
+
values: Object.fromEntries(entries.map(([id, metric]) => [id, metric.schema])),
|
|
32
|
+
},
|
|
33
|
+
async evaluate(invocation) {
|
|
34
|
+
const result = await callback(invocation);
|
|
35
|
+
// Preserve malformed results for the canonical validator, including reported usage.
|
|
36
|
+
if (result?.resultKind !== 'completed')
|
|
37
|
+
return result;
|
|
38
|
+
if (result.results == null || typeof result.results !== 'object' || Array.isArray(result.results)
|
|
39
|
+
|| Object.values(result.results).some((item) => item != null && typeof item === 'object' && 'metricId' in item)) {
|
|
40
|
+
return { ...result, results: null };
|
|
41
|
+
}
|
|
42
|
+
return {
|
|
43
|
+
...result,
|
|
44
|
+
results: Object.entries(result.results).map(([metricId, item]) => ({ ...item, metricId })),
|
|
45
|
+
};
|
|
46
|
+
},
|
|
47
|
+
},
|
|
48
|
+
};
|
|
49
|
+
if (typeof callback !== 'function') {
|
|
50
|
+
throw new CustomEvaluatorDeclarationError([{ path: ['implementation', 'evaluate'], reasonCode: 'invalid-value' }]);
|
|
51
|
+
}
|
|
52
|
+
captureCustomEvaluator(evaluator);
|
|
53
|
+
return evaluator;
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
throw new EvaluationConfigurationError('EVAL_RUNTIME_EVALUATOR_INVALID', 'Custom Evaluator 配置无效。请查看 issues 中的字段位置。', undefined, error instanceof CustomEvaluatorDeclarationError ? error.issues.map((issue) => {
|
|
57
|
+
const [root, index, ...rest] = issue.path;
|
|
58
|
+
if (root === 'metrics' && typeof index === 'number') {
|
|
59
|
+
return { ...issue, path: ['metrics', metricIds[index], ...rest] };
|
|
60
|
+
}
|
|
61
|
+
if (root === 'implementation' && index === 'schemas' && rest[0] === 'values' && typeof rest[1] === 'string') {
|
|
62
|
+
return { ...issue, path: ['metrics', rest[1], 'schema'] };
|
|
63
|
+
}
|
|
64
|
+
return issue;
|
|
65
|
+
}) : []);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import type { Rubric, RubricJudgeEvaluator } from '../evaluation/contracts.js';
|
|
2
|
+
/** Rubric map keys are metric IDs; all measurement choices retain the standard contract. */
|
|
3
|
+
export type CreateRubricEvaluatorInput = Omit<RubricJudgeEvaluator, 'evaluatorKind' | 'rubrics'> & Readonly<{
|
|
4
|
+
rubrics: Readonly<Record<string, Rubric>>;
|
|
5
|
+
}>;
|
|
6
|
+
/** Validates without invoking a provider and returns a standard Rubric declaration. */
|
|
7
|
+
export declare function createRubricEvaluator(input: Readonly<CreateRubricEvaluatorInput>): RubricJudgeEvaluator;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { IdentifierSchema } from '../../eval-core/contracts/index.js';
|
|
2
|
+
import { captureJudge } from '../evaluation/capture-judge.js';
|
|
3
|
+
import { EvaluationConfigurationError } from '../evaluation/errors.js';
|
|
4
|
+
import { captureRubricDeclaration, rubricAt, rubricIssue } from '../evaluation/rubric-declaration.js';
|
|
5
|
+
/** Validates without invoking a provider and returns a standard Rubric declaration. */
|
|
6
|
+
export function createRubricEvaluator(input) {
|
|
7
|
+
let metricIds = [];
|
|
8
|
+
try {
|
|
9
|
+
return rubricAt(0, [], () => {
|
|
10
|
+
if (input === null || typeof input !== 'object' || Array.isArray(input))
|
|
11
|
+
rubricIssue(0, []);
|
|
12
|
+
if (Object.hasOwn(input, 'evaluatorKind'))
|
|
13
|
+
rubricIssue(0, [], 'unsupported-field');
|
|
14
|
+
const source = rubricAt(0, ['rubrics'], () => input.rubrics);
|
|
15
|
+
if (source === null || typeof source !== 'object' || Array.isArray(source))
|
|
16
|
+
rubricIssue(0, ['rubrics']);
|
|
17
|
+
metricIds = Object.keys(source);
|
|
18
|
+
// Invalid map keys are rejected at the map itself rather than copied into diagnostics.
|
|
19
|
+
if (metricIds.some((id) => !IdentifierSchema.safeParse(id).success))
|
|
20
|
+
rubricIssue(0, ['rubrics']);
|
|
21
|
+
const rubrics = metricIds.map((metricId, index) => rubricAt(0, ['rubrics', index], () => {
|
|
22
|
+
const rubric = source[metricId];
|
|
23
|
+
if (rubric === null || typeof rubric !== 'object' || Array.isArray(rubric))
|
|
24
|
+
rubricIssue(0, ['rubrics', index]);
|
|
25
|
+
if (Object.keys(rubric).some((key) => !['criterionId', 'prompt', 'rubric'].includes(key))) {
|
|
26
|
+
rubricIssue(0, ['rubrics', index], 'unsupported-field');
|
|
27
|
+
}
|
|
28
|
+
return { ...rubric, metricId };
|
|
29
|
+
}));
|
|
30
|
+
const evaluator = { ...input, evaluatorKind: 'rubric-judge', rubrics };
|
|
31
|
+
const { panelJudges } = captureRubricDeclaration(evaluator, 0);
|
|
32
|
+
panelJudges.forEach((member, index) => rubricAt(0, ['judges', index, 'judge'], () => captureJudge(member.judge)));
|
|
33
|
+
return evaluator;
|
|
34
|
+
});
|
|
35
|
+
}
|
|
36
|
+
catch (error) {
|
|
37
|
+
// rubricAt wraps host exceptions before their fields can enter this public error.
|
|
38
|
+
throw new EvaluationConfigurationError('EVAL_RUNTIME_EVALUATOR_INVALID', 'Rubric 评委配置无效。请查看 issues 中的字段位置。', undefined, error instanceof EvaluationConfigurationError ? error.issues.map((issue) => {
|
|
39
|
+
const [, , root, index, ...rest] = issue.path;
|
|
40
|
+
const path = root === 'rubrics' && typeof index === 'number'
|
|
41
|
+
? ['rubrics', metricIds[index], ...rest]
|
|
42
|
+
: issue.path.slice(2);
|
|
43
|
+
return { ...issue, path };
|
|
44
|
+
}) : []);
|
|
45
|
+
}
|
|
46
|
+
}
|
|
@@ -76,12 +76,12 @@ function evaluator(onCall) {
|
|
|
76
76
|
evaluatorKind: 'custom',
|
|
77
77
|
evaluatorId: 'omk-runtime-check-cache-evaluator',
|
|
78
78
|
instrumentId: 'omk.runtime-check.cache-evaluator/v1',
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
79
|
+
metrics: [{
|
|
80
|
+
metricId: 'omk-runtime-check-cache-match',
|
|
81
|
+
valueType: 'boolean',
|
|
82
|
+
direction: 'higher-is-better',
|
|
83
|
+
missingPolicyId: 'exclude/v1',
|
|
84
|
+
}],
|
|
85
85
|
bindings: [
|
|
86
86
|
{ bindingId: 'actual', sourceKind: 'output', pointer: '' },
|
|
87
87
|
{ bindingId: 'expected', sourceKind: 'expected', pointer: '' },
|
|
@@ -91,13 +91,20 @@ function evaluator(onCall) {
|
|
|
91
91
|
version: '1.0.0',
|
|
92
92
|
schemas: {
|
|
93
93
|
bindings: z.object({ actual: z.string(), expected: z.string() }).strict(),
|
|
94
|
-
|
|
94
|
+
values: { 'omk-runtime-check-cache-match': z.boolean() },
|
|
95
95
|
fingerprintFacets: { bindings: 'two-strings/v1', value: 'boolean/v1' },
|
|
96
96
|
},
|
|
97
97
|
fingerprintFacets: { probe: 'omk.runtime-check.cache/v1' },
|
|
98
98
|
async evaluate({ bindings }) {
|
|
99
99
|
onCall();
|
|
100
|
-
return {
|
|
100
|
+
return {
|
|
101
|
+
resultKind: 'completed',
|
|
102
|
+
results: [{
|
|
103
|
+
metricId: 'omk-runtime-check-cache-match',
|
|
104
|
+
resultKind: 'score',
|
|
105
|
+
value: bindings.actual === bindings.expected,
|
|
106
|
+
}],
|
|
107
|
+
};
|
|
101
108
|
},
|
|
102
109
|
},
|
|
103
110
|
};
|
|
@@ -9,13 +9,13 @@ export interface EvaluatorConformanceProbeSources {
|
|
|
9
9
|
export interface EvaluatorConformanceProbeInput<Bindings extends Record<string, JsonValue> = Record<string, JsonValue>, Parameters extends JsonValue | undefined = JsonValue | undefined> {
|
|
10
10
|
readonly evaluator: CustomEvaluator<Bindings, Parameters>;
|
|
11
11
|
readonly score: EvaluatorConformanceProbeSources & {
|
|
12
|
-
readonly
|
|
12
|
+
readonly expectedValues: Readonly<Record<string, JsonValue>>;
|
|
13
13
|
};
|
|
14
14
|
readonly missing: EvaluatorConformanceProbeSources & {
|
|
15
|
-
readonly
|
|
15
|
+
readonly expectedReasonCodes: Readonly<Record<string, string>>;
|
|
16
16
|
};
|
|
17
17
|
readonly invalid: EvaluatorConformanceProbeSources & {
|
|
18
|
-
readonly
|
|
18
|
+
readonly expectedReasonCodes: Readonly<Record<string, string>>;
|
|
19
19
|
};
|
|
20
20
|
readonly failure: EvaluatorConformanceProbeSources & {
|
|
21
21
|
readonly expectedErrorCode: string;
|
|
@@ -115,14 +115,25 @@ export async function runEvaluatorConformance(input) {
|
|
|
115
115
|
return result([check('configuration', false, 'runtime-evaluator-configuration-invalid')]);
|
|
116
116
|
}
|
|
117
117
|
const callback = input.evaluator?.implementation?.evaluate;
|
|
118
|
+
const metricIds = Array.isArray(input.evaluator?.metrics)
|
|
119
|
+
? input.evaluator.metrics.map((metric) => metric?.metricId).sort()
|
|
120
|
+
: [];
|
|
121
|
+
const coversMetrics = (values) => (values !== null && typeof values === 'object' && !Array.isArray(values)
|
|
122
|
+
&& metricIds.length > 0
|
|
123
|
+
&& new Set(metricIds).size === metricIds.length
|
|
124
|
+
&& canonicalizeJson(Object.keys(values).sort()) === canonicalizeJson(metricIds));
|
|
118
125
|
if (typeof callback !== 'function'
|
|
126
|
+
|| metricIds.some((metricId) => !IdentifierSchema.safeParse(metricId).success)
|
|
119
127
|
|| !IdentifierSchema.safeParse(input.probeNamespace).success
|
|
120
128
|
|| !validSources(input.score)
|
|
121
|
-
|| !JsonValueSchema.safeParse(input.score.
|
|
129
|
+
|| !z.record(IdentifierSchema, JsonValueSchema).safeParse(input.score.expectedValues).success
|
|
130
|
+
|| !coversMetrics(input.score.expectedValues)
|
|
122
131
|
|| !validSources(input.missing)
|
|
123
|
-
|| !IdentifierSchema.safeParse(input.missing.
|
|
132
|
+
|| !z.record(IdentifierSchema, IdentifierSchema).safeParse(input.missing.expectedReasonCodes).success
|
|
133
|
+
|| !coversMetrics(input.missing.expectedReasonCodes)
|
|
124
134
|
|| !validSources(input.invalid)
|
|
125
|
-
|| !IdentifierSchema.safeParse(input.invalid.
|
|
135
|
+
|| !z.record(IdentifierSchema, IdentifierSchema).safeParse(input.invalid.expectedReasonCodes).success
|
|
136
|
+
|| !coversMetrics(input.invalid.expectedReasonCodes)
|
|
126
137
|
|| !validSources(input.failure)
|
|
127
138
|
|| !IdentifierSchema.safeParse(input.failure.expectedErrorCode).success
|
|
128
139
|
|| !validSources(input.cancellation)) {
|
|
@@ -247,18 +258,18 @@ export async function runEvaluatorConformance(input) {
|
|
|
247
258
|
&& Object.keys(scoreInvocation.bindings).sort().join(',')
|
|
248
259
|
=== input.evaluator.bindings.map((binding) => binding.bindingId).sort().join(','), 'runtime-evaluator-binding-mismatch'),
|
|
249
260
|
check('score-contract', scoreRecord?.evaluationStatus === 'completed'
|
|
250
|
-
&& scoreRecord.observations.length ===
|
|
251
|
-
&& scoreRecord.observations
|
|
252
|
-
|
|
253
|
-
|
|
261
|
+
&& scoreRecord.observations.length === metricIds.length
|
|
262
|
+
&& scoreRecord.observations.every((observation) => (observation.observationStatus === 'observed'
|
|
263
|
+
&& canonicalizeJson(observation.value)
|
|
264
|
+
=== canonicalizeJson(input.score.expectedValues[observation.metricId]))), 'runtime-evaluator-score-contract-invalid'),
|
|
254
265
|
check('missing-contract', missingRecord?.evaluationStatus === 'completed'
|
|
255
|
-
&& missingRecord.observations.length ===
|
|
256
|
-
&& missingRecord.observations
|
|
257
|
-
|
|
266
|
+
&& missingRecord.observations.length === metricIds.length
|
|
267
|
+
&& missingRecord.observations.every((observation) => (observation.observationStatus === 'missing'
|
|
268
|
+
&& observation.reasonCode === input.missing.expectedReasonCodes[observation.metricId])), 'runtime-evaluator-missing-contract-invalid'),
|
|
258
269
|
check('invalid-contract', invalidRecord?.evaluationStatus === 'completed'
|
|
259
|
-
&& invalidRecord.observations.length ===
|
|
260
|
-
&& invalidRecord.observations
|
|
261
|
-
|
|
270
|
+
&& invalidRecord.observations.length === metricIds.length
|
|
271
|
+
&& invalidRecord.observations.every((observation) => (observation.observationStatus === 'invalid'
|
|
272
|
+
&& observation.reasonCode === input.invalid.expectedReasonCodes[observation.metricId])), 'runtime-evaluator-invalid-contract-invalid'),
|
|
262
273
|
check('failure-contract', failureRecord?.evaluationStatus === 'failed'
|
|
263
274
|
&& failureRecord.error.code === input.failure.expectedErrorCode, 'runtime-evaluator-failure-contract-invalid'),
|
|
264
275
|
check('cancellation-contract', runs[4]?.status === 'cancelled'
|
|
@@ -269,10 +280,10 @@ export async function runEvaluatorConformance(input) {
|
|
|
269
280
|
&& concurrentRun.artifacts.evaluation.records.length === 2
|
|
270
281
|
&& maximumEvaluatorInvocationsInFlight >= 2
|
|
271
282
|
&& concurrentRun.artifacts.evaluation.records.every((record) => (record.evaluationStatus === 'completed'
|
|
272
|
-
&& record.observations.length ===
|
|
273
|
-
&& record.observations
|
|
274
|
-
|
|
275
|
-
|
|
283
|
+
&& record.observations.length === metricIds.length
|
|
284
|
+
&& record.observations.every((observation) => (observation.observationStatus === 'observed'
|
|
285
|
+
&& canonicalizeJson(observation.value)
|
|
286
|
+
=== canonicalizeJson(input.score.expectedValues[observation.metricId]))))), 'runtime-evaluator-concurrency-invalid'),
|
|
276
287
|
check('telemetry-contract', allUsageValid, 'runtime-evaluator-telemetry-invalid'),
|
|
277
288
|
check('core-roundtrip', runs.length === 5
|
|
278
289
|
&& runs.slice(0, 4).every((run) => run.status === 'completed')
|
|
@@ -72,12 +72,11 @@ function evaluationInput(namespace, phase, cases, model, judge) {
|
|
|
72
72
|
evaluators: [{
|
|
73
73
|
evaluatorKind: 'rubric-judge',
|
|
74
74
|
evaluatorId: 'runtime-check-judge',
|
|
75
|
-
metricId: 'runtime-check-judge-score',
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
},
|
|
75
|
+
rubrics: [{ metricId: 'runtime-check-judge-score',
|
|
76
|
+
criterionId: `runtime-check-${phase}`,
|
|
77
|
+
prompt: `OMK Runtime Judge behavioral probe: ${phase}.`,
|
|
78
|
+
rubric: 'Return a numeric score from 1 through 5 using the required JSON shape.',
|
|
79
|
+
}],
|
|
81
80
|
judges: [{ memberId: 'primary', model, judge }],
|
|
82
81
|
aggregation: { method: 'mean', missing: 'require-complete' },
|
|
83
82
|
}],
|
|
@@ -31,7 +31,7 @@ export type WorkspaceProviderRuntimeCheckResult = Readonly<RuntimeCheckResultEnv
|
|
|
31
31
|
export type EvaluatorRuntimeCheckInput<Bindings extends Record<string, JsonValue> = Record<string, JsonValue>, Parameters extends JsonValue | undefined = JsonValue | undefined> = Readonly<{
|
|
32
32
|
readonly runtimeKind: 'evaluator';
|
|
33
33
|
} & EvaluatorConformanceProbeInput<Bindings, Parameters>>;
|
|
34
|
-
export type EvaluatorRuntimeCheckResult = Readonly<RuntimeCheckResultEnvelope<'evaluator', 'omk.runtime-check.custom-evaluator/
|
|
34
|
+
export type EvaluatorRuntimeCheckResult = Readonly<RuntimeCheckResultEnvelope<'evaluator', 'omk.runtime-check.custom-evaluator/v2'> & EvaluatorConformanceResult>;
|
|
35
35
|
export type JudgeRuntimeCheckInput = Readonly<{
|
|
36
36
|
readonly runtimeKind: 'judge';
|
|
37
37
|
} & JudgeConformanceProbeInput>;
|
|
@@ -103,7 +103,7 @@ export async function checkRuntime(input) {
|
|
|
103
103
|
cancellation: input.cancellation,
|
|
104
104
|
probeNamespace: input.probeNamespace,
|
|
105
105
|
...(input.timeoutMs === undefined ? {} : { timeoutMs: input.timeoutMs }),
|
|
106
|
-
}).then((result) => envelope('evaluator', 'omk.runtime-check.custom-evaluator/
|
|
106
|
+
}).then((result) => envelope('evaluator', 'omk.runtime-check.custom-evaluator/v2', requireConfigured(result, ['configuration'], 'Evaluator runtime check declaration 无效。')));
|
|
107
107
|
}
|
|
108
108
|
if (input.runtimeKind === 'judge') {
|
|
109
109
|
if (!hasOnlyKeys(input, [
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type EvaluatorDefinition, type JsonValue, type MetricDefinition, type UsageRecord } from '../eval-core/contracts/index.js';
|
|
2
2
|
import { type EvaluationEvaluator } from '../eval-core/evaluation/index.js';
|
|
3
3
|
import type { RuntimeValueParser } from './adapters/json-executor.js';
|
|
4
|
+
import type { EvaluationConfigurationIssue } from './evaluation/errors.js';
|
|
4
5
|
interface MetricBase {
|
|
5
6
|
readonly metricId: string;
|
|
6
7
|
readonly unit?: string;
|
|
@@ -32,21 +33,26 @@ export interface CustomEvaluatorContent {
|
|
|
32
33
|
readonly classification: 'public' | 'sensitive' | 'secret' | 'gold';
|
|
33
34
|
readonly mediaType?: string;
|
|
34
35
|
}
|
|
35
|
-
export type
|
|
36
|
+
export type CustomMetricResult = Readonly<{
|
|
37
|
+
metricId: string;
|
|
36
38
|
resultKind: 'score';
|
|
37
39
|
value: JsonValue;
|
|
38
40
|
evidence?: CustomEvaluatorContent;
|
|
39
|
-
usage?: UsageRecord;
|
|
40
41
|
}> | Readonly<{
|
|
42
|
+
metricId: string;
|
|
41
43
|
resultKind: 'missing';
|
|
42
44
|
reasonCode: string;
|
|
43
45
|
evidence?: CustomEvaluatorContent;
|
|
44
|
-
usage?: UsageRecord;
|
|
45
46
|
}> | Readonly<{
|
|
47
|
+
metricId: string;
|
|
46
48
|
resultKind: 'invalid';
|
|
47
49
|
reasonCode: string;
|
|
48
50
|
invalidValue?: CustomEvaluatorContent;
|
|
49
51
|
evidence?: CustomEvaluatorContent;
|
|
52
|
+
}>;
|
|
53
|
+
export type CustomEvaluatorResult = Readonly<{
|
|
54
|
+
resultKind: 'completed';
|
|
55
|
+
results: readonly CustomMetricResult[];
|
|
50
56
|
usage?: UsageRecord;
|
|
51
57
|
}> | Readonly<{
|
|
52
58
|
resultKind: 'failed';
|
|
@@ -69,7 +75,7 @@ export interface CustomEvaluator<Bindings extends Record<string, JsonValue> = Re
|
|
|
69
75
|
readonly evaluatorKind: 'custom';
|
|
70
76
|
readonly evaluatorId: string;
|
|
71
77
|
readonly instrumentId: string;
|
|
72
|
-
readonly
|
|
78
|
+
readonly metrics: readonly Metric[];
|
|
73
79
|
readonly bindings: readonly CustomEvaluatorBinding[];
|
|
74
80
|
readonly parameters?: Parameters;
|
|
75
81
|
readonly implementation: Readonly<{
|
|
@@ -77,7 +83,7 @@ export interface CustomEvaluator<Bindings extends Record<string, JsonValue> = Re
|
|
|
77
83
|
version: string;
|
|
78
84
|
schemas: Readonly<{
|
|
79
85
|
bindings: RuntimeValueParser<Bindings>;
|
|
80
|
-
|
|
86
|
+
values: Readonly<Record<string, RuntimeValueParser<JsonValue>>>;
|
|
81
87
|
fingerprintFacets: JsonValue;
|
|
82
88
|
}>;
|
|
83
89
|
providerCost?: Readonly<{
|
|
@@ -93,13 +99,14 @@ export interface CustomEvaluator<Bindings extends Record<string, JsonValue> = Re
|
|
|
93
99
|
}
|
|
94
100
|
export interface CapturedCustomEvaluator {
|
|
95
101
|
readonly definition: EvaluatorDefinition;
|
|
96
|
-
readonly
|
|
102
|
+
readonly metrics: readonly MetricDefinition[];
|
|
97
103
|
readonly port: EvaluationEvaluator;
|
|
98
104
|
readonly implementationId: string;
|
|
99
105
|
readonly version: string;
|
|
100
106
|
}
|
|
101
107
|
export declare class CustomEvaluatorDeclarationError extends TypeError {
|
|
102
|
-
|
|
108
|
+
readonly issues: readonly EvaluationConfigurationIssue[];
|
|
109
|
+
constructor(issues?: readonly EvaluationConfigurationIssue[]);
|
|
103
110
|
}
|
|
104
111
|
/** Captures one canonical custom declaration and adapts it to the Core Evaluator port. */
|
|
105
112
|
export declare function captureCustomEvaluator(value: Readonly<CustomEvaluator>): Readonly<CapturedCustomEvaluator>;
|