oh-my-knowledge 0.47.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +19 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +44 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +35 -4
- package/dist/cli/commands/sample.d.ts +9 -3
- package/dist/cli/commands/sample.js +91 -74
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +2 -1
- package/dist/observability/inbox-view-model.js +20 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +227 -91
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/shell-quote.d.ts +2 -0
- package/dist/shared/shell-quote.js +7 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -3,8 +3,8 @@ import type { ObservationInboxItem } from '../types/observability.js';
|
|
|
3
3
|
interface GenerateSamplesOptions {
|
|
4
4
|
skillContent: string;
|
|
5
5
|
count?: number;
|
|
6
|
-
model
|
|
7
|
-
executorName
|
|
6
|
+
model: string;
|
|
7
|
+
executorName: string;
|
|
8
8
|
/**
|
|
9
9
|
* 自然语言描述用户希望重点覆盖的场景。会作为额外约束追加到 prompt 末尾,
|
|
10
10
|
* 优先于"自由发挥"的多样性。空串 / undefined 表示不施加额外约束。
|
|
@@ -66,7 +66,7 @@ export declare function buildSamplesFromTracesPrompt(items: TraceSignalItem[], c
|
|
|
66
66
|
export interface GenerateSamplesFromTracesOptions {
|
|
67
67
|
items: TraceSignalItem[];
|
|
68
68
|
count?: number;
|
|
69
|
-
model
|
|
69
|
+
model: string;
|
|
70
70
|
executorName?: string;
|
|
71
71
|
/** Injectable executor (tests). Defaults to createExecutor(executorName). */
|
|
72
72
|
executor?: ExecutorFn;
|
|
@@ -1,13 +1,5 @@
|
|
|
1
1
|
import { createExecutor } from '../executors/index.js';
|
|
2
2
|
import { DEFAULT_GATE_THRESHOLD } from '../eval-core/verdict.js';
|
|
3
|
-
/**
|
|
4
|
-
* Generator 默认模型 'opus' (跟 eval 默认对齐)。
|
|
5
|
-
* lean=true 路径会自动追加 `--effort low`,关掉 opus 默认的扩展思考,
|
|
6
|
-
* 所以 opus + lean + effort-low 在 generator 场景下速度仍然可控(单 skill ~30-60s)。
|
|
7
|
-
* 成本约 sonnet 的 5x,但 opus 在结构化指令遵循 / 长 prompt 一致性上更稳。
|
|
8
|
-
* 用户想要省钱时显式 `--model sonnet` 即可。
|
|
9
|
-
*/
|
|
10
|
-
const GENERATOR_DEFAULT_MODEL = 'sonnet';
|
|
11
3
|
const SYSTEM_PROMPT = `你是一个评测用例生成器。你的任务是根据用户提供的 skill(系统提示词)内容,生成高质量的评测用例。
|
|
12
4
|
|
|
13
5
|
样本结构决策(必须先做):先扫一遍 skill 内容判断它属于哪一类,按对应配比和数量生成。
|
|
@@ -398,7 +390,7 @@ ${skillContent}
|
|
|
398
390
|
|
|
399
391
|
${countLine}直接输出 JSON 数组。${focusBlock}${noMockBlock}`;
|
|
400
392
|
}
|
|
401
|
-
export async function generateSamples({ skillContent, count, model
|
|
393
|
+
export async function generateSamples({ skillContent, count, model, executorName, focus, noMock }) {
|
|
402
394
|
const executor = createExecutor(executorName);
|
|
403
395
|
const prompt = buildSamplesPrompt({ skillContent, count, focus, noMock });
|
|
404
396
|
// 生成场景比单次 eval 调用更重(LLM 要思考结构 + 输出大段 JSON),
|
|
@@ -547,9 +539,12 @@ function traceSanitizeContext(items) {
|
|
|
547
539
|
* stamps `provenance: 'production-trace'`. Output is meant to land in a review draft,
|
|
548
540
|
* not the live dataset (the CLI enforces that).
|
|
549
541
|
*/
|
|
550
|
-
export async function generateSamplesFromTraces({ items, count, model
|
|
542
|
+
export async function generateSamplesFromTraces({ items, count, model, executorName, executor: injectedExecutor, }) {
|
|
551
543
|
if (items.length === 0)
|
|
552
544
|
return { samples: [], costUSD: 0 };
|
|
545
|
+
if (!injectedExecutor && !executorName) {
|
|
546
|
+
throw new Error('executorName is required when no executor is injected');
|
|
547
|
+
}
|
|
553
548
|
const executor = injectedExecutor ?? createExecutor(executorName);
|
|
554
549
|
const prompt = buildSamplesFromTracesPrompt(items, count);
|
|
555
550
|
const sanitizeContext = traceSanitizeContext(items);
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { EvaluationReport } from '../types/report.js';
|
|
2
|
+
import type { ToolCallInfo } from '../types/index.js';
|
|
2
3
|
export interface FixContext {
|
|
3
4
|
sampleId: string;
|
|
4
5
|
originalSample: Record<string, unknown>;
|
|
@@ -12,11 +13,7 @@ export interface FixContext {
|
|
|
12
13
|
value: string;
|
|
13
14
|
passed: boolean;
|
|
14
15
|
}>;
|
|
15
|
-
toolCalls: Array<
|
|
16
|
-
tool: string;
|
|
17
|
-
input: unknown;
|
|
18
|
-
success: boolean;
|
|
19
|
-
}>;
|
|
16
|
+
toolCalls: Array<Pick<ToolCallInfo, 'tool' | 'input' | 'success' | 'status'>>;
|
|
20
17
|
}
|
|
21
18
|
export interface FixSamplesOptions {
|
|
22
19
|
skillContent: string;
|
|
@@ -33,18 +30,23 @@ export interface FixSamplesOptions {
|
|
|
33
30
|
ok: boolean;
|
|
34
31
|
text: string;
|
|
35
32
|
costUSD: number;
|
|
33
|
+
costReported?: boolean;
|
|
36
34
|
}>;
|
|
37
|
-
model
|
|
35
|
+
model: string;
|
|
38
36
|
maxAttemptsPerSample?: number;
|
|
39
37
|
}
|
|
40
38
|
export interface FixSamplesResult {
|
|
41
39
|
samples: Record<string, unknown>[];
|
|
42
40
|
fixedCount: number;
|
|
43
41
|
costUSD: number;
|
|
42
|
+
/** false 表示 costUSD 只是执行器已上报部分的下界。 */
|
|
43
|
+
costReported: boolean;
|
|
44
44
|
fixes: Array<{
|
|
45
45
|
sampleId: string;
|
|
46
46
|
changed: boolean;
|
|
47
47
|
error?: string;
|
|
48
48
|
}>;
|
|
49
49
|
}
|
|
50
|
+
export declare function sampleFixWithinScope(previous: Record<string, unknown>, next: Record<string, unknown>): boolean;
|
|
51
|
+
export declare function stampFixMetadata(sample: Record<string, unknown>, reportId: string): void;
|
|
50
52
|
export declare function fixSamples(options: FixSamplesOptions): Promise<FixSamplesResult>;
|
|
@@ -1,10 +1,11 @@
|
|
|
1
|
+
import { toolCallStatus } from '../shared/tool-call-status.js';
|
|
1
2
|
import { sanitizeGeneratedSamples } from './generator.js';
|
|
2
3
|
const FIX_SYSTEM_PROMPT = `你是一个评测用例修复专家。根据诊断信息和失败断言修复评测用例(sample)。
|
|
3
4
|
|
|
4
5
|
修复原则:
|
|
5
6
|
1. 先分析失败原因——是 sample 设计问题(mock 缺失 / 断言写错 / 工具名不匹配)还是 LLM 行为问题(LLM 真的做错了,低分合理)
|
|
6
7
|
2. **如果是 LLM 行为问题(低分合理),不要改 sample,原样返回**
|
|
7
|
-
3. 如果是 sample
|
|
8
|
+
3. 如果是 sample 设计问题,只修复 mocks / mocksStrict / assertions / environment
|
|
8
9
|
4. 保持 sample_id 不变,保持测试意图不变
|
|
9
10
|
5. provenance 只能使用现有合法枚举: "human" / "llm-generated" / "production-trace"。不要发明 "llm-generated-fixed"、"fixed" 等新值;不确定时保留原值或用 "llm-generated"
|
|
10
11
|
6. 断言修复只保留核心行为节点:必要输出位置、必须/禁止的关键副作用、关键安全约束。不要为了贴合某次执行轨迹而绑定具体工具、命令形态、文案、完整步骤顺序或临时实现细节
|
|
@@ -23,6 +24,30 @@ function sanitizeFixedSamples(samples, skillContent) {
|
|
|
23
24
|
sanitizeGeneratedSamples(samples, { skillContent });
|
|
24
25
|
return samples;
|
|
25
26
|
}
|
|
27
|
+
const FIXABLE_SAMPLE_FIELDS = new Set([
|
|
28
|
+
'assertions',
|
|
29
|
+
'mocks',
|
|
30
|
+
'mocksStrict',
|
|
31
|
+
'environment',
|
|
32
|
+
]);
|
|
33
|
+
function canonicalStringify(value) {
|
|
34
|
+
if (value === null || typeof value !== 'object')
|
|
35
|
+
return JSON.stringify(value);
|
|
36
|
+
if (Array.isArray(value))
|
|
37
|
+
return `[${value.map(canonicalStringify).join(',')}]`;
|
|
38
|
+
const record = value;
|
|
39
|
+
return `{${Object.keys(record)
|
|
40
|
+
.sort()
|
|
41
|
+
.map((key) => `${JSON.stringify(key)}:${canonicalStringify(record[key])}`)
|
|
42
|
+
.join(',')}}`;
|
|
43
|
+
}
|
|
44
|
+
function sampleOutsideFixScope(sample) {
|
|
45
|
+
return Object.fromEntries(Object.entries(sample).filter(([key]) => !FIXABLE_SAMPLE_FIELDS.has(key)));
|
|
46
|
+
}
|
|
47
|
+
export function sampleFixWithinScope(previous, next) {
|
|
48
|
+
return canonicalStringify(sampleOutsideFixScope(previous))
|
|
49
|
+
=== canonicalStringify(sampleOutsideFixScope(next));
|
|
50
|
+
}
|
|
26
51
|
function fixAttempts(sample) {
|
|
27
52
|
const meta = sample.omkFix;
|
|
28
53
|
if (!meta || typeof meta !== 'object')
|
|
@@ -30,7 +55,7 @@ function fixAttempts(sample) {
|
|
|
30
55
|
const attempts = meta.attempts;
|
|
31
56
|
return typeof attempts === 'number' && Number.isFinite(attempts) ? attempts : 0;
|
|
32
57
|
}
|
|
33
|
-
function stampFixMetadata(sample, reportId) {
|
|
58
|
+
export function stampFixMetadata(sample, reportId) {
|
|
34
59
|
sample.omkFix = {
|
|
35
60
|
...(typeof sample.omkFix === 'object' && sample.omkFix !== null ? sample.omkFix : {}),
|
|
36
61
|
attempts: fixAttempts(sample) + 1,
|
|
@@ -72,7 +97,7 @@ function parseFixedSamples(text) {
|
|
|
72
97
|
return null;
|
|
73
98
|
}
|
|
74
99
|
export async function fixSamples(options) {
|
|
75
|
-
const { skillContent, samples, report, treatmentKey, executor, model
|
|
100
|
+
const { skillContent, samples, report, treatmentKey, executor, model, maxAttemptsPerSample = 2 } = options;
|
|
76
101
|
const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
|
|
77
102
|
// Collect all fixable samples into one batch
|
|
78
103
|
const fixContexts = [];
|
|
@@ -114,6 +139,7 @@ export async function fixSamples(options) {
|
|
|
114
139
|
samples,
|
|
115
140
|
fixedCount: 0,
|
|
116
141
|
costUSD: 0,
|
|
142
|
+
costReported: true,
|
|
117
143
|
fixes: skippedByAttempts.map((sampleId) => ({ sampleId, changed: false, error: `max attempts reached (${maxAttemptsPerSample})` })),
|
|
118
144
|
};
|
|
119
145
|
}
|
|
@@ -127,7 +153,7 @@ export async function fixSamples(options) {
|
|
|
127
153
|
.map((a) => ` - ${a.type}: ${a.value}`)
|
|
128
154
|
.join('\n');
|
|
129
155
|
const toolCallsSummary = ctx.toolCalls.length > 0
|
|
130
|
-
? ctx.toolCalls.map((tc, i) => ` [${i}] ${tc.tool}
|
|
156
|
+
? ctx.toolCalls.map((tc, i) => ` [${i}] ${tc.tool} status=${toolCallStatus(tc)} input=${JSON.stringify(tc.input).slice(0, 150)}`).join('\n')
|
|
131
157
|
: ' (无工具调用)';
|
|
132
158
|
return `### ${ctx.sampleId}
|
|
133
159
|
|
|
@@ -156,13 +182,18 @@ ${skillPreview}
|
|
|
156
182
|
${sampleSections}
|
|
157
183
|
|
|
158
184
|
请分析每条用例的失败原因,判断是 sample 设计问题还是 LLM 行为问题。若是 sample 设计问题,优先把断言收敛到核心行为节点,放松易随机的工具/命令/文案/完整步骤绑定;不要为了过用例而要求修改 SKILL.md,也不要把 skill 的完整流程细节塞进 sample。输出一个 JSON 数组,包含所有 ${fixContexts.length} 条 sample(修改过的和原样保留的都要包含)。`;
|
|
185
|
+
let incurredCostUSD = 0;
|
|
186
|
+
let incurredCostReported = false;
|
|
159
187
|
try {
|
|
160
188
|
const result = await executor({ model, system: FIX_SYSTEM_PROMPT, prompt, timeoutMs: 300_000 });
|
|
189
|
+
incurredCostUSD = result.costUSD;
|
|
190
|
+
incurredCostReported = result.costReported !== false;
|
|
161
191
|
if (!result.ok) {
|
|
162
192
|
return {
|
|
163
193
|
samples,
|
|
164
194
|
fixedCount: 0,
|
|
165
195
|
costUSD: result.costUSD,
|
|
196
|
+
costReported: incurredCostReported,
|
|
166
197
|
fixes: fixContexts.map((ctx) => ({ sampleId: ctx.sampleId, changed: false, error: 'executor failed' })),
|
|
167
198
|
};
|
|
168
199
|
}
|
|
@@ -172,10 +203,48 @@ ${sampleSections}
|
|
|
172
203
|
samples,
|
|
173
204
|
fixedCount: 0,
|
|
174
205
|
costUSD: result.costUSD,
|
|
206
|
+
costReported: incurredCostReported,
|
|
175
207
|
fixes: fixContexts.map((ctx) => ({ sampleId: ctx.sampleId, changed: false, error: 'failed to parse LLM response as JSON array' })),
|
|
176
208
|
};
|
|
177
209
|
}
|
|
210
|
+
const expectedIds = fixContexts.map((context) => context.sampleId);
|
|
211
|
+
const returnedIds = fixedArr.map((sample) => sample.sample_id);
|
|
212
|
+
const returnedIdSet = new Set(returnedIds);
|
|
213
|
+
if (returnedIds.some((sampleId) => typeof sampleId !== 'string')
|
|
214
|
+
|| returnedIdSet.size !== returnedIds.length
|
|
215
|
+
|| returnedIdSet.size !== expectedIds.length
|
|
216
|
+
|| expectedIds.some((sampleId) => !returnedIdSet.has(sampleId))) {
|
|
217
|
+
return {
|
|
218
|
+
samples,
|
|
219
|
+
fixedCount: 0,
|
|
220
|
+
costUSD: result.costUSD,
|
|
221
|
+
costReported: incurredCostReported,
|
|
222
|
+
fixes: fixContexts.map((ctx) => ({
|
|
223
|
+
sampleId: ctx.sampleId,
|
|
224
|
+
changed: false,
|
|
225
|
+
error: 'fixer must return every requested sample_id exactly once and no others',
|
|
226
|
+
})),
|
|
227
|
+
};
|
|
228
|
+
}
|
|
178
229
|
const sanitizedFixedArr = sanitizeFixedSamples(fixedArr, skillContent);
|
|
230
|
+
const outOfScope = sanitizedFixedArr.find((fixed) => {
|
|
231
|
+
const sid = fixed.sample_id;
|
|
232
|
+
const original = sampleMap.get(sid);
|
|
233
|
+
return !original || !sampleFixWithinScope(original, fixed);
|
|
234
|
+
});
|
|
235
|
+
if (outOfScope) {
|
|
236
|
+
return {
|
|
237
|
+
samples,
|
|
238
|
+
fixedCount: 0,
|
|
239
|
+
costUSD: result.costUSD,
|
|
240
|
+
costReported: incurredCostReported,
|
|
241
|
+
fixes: fixContexts.map((ctx) => ({
|
|
242
|
+
sampleId: ctx.sampleId,
|
|
243
|
+
changed: false,
|
|
244
|
+
error: `fixer changed protected fields in sample ${String(outOfScope.sample_id)}`,
|
|
245
|
+
})),
|
|
246
|
+
};
|
|
247
|
+
}
|
|
179
248
|
const fixes = [];
|
|
180
249
|
for (const fixed of sanitizedFixedArr) {
|
|
181
250
|
const sid = fixed.sample_id;
|
|
@@ -195,6 +264,7 @@ ${sampleSections}
|
|
|
195
264
|
samples: samples.map((s) => sampleMap.get(s.sample_id) ?? s),
|
|
196
265
|
fixedCount: fixes.filter((f) => f.changed).length,
|
|
197
266
|
costUSD: result.costUSD,
|
|
267
|
+
costReported: incurredCostReported,
|
|
198
268
|
fixes: [
|
|
199
269
|
...fixes,
|
|
200
270
|
...skippedByAttempts.map((sampleId) => ({ sampleId, changed: false, error: `max attempts reached (${maxAttemptsPerSample})` })),
|
|
@@ -205,7 +275,8 @@ ${sampleSections}
|
|
|
205
275
|
return {
|
|
206
276
|
samples,
|
|
207
277
|
fixedCount: 0,
|
|
208
|
-
costUSD:
|
|
278
|
+
costUSD: incurredCostUSD,
|
|
279
|
+
costReported: incurredCostReported,
|
|
209
280
|
fixes: fixContexts.map((ctx) => ({ sampleId: ctx.sampleId, changed: false, error: String(err) })),
|
|
210
281
|
};
|
|
211
282
|
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { mkdirSync, readFileSync, readdirSync, unlinkSync
|
|
1
|
+
import { mkdirSync, readFileSync, readdirSync, unlinkSync } from 'node:fs';
|
|
2
2
|
import { join, resolve } from 'node:path';
|
|
3
3
|
import { Args, Flags } from '@oclif/core';
|
|
4
4
|
import { LANG_FLAG, bilingual } from '../oclif/i18n.js';
|
|
@@ -7,12 +7,15 @@ import { enumStringParser, integerStringParser, numberStringParser } from '../oc
|
|
|
7
7
|
import { CliExit } from '../lib/cli-exit.js';
|
|
8
8
|
import { tCli } from '../lib/i18n.js';
|
|
9
9
|
import { makeDoctorProgress } from '../lib/progress.js';
|
|
10
|
+
import { resolveCliExecutor, resolveRuntimeSelection } from '../lib/runtime-defaults.js';
|
|
10
11
|
import { DEFAULT_DOCTORS_DIR } from '../../eval-core/default-dirs.js';
|
|
11
12
|
import { indexDoctorWrite, removeDoctorCard } from '../../eval-core/artifact-index.js';
|
|
12
|
-
import { doctorReportFileStem, isReportFileName, reportFilePath } from '../../eval-core/artifact-file-names.js';
|
|
13
|
+
import { doctorReportFileStem, isReportFileName, reportFilePath, reportFileStem } from '../../eval-core/artifact-file-names.js';
|
|
13
14
|
import { migrateLegacyReportFiles } from '../../eval-core/report-file-migration.js';
|
|
14
15
|
import { projectDoctorsDir, globalDoctorsDir } from '../../eval-core/measurement-dirs.js';
|
|
15
16
|
import { persistDoctorGraphSidecars, removeDoctorGraphSidecars } from '../../artifact-graph/doctor.js';
|
|
17
|
+
import { parseDoctorReport } from '../../shared/doctor-report.js';
|
|
18
|
+
import { writeJsonFileAtomic } from '../../shared/atomic-json.js';
|
|
16
19
|
export default class Doctor extends BaseCommand {
|
|
17
20
|
static description = bilingual({
|
|
18
21
|
zh: '体检 omk 工作目录:先跑静态规则,再对 skill 做多维度 LLM 健康度审计(默认 --repeat 2 采样 + 共识归并)。',
|
|
@@ -75,14 +78,14 @@ export default class Doctor extends BaseCommand {
|
|
|
75
78
|
}),
|
|
76
79
|
executor: Flags.string({
|
|
77
80
|
description: bilingual({
|
|
78
|
-
zh: '
|
|
79
|
-
en: 'Executor name
|
|
81
|
+
zh: '执行器名。Codex 任务内自动用 codex;也可用 OMK_EXECUTOR 设置环境偏好。指定为测试 fixture 路径可在测试里跑。',
|
|
82
|
+
en: 'Executor name. Defaults to codex inside Codex tasks; OMK_EXECUTOR sets an environment preference. A test fixture path is also accepted in tests.',
|
|
80
83
|
}),
|
|
81
84
|
}),
|
|
82
85
|
model: Flags.string({
|
|
83
86
|
description: bilingual({
|
|
84
|
-
zh: 'LLM model
|
|
85
|
-
en: 'LLM model name
|
|
87
|
+
zh: 'LLM model 名。Codex 自动读取本机配置;也可用 OMK_MODEL 设置环境偏好。',
|
|
88
|
+
en: 'LLM model name. Codex reads the local configured model; OMK_MODEL sets an environment preference.',
|
|
86
89
|
}),
|
|
87
90
|
}),
|
|
88
91
|
timeout: Flags.string({
|
|
@@ -151,8 +154,15 @@ export default class Doctor extends BaseCommand {
|
|
|
151
154
|
const lang = this.lang;
|
|
152
155
|
await this.runWithCliExit(async () => {
|
|
153
156
|
const target = args.target ?? null;
|
|
154
|
-
const
|
|
155
|
-
const
|
|
157
|
+
const staticOnly = flags['static-only'];
|
|
158
|
+
const runtime = staticOnly
|
|
159
|
+
? {
|
|
160
|
+
executor: resolveCliExecutor(flags.executor),
|
|
161
|
+
model: flags.model ?? 'static-only',
|
|
162
|
+
}
|
|
163
|
+
: resolveRuntimeSelection({ executor: flags.executor, model: flags.model }, { lang });
|
|
164
|
+
const executorName = runtime.executor;
|
|
165
|
+
const model = runtime.model;
|
|
156
166
|
// omk doctor 默认 = 静态规则 + LLM 健康度审计(7 内置维度 + 用户注册的自定义维度);
|
|
157
167
|
// --static-only = 只跑静态检测(readable / metadata / 正文依赖,不调 LLM、不读 samples)。
|
|
158
168
|
// samples_contract_aligned 仍只归 eval preflight(它要 samples.json,与离线解耦)。
|
|
@@ -184,7 +194,6 @@ export default class Doctor extends BaseCommand {
|
|
|
184
194
|
// 默认:静态规则 + 在线检查(LLM health composer + endpoint 自定义维度 external=true)。
|
|
185
195
|
// --static-only:只跑静态检测(纯静态内置 rule),但排除 samples_contract_aligned
|
|
186
196
|
// (那条要 samples.json,与离线解耦) → 且不加载 samples,依赖检查只扫 skill 正文。
|
|
187
|
-
const staticOnly = flags['static-only'];
|
|
188
197
|
const isOnline = (r) => isComposerRule(r) || r.external === true;
|
|
189
198
|
const rulesOverride = staticOnly
|
|
190
199
|
? getRegisteredRules().filter((r) => !isOnline(r) && r.id !== 'samples_contract_aligned')
|
|
@@ -298,7 +307,10 @@ function persistDoctorReport(report, outputDir, lang = 'zh') {
|
|
|
298
307
|
};
|
|
299
308
|
const cardId = doctorReportFileStem(skill.skillName, report.id);
|
|
300
309
|
const filePath = reportFilePath(dir, cardId);
|
|
301
|
-
|
|
310
|
+
const parsed = parseDoctorReport(perSkill);
|
|
311
|
+
if (!parsed)
|
|
312
|
+
throw new Error('invalid doctor report');
|
|
313
|
+
writeJsonFileAtomic(filePath, parsed);
|
|
302
314
|
// 产物发现索引:per-skill 报告落项目本地后,best-effort 追加全局轻卡片,让 studio 跨项目聚合。
|
|
303
315
|
indexDoctorWrite({
|
|
304
316
|
id: cardId, path: filePath, skillName: skill.skillName, reportId: report.id, timestamp: report.timestamp,
|
|
@@ -328,19 +340,24 @@ function persistDoctorReport(report, outputDir, lang = 'zh') {
|
|
|
328
340
|
// 按 timestamp 倒排,保留 maxKeep 份最近的,其余删。按 content 匹配 skillName 不
|
|
329
341
|
// 看文件名,所以清理逻辑不依赖 readdir 顺序或 stem 推断 skill 名。
|
|
330
342
|
export function pruneDoctorHistory(dir, skillName, maxKeep) {
|
|
343
|
+
if (!Number.isSafeInteger(maxKeep) || maxKeep < 0) {
|
|
344
|
+
throw new TypeError('maxKeep must be a non-negative safe integer');
|
|
345
|
+
}
|
|
331
346
|
migrateLegacyReportFiles(dir, 'doctor');
|
|
332
347
|
const candidates = [];
|
|
333
348
|
for (const file of readdirSync(dir)) {
|
|
334
349
|
if (!isReportFileName(file))
|
|
335
350
|
continue;
|
|
336
351
|
try {
|
|
337
|
-
const data = JSON.parse(readFileSync(join(dir, file), 'utf-8'));
|
|
338
|
-
|
|
339
|
-
if (!kind || !Array.isArray(data.skills) || data.skills.length !== 1)
|
|
352
|
+
const data = parseDoctorReport(JSON.parse(readFileSync(join(dir, file), 'utf-8')));
|
|
353
|
+
if (!data || data.skills.length !== 1)
|
|
340
354
|
continue;
|
|
341
355
|
if (data.skills[0].skillName !== skillName)
|
|
342
356
|
continue;
|
|
343
|
-
|
|
357
|
+
const expectedStem = doctorReportFileStem(skillName, data.id);
|
|
358
|
+
if (reportFileStem(file) !== expectedStem)
|
|
359
|
+
continue;
|
|
360
|
+
candidates.push({ file, graphStem: expectedStem, timestamp: data.timestamp });
|
|
344
361
|
}
|
|
345
362
|
catch { /* skip corrupt / unrelated json */ }
|
|
346
363
|
}
|
|
@@ -1,4 +1,7 @@
|
|
|
1
1
|
import { BaseCommand } from '../../oclif/base-command.js';
|
|
2
|
+
import { type CliLang } from '../../lib/i18n.js';
|
|
3
|
+
import { type RunConfig } from '../../lib/parse-run-config.js';
|
|
4
|
+
export declare function formatConnectivityFailureHint(message: string, config: Pick<RunConfig, 'executorName' | 'model' | 'judgeModels' | 'noJudge'>, lang: CliLang, env?: NodeJS.ProcessEnv): string;
|
|
2
5
|
export default class Eval extends BaseCommand {
|
|
3
6
|
static description: string;
|
|
4
7
|
static examples: {
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { existsSync } from 'node:fs';
|
|
2
|
+
import { join, relative, sep } from 'node:path';
|
|
1
3
|
import { Flags } from '@oclif/core';
|
|
2
4
|
import { LANG_FLAG, bilingual } from '../../oclif/i18n.js';
|
|
3
5
|
import { BaseCommand } from '../../oclif/base-command.js';
|
|
@@ -7,10 +9,17 @@ import { tCli } from '../../lib/i18n.js';
|
|
|
7
9
|
import { parseRunConfig } from '../../lib/parse-run-config.js';
|
|
8
10
|
import { makeOnProgress } from '../../lib/progress.js';
|
|
9
11
|
import { computeRunTally } from '../../lib/run-tally.js';
|
|
12
|
+
import { codexModelFlagValue, codexModelHint } from '../../lib/codex-model-hint.js';
|
|
13
|
+
import { looksLikeModelUnavailableFailure } from '../../lib/llm-failure-classifier.js';
|
|
10
14
|
import { DEFAULT_BOOTSTRAP_SAMPLES } from '../../../eval-core/bootstrap.js';
|
|
11
15
|
import { DEFAULT_GATE_THRESHOLD } from '../../../eval-core/verdict.js';
|
|
12
16
|
import { EVALUATION_REPORT_SCHEMA_VERSION } from '../../../eval-core/evaluation-reporting.js';
|
|
13
17
|
import { findSingleTreatmentDeprecatedSamplesHint, hasUsableSamplesPath, } from '../../../inputs/sample-locator.js';
|
|
18
|
+
import { shellQuoteArg } from '../../../shared/shell-quote.js';
|
|
19
|
+
const CLAUDE_EXECUTORS = new Set(['claude', 'claude-sdk']);
|
|
20
|
+
const CODEX_EXECUTORS = new Set(['codex', 'codex-sdk']);
|
|
21
|
+
const OPENAI_API_EXECUTORS = new Set(['openai-api']);
|
|
22
|
+
const ANTHROPIC_API_EXECUTORS = new Set(['anthropic-api']);
|
|
14
23
|
function isDryRunReport(report) {
|
|
15
24
|
return Boolean(report && typeof report === 'object' && report.dryRun === true);
|
|
16
25
|
}
|
|
@@ -42,6 +51,25 @@ function applyGateExitCode(code, values, lang) {
|
|
|
42
51
|
process.stderr.write(tCli('cli.run.report_only_gate_skipped', lang));
|
|
43
52
|
return 0;
|
|
44
53
|
}
|
|
54
|
+
function userFacingPath(path) {
|
|
55
|
+
const rel = relative(process.cwd(), path);
|
|
56
|
+
if (rel && rel !== '..' && !rel.startsWith(`..${sep}`))
|
|
57
|
+
return rel;
|
|
58
|
+
return path;
|
|
59
|
+
}
|
|
60
|
+
function sampleCommandForSingleTreatment(treatment, skillDir) {
|
|
61
|
+
if (existsSync(treatment))
|
|
62
|
+
return `omk sample ${shellQuoteArg(treatment)}`;
|
|
63
|
+
const dirSkillPath = join(skillDir, treatment);
|
|
64
|
+
if (existsSync(join(dirSkillPath, 'SKILL.md'))) {
|
|
65
|
+
return `omk sample ${shellQuoteArg(userFacingPath(dirSkillPath))}`;
|
|
66
|
+
}
|
|
67
|
+
const flatSkillPath = join(skillDir, `${treatment}.md`);
|
|
68
|
+
if (existsSync(flatSkillPath)) {
|
|
69
|
+
return `omk sample ${shellQuoteArg(userFacingPath(flatSkillPath))}`;
|
|
70
|
+
}
|
|
71
|
+
return null;
|
|
72
|
+
}
|
|
45
73
|
/**
|
|
46
74
|
* 完整 report JSON 是**机器输出**:重定向 / 管道(`omk eval > r.json`、`| jq`)时吐到 stdout 供下游消费。
|
|
47
75
|
* 交互式 TTY 下报告已存盘、(默认)还起了 report server,再刷上千行 JSON 只会把 verdict 淹没在屏幕外 ——
|
|
@@ -62,11 +90,118 @@ function emitVerdictText(text) {
|
|
|
62
90
|
const stream = process.stdout.isTTY ? process.stdout : process.stderr;
|
|
63
91
|
stream.write(text + '\n');
|
|
64
92
|
}
|
|
93
|
+
function preflightFailedTarget(message) {
|
|
94
|
+
return message.match(/(^|\n)preflight failed \[([^\]]+)\]:/)?.[2] ?? null;
|
|
95
|
+
}
|
|
96
|
+
function runtimeKey(executor, model) {
|
|
97
|
+
return `${executor}:${model}`;
|
|
98
|
+
}
|
|
99
|
+
function matchesPreflightTarget(target, executor, model) {
|
|
100
|
+
return target.includes(':') ? target === runtimeKey(executor, model) : target === model;
|
|
101
|
+
}
|
|
102
|
+
function matchingPreflightRuntimes(config, failedTarget) {
|
|
103
|
+
const matches = [];
|
|
104
|
+
if (config.executorName && matchesPreflightTarget(failedTarget, config.executorName, config.model ?? '')) {
|
|
105
|
+
matches.push({ executor: config.executorName, role: 'task' });
|
|
106
|
+
}
|
|
107
|
+
if (!config.noJudge) {
|
|
108
|
+
for (const judge of config.judgeModels) {
|
|
109
|
+
if (matchesPreflightTarget(failedTarget, judge.executor, judge.model))
|
|
110
|
+
matches.push({ executor: judge.executor, role: 'judge' });
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
return matches;
|
|
114
|
+
}
|
|
115
|
+
function hasExecutor(matches, executors) {
|
|
116
|
+
return matches.some((match) => executors.has(match.executor));
|
|
117
|
+
}
|
|
118
|
+
function configHasExecutor(config, executors) {
|
|
119
|
+
if (config.executorName && executors.has(config.executorName))
|
|
120
|
+
return true;
|
|
121
|
+
return !config.noJudge && config.judgeModels.some((judge) => executors.has(judge.executor));
|
|
122
|
+
}
|
|
123
|
+
function shouldSwitchJudgeWithTask(config, executors) {
|
|
124
|
+
return !config.noJudge && config.judgeModels.length === 1 && executors.has(config.judgeModels[0].executor);
|
|
125
|
+
}
|
|
126
|
+
function fallbackFlags(matches, executor, taskModel, judgeModel, includeJudgeWithTask) {
|
|
127
|
+
const hasTask = matches.some((match) => match.role === 'task');
|
|
128
|
+
const hasJudge = matches.some((match) => match.role === 'judge') || (hasTask && includeJudgeWithTask);
|
|
129
|
+
return [
|
|
130
|
+
hasTask ? `--executor ${executor} --model ${taskModel}` : '',
|
|
131
|
+
hasJudge ? `--judge-models ${executor}:${judgeModel}` : '',
|
|
132
|
+
].filter(Boolean).join(' ');
|
|
133
|
+
}
|
|
134
|
+
export function formatConnectivityFailureHint(message, config, lang, env = process.env) {
|
|
135
|
+
const failedTarget = preflightFailedTarget(message);
|
|
136
|
+
if (failedTarget) {
|
|
137
|
+
const matches = matchingPreflightRuntimes(config, failedTarget);
|
|
138
|
+
if (matches.length === 0)
|
|
139
|
+
return '';
|
|
140
|
+
if (hasExecutor(matches, CLAUDE_EXECUTORS)) {
|
|
141
|
+
const codexModel = codexModelFlagValue(env);
|
|
142
|
+
return tCli('cli.run.codex_fallback_hint', lang, {
|
|
143
|
+
flags: fallbackFlags(matches, 'codex', codexModel, codexModel, shouldSwitchJudgeWithTask(config, CLAUDE_EXECUTORS)),
|
|
144
|
+
codexModelHint: codexModelHint(lang, env),
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
if (hasExecutor(matches, CODEX_EXECUTORS)) {
|
|
148
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
149
|
+
const codexModel = codexModelFlagValue(env);
|
|
150
|
+
return tCli('cli.run.codex_model_hint', lang, {
|
|
151
|
+
codexFlags: fallbackFlags(matches, 'codex', codexModel, codexModel, shouldSwitchJudgeWithTask(config, CODEX_EXECUTORS)),
|
|
152
|
+
codexExec: `codex exec -m ${codexModel} "hi"`,
|
|
153
|
+
claudeFlags: fallbackFlags(matches, 'claude', 'sonnet', 'haiku', shouldSwitchJudgeWithTask(config, CODEX_EXECUTORS)),
|
|
154
|
+
openaiFlags: fallbackFlags(matches, 'openai-api', '<openai-model>', '<openai-model>', shouldSwitchJudgeWithTask(config, CODEX_EXECUTORS)),
|
|
155
|
+
codexModelHint: codexModelHint(lang, env),
|
|
156
|
+
});
|
|
157
|
+
}
|
|
158
|
+
return tCli('cli.run.codex_auth_hint', lang, {
|
|
159
|
+
claudeFlags: fallbackFlags(matches, 'claude', 'sonnet', 'haiku', shouldSwitchJudgeWithTask(config, CODEX_EXECUTORS)),
|
|
160
|
+
openaiFlags: fallbackFlags(matches, 'openai-api', '<openai-model>', '<openai-model>', shouldSwitchJudgeWithTask(config, CODEX_EXECUTORS)),
|
|
161
|
+
});
|
|
162
|
+
}
|
|
163
|
+
if (hasExecutor(matches, OPENAI_API_EXECUTORS)) {
|
|
164
|
+
return tCli(looksLikeModelUnavailableFailure(message) ? 'cli.run.openai_api_model_hint' : 'cli.run.openai_api_auth_hint', lang);
|
|
165
|
+
}
|
|
166
|
+
if (hasExecutor(matches, ANTHROPIC_API_EXECUTORS)) {
|
|
167
|
+
return tCli(looksLikeModelUnavailableFailure(message) ? 'cli.run.anthropic_api_model_hint' : 'cli.run.anthropic_api_auth_hint', lang);
|
|
168
|
+
}
|
|
169
|
+
return '';
|
|
170
|
+
}
|
|
171
|
+
if (message.includes('OPENAI_API_KEY') && configHasExecutor(config, OPENAI_API_EXECUTORS))
|
|
172
|
+
return tCli('cli.run.openai_api_auth_hint', lang);
|
|
173
|
+
if (message.includes('ANTHROPIC_API_KEY') && configHasExecutor(config, ANTHROPIC_API_EXECUTORS))
|
|
174
|
+
return tCli('cli.run.anthropic_api_auth_hint', lang);
|
|
175
|
+
return '';
|
|
176
|
+
}
|
|
177
|
+
function treatmentVariantFromVerdict(result) {
|
|
178
|
+
if (result.representative?.treatment)
|
|
179
|
+
return result.representative.treatment;
|
|
180
|
+
if (result.level === 'SOLO')
|
|
181
|
+
return result.variants[0];
|
|
182
|
+
return result.variants[1];
|
|
183
|
+
}
|
|
184
|
+
function promoteTargetFromEvidence(result, records) {
|
|
185
|
+
const treatment = treatmentVariantFromVerdict(result);
|
|
186
|
+
const matched = treatment ? records.filter((r) => r.variant === treatment) : records;
|
|
187
|
+
const names = new Set(matched.filter((r) => r.bound).map((r) => r.name));
|
|
188
|
+
return names.size === 1 ? [...names][0] : undefined;
|
|
189
|
+
}
|
|
190
|
+
function emitRecordedEvidence(records, lang, verdict) {
|
|
191
|
+
for (const w of records) {
|
|
192
|
+
const command = `omk promote ${shellQuoteArg(w.name)}`;
|
|
193
|
+
const key = w.bound && verdict === 'PROGRESS'
|
|
194
|
+
? 'cli.run.evidence_recorded_promotable'
|
|
195
|
+
: (w.bound ? 'cli.run.evidence_recorded' : 'cli.run.evidence_recorded_unbound');
|
|
196
|
+
process.stderr.write(tCli(key, lang, { name: w.name, command }));
|
|
197
|
+
}
|
|
198
|
+
}
|
|
65
199
|
async function emitEvaluationVerdict(report, values, lang) {
|
|
66
200
|
const { computeVerdict, formatVerdictText } = await import('../../../eval-core/verdict.js');
|
|
67
201
|
const result = computeVerdict(report, verdictOptions(values));
|
|
68
|
-
|
|
69
|
-
|
|
202
|
+
const written = await recordEvidenceSafely(report, result.level, values);
|
|
203
|
+
emitVerdictText(formatVerdictText(result, { verbose: true, lang, promoteTarget: promoteTargetFromEvidence(result, written) }));
|
|
204
|
+
emitRecordedEvidence(written, lang, result.level);
|
|
70
205
|
return verdictPasses(result.level, result.headline) ? 0 : 1;
|
|
71
206
|
}
|
|
72
207
|
/**
|
|
@@ -74,18 +209,16 @@ async function emitEvaluationVerdict(report, values, lang) {
|
|
|
74
209
|
* 永不致命:管理是 eval 的旁路,写入失败 / 无匹配记录都不影响 verdict 与 exit code。
|
|
75
210
|
* `--no-evidence` 关闭。仅对实际写入的记录打一行提示(无匹配则全静默)。
|
|
76
211
|
*/
|
|
77
|
-
async function recordEvidenceSafely(report, verdict, values
|
|
212
|
+
async function recordEvidenceSafely(report, verdict, values) {
|
|
78
213
|
if (values['no-evidence'] === true)
|
|
79
|
-
return;
|
|
214
|
+
return [];
|
|
80
215
|
try {
|
|
81
216
|
const { recordEvalEvidence } = await import('../../../managed/index.js');
|
|
82
|
-
|
|
83
|
-
for (const w of written) {
|
|
84
|
-
process.stderr.write(tCli(w.bound ? 'cli.run.evidence_recorded' : 'cli.run.evidence_recorded_unbound', lang, { name: w.name }));
|
|
85
|
-
}
|
|
217
|
+
return recordEvalEvidence(report, verdict, new Date().toISOString());
|
|
86
218
|
}
|
|
87
219
|
catch {
|
|
88
220
|
// 证据写入是旁路,任何异常都不该让评测失败
|
|
221
|
+
return [];
|
|
89
222
|
}
|
|
90
223
|
}
|
|
91
224
|
function batchItemFallbackReport(batch, item) {
|
|
@@ -137,7 +270,7 @@ async function emitBatchVerdict(report, reportsDir, values, lang) {
|
|
|
137
270
|
// batch 每个子报告各自是一份独立 skill 的评测 → 各自写证据。
|
|
138
271
|
for (const child of childReports) {
|
|
139
272
|
const v = results.find((r) => r.id === child.id)?.verdict.level ?? 'SOLO';
|
|
140
|
-
await recordEvidenceSafely(child, v, values, lang);
|
|
273
|
+
emitRecordedEvidence(await recordEvidenceSafely(child, v, values), lang, v);
|
|
141
274
|
}
|
|
142
275
|
const passed = results.filter((r) => verdictPasses(r.verdict.level, r.verdict.headline)).length;
|
|
143
276
|
const failed = results.length - passed;
|
|
@@ -188,7 +321,7 @@ async function announceSavedReport({ report, filePath, reportsDir, values, lang,
|
|
|
188
321
|
}
|
|
189
322
|
}
|
|
190
323
|
async function runEval(_args, flags, lang) {
|
|
191
|
-
const { values, config, evalConfig } = parseRunConfig({ ...flags });
|
|
324
|
+
const { values, config, evalConfig } = parseRunConfig({ ...flags }, { lang });
|
|
192
325
|
if (!values.batch && !hasUsableSamplesPath(config.samplesPath)) {
|
|
193
326
|
const treatmentRaw = typeof values.treatment === 'string' ? values.treatment : '';
|
|
194
327
|
const treatments = treatmentRaw.split(',').map((v) => v.trim()).filter(Boolean);
|
|
@@ -201,8 +334,15 @@ async function runEval(_args, flags, lang) {
|
|
|
201
334
|
newPath: deprecatedSamplesHint.newPath,
|
|
202
335
|
}));
|
|
203
336
|
}
|
|
337
|
+
const sampleCommand = !deprecatedSamplesHint && !values.samples && !evalConfig?.samples && treatments.length === 1
|
|
338
|
+
? sampleCommandForSingleTreatment(treatments[0], config.skillDir)
|
|
339
|
+
: null;
|
|
340
|
+
const missingSamplesMessage = [
|
|
341
|
+
tCli('cli.common.samples_not_found', lang, { path: config.samplesPath }),
|
|
342
|
+
sampleCommand ? tCli('cli.common.samples_not_found_hint', lang, { command: sampleCommand }) : '',
|
|
343
|
+
].filter(Boolean).join('\n');
|
|
204
344
|
console.error(tCli('cli.common.error_prefix', lang, {
|
|
205
|
-
message:
|
|
345
|
+
message: missingSamplesMessage,
|
|
206
346
|
}));
|
|
207
347
|
throw new CliExit(1);
|
|
208
348
|
}
|
|
@@ -349,7 +489,10 @@ async function runEval(_args, flags, lang) {
|
|
|
349
489
|
catch (err) {
|
|
350
490
|
if (err instanceof CliExit)
|
|
351
491
|
throw err;
|
|
352
|
-
|
|
492
|
+
const message = err.message;
|
|
493
|
+
console.error(tCli('cli.common.error_prefix', lang, {
|
|
494
|
+
message: `${message}${formatConnectivityFailureHint(message, config, lang)}`,
|
|
495
|
+
}));
|
|
353
496
|
throw new CliExit(1);
|
|
354
497
|
}
|
|
355
498
|
}
|
|
@@ -416,14 +559,14 @@ export default class Eval extends BaseCommand {
|
|
|
416
559
|
}),
|
|
417
560
|
executor: Flags.string({
|
|
418
561
|
description: bilingual({
|
|
419
|
-
zh: '
|
|
420
|
-
en: 'Executor: claude / claude-sdk / codex / codex-sdk / openai-api / gemini / custom
|
|
562
|
+
zh: '执行器:claude / claude-sdk / codex / codex-sdk / openai-api / gemini / 自定义命令。Codex 任务内自动用 codex;也可用 OMK_EXECUTOR 设置环境偏好。',
|
|
563
|
+
en: 'Executor: claude / claude-sdk / codex / codex-sdk / openai-api / gemini / custom. Defaults to codex inside Codex tasks; OMK_EXECUTOR sets an environment preference.',
|
|
421
564
|
}),
|
|
422
565
|
}),
|
|
423
566
|
'judge-models': Flags.string({
|
|
424
567
|
description: bilingual({
|
|
425
|
-
zh: '评委配置,格式 executor:model[,...],例 claude:haiku 或
|
|
426
|
-
en: 'Judge config: executor:model[,...]
|
|
568
|
+
zh: '评委配置,格式 executor:model[,...],例 claude:haiku 或 codex:<model>(≥ 2 个 = ensemble)。默认跟随所选执行器;Codex 沿用被测模型。',
|
|
569
|
+
en: 'Judge config: executor:model[,...], e.g. claude:haiku or codex:<model> (≥ 2 = ensemble). Defaults to the selected executor; Codex reuses the evaluated model.',
|
|
427
570
|
}),
|
|
428
571
|
}),
|
|
429
572
|
'output-dir': Flags.string({
|
|
@@ -482,7 +625,10 @@ export default class Eval extends BaseCommand {
|
|
|
482
625
|
parse: integerStringParser('--retry', { min: 0 }),
|
|
483
626
|
}),
|
|
484
627
|
resume: Flags.string({
|
|
485
|
-
description: bilingual({
|
|
628
|
+
description: bilingual({
|
|
629
|
+
zh: '从契约兼容的报告恢复成功项;不兼容则从头运行',
|
|
630
|
+
en: 'Reuse successful entries from a contract-compatible report; otherwise start over',
|
|
631
|
+
}),
|
|
486
632
|
}),
|
|
487
633
|
'layered-stats': Flags.boolean({
|
|
488
634
|
description: bilingual({ zh: '输出分层统计', en: 'Emit layered stats' }),
|