oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -20,7 +20,7 @@ export interface ClaudeCliResponse {
|
|
|
20
20
|
usage?: TokenUsage;
|
|
21
21
|
total_cost_usd?: number;
|
|
22
22
|
result?: string;
|
|
23
|
-
stop_reason?: string;
|
|
23
|
+
stop_reason?: string | null;
|
|
24
24
|
num_turns?: number;
|
|
25
25
|
}
|
|
26
26
|
export interface OpenAiUsage {
|
|
@@ -34,7 +34,8 @@ export interface OpenAiResponse {
|
|
|
34
34
|
usage?: OpenAiUsage;
|
|
35
35
|
choices?: Array<{
|
|
36
36
|
message?: {
|
|
37
|
-
content?: string;
|
|
37
|
+
content?: string | null;
|
|
38
|
+
refusal?: string | null;
|
|
38
39
|
};
|
|
39
40
|
finish_reason?: string;
|
|
40
41
|
}>;
|
|
@@ -52,6 +53,7 @@ export interface GeminiResponse {
|
|
|
52
53
|
export interface AnthropicResponse {
|
|
53
54
|
usage?: TokenUsage;
|
|
54
55
|
content?: Array<{
|
|
56
|
+
type?: string;
|
|
55
57
|
text?: string;
|
|
56
58
|
}>;
|
|
57
59
|
stop_reason?: string;
|
|
@@ -99,6 +101,13 @@ export interface ClaudeSdkResultMessage extends ClaudeSdkBaseMessage {
|
|
|
99
101
|
duration_api_ms?: number;
|
|
100
102
|
duration_ms?: number;
|
|
101
103
|
num_turns?: number;
|
|
104
|
+
stop_reason?: string | null;
|
|
105
|
+
modelUsage?: Record<string, {
|
|
106
|
+
inputTokens?: number;
|
|
107
|
+
outputTokens?: number;
|
|
108
|
+
cacheReadInputTokens?: number;
|
|
109
|
+
cacheCreationInputTokens?: number;
|
|
110
|
+
}>;
|
|
102
111
|
subtype?: string;
|
|
103
112
|
errors?: string[];
|
|
104
113
|
}
|
|
@@ -130,18 +139,30 @@ export interface CodexEvent {
|
|
|
130
139
|
results?: unknown[];
|
|
131
140
|
changes?: Array<{
|
|
132
141
|
path?: string;
|
|
142
|
+
changeKind?: string;
|
|
133
143
|
}>;
|
|
134
144
|
server?: string;
|
|
135
145
|
tool?: string;
|
|
146
|
+
name?: string;
|
|
136
147
|
arguments?: unknown;
|
|
137
148
|
result?: unknown;
|
|
138
149
|
message?: string;
|
|
150
|
+
error?: {
|
|
151
|
+
message?: string;
|
|
152
|
+
};
|
|
139
153
|
};
|
|
140
154
|
error?: {
|
|
141
155
|
message?: string;
|
|
142
156
|
};
|
|
157
|
+
message?: string;
|
|
143
158
|
ts?: number;
|
|
144
159
|
}
|
|
160
|
+
/**
|
|
161
|
+
* Translate Codex's external event shape into omk's internal protocol model.
|
|
162
|
+
* Codex currently calls file-change discriminators `kind`; omk reserves bare
|
|
163
|
+
* `kind` for ArtifactKind, so the raw field is qualified at the boundary.
|
|
164
|
+
*/
|
|
165
|
+
export declare function normalizeCodexProtocolEvent(value: unknown): CodexEvent | null;
|
|
145
166
|
export interface ExecutorErrorLike {
|
|
146
167
|
message?: string;
|
|
147
168
|
name?: string;
|
|
@@ -151,9 +172,20 @@ export interface ExecutorErrorLike {
|
|
|
151
172
|
export declare function asErrorLike(err: unknown): ExecutorErrorLike;
|
|
152
173
|
export declare function errorMessage(err: unknown, fallback?: string): string;
|
|
153
174
|
export declare function parseJson<T>(content: string): T;
|
|
175
|
+
export interface JsonResponseBody<T> {
|
|
176
|
+
data: T | null;
|
|
177
|
+
rawBody: string;
|
|
178
|
+
}
|
|
179
|
+
export declare function readJsonResponse<T>(response: Response): Promise<JsonResponseBody<T>>;
|
|
180
|
+
export declare function responseBodyPreview(rawBody: string, maxLength?: number): string;
|
|
154
181
|
export declare function buildExecEnv(skillDir?: string | null): NodeJS.ProcessEnv;
|
|
155
182
|
export declare function timeoutExecResult(timeoutMs: number, durationMs: number): ExecResult;
|
|
156
183
|
export declare function interruptedExecResult(durationMs: number): ExecResult;
|
|
184
|
+
/**
|
|
185
|
+
* Register an in-process runtime (for example an SDK-owned child) with the
|
|
186
|
+
* same SIGINT coordinator used by spawned executors.
|
|
187
|
+
*/
|
|
188
|
+
export declare function registerSigintSubscriber(subscriber: () => void): () => void;
|
|
157
189
|
export declare function __resetSigintRegistryForTest(): void;
|
|
158
190
|
export interface SpawnHelperResult {
|
|
159
191
|
stdout: string;
|
|
@@ -176,7 +208,7 @@ export interface SpawnHelperOptions {
|
|
|
176
208
|
env?: NodeJS.ProcessEnv;
|
|
177
209
|
/** kill child after this many ms; reject with killedByTimeout=true */
|
|
178
210
|
timeoutMs?: number;
|
|
179
|
-
/** stdout
|
|
211
|
+
/** per-stream stdout/stderr byte limit; reject when either stream exceeds it */
|
|
180
212
|
maxBuffer?: number;
|
|
181
213
|
/** external abort signal; abort() 走跟 SIGINT 同一 grace 路径 */
|
|
182
214
|
abortSignal?: AbortSignal;
|
package/dist/executors/shared.js
CHANGED
|
@@ -24,6 +24,37 @@ const EXECUTOR_VENDOR = {
|
|
|
24
24
|
export function executorVendor(executor) {
|
|
25
25
|
return EXECUTOR_VENDOR[executor] ?? 'unknown';
|
|
26
26
|
}
|
|
27
|
+
/**
|
|
28
|
+
* Translate Codex's external event shape into omk's internal protocol model.
|
|
29
|
+
* Codex currently calls file-change discriminators `kind`; omk reserves bare
|
|
30
|
+
* `kind` for ArtifactKind, so the raw field is qualified at the boundary.
|
|
31
|
+
*/
|
|
32
|
+
export function normalizeCodexProtocolEvent(value) {
|
|
33
|
+
if (typeof value !== 'object' || value === null || Array.isArray(value))
|
|
34
|
+
return null;
|
|
35
|
+
const event = value;
|
|
36
|
+
const rawItem = event.item;
|
|
37
|
+
if (typeof rawItem !== 'object' || rawItem === null || Array.isArray(rawItem)) {
|
|
38
|
+
return event;
|
|
39
|
+
}
|
|
40
|
+
const item = rawItem;
|
|
41
|
+
const rawChanges = item.changes;
|
|
42
|
+
const normalizedItem = {
|
|
43
|
+
...item,
|
|
44
|
+
...(Array.isArray(rawChanges) && {
|
|
45
|
+
changes: rawChanges.flatMap((change) => {
|
|
46
|
+
if (typeof change !== 'object' || change === null || Array.isArray(change))
|
|
47
|
+
return [];
|
|
48
|
+
const rawChange = change;
|
|
49
|
+
return [{
|
|
50
|
+
...(typeof rawChange.path === 'string' && { path: rawChange.path }),
|
|
51
|
+
...(typeof rawChange.kind === 'string' && { changeKind: rawChange.kind }),
|
|
52
|
+
}];
|
|
53
|
+
}),
|
|
54
|
+
}),
|
|
55
|
+
};
|
|
56
|
+
return { ...event, item: normalizedItem };
|
|
57
|
+
}
|
|
27
58
|
export function asErrorLike(err) {
|
|
28
59
|
return typeof err === 'object' && err !== null ? err : {};
|
|
29
60
|
}
|
|
@@ -34,6 +65,25 @@ export function errorMessage(err, fallback = 'unknown error') {
|
|
|
34
65
|
export function parseJson(content) {
|
|
35
66
|
return JSON.parse(content);
|
|
36
67
|
}
|
|
68
|
+
export async function readJsonResponse(response) {
|
|
69
|
+
const rawBody = await response.text();
|
|
70
|
+
if (!rawBody.trim())
|
|
71
|
+
return { data: null, rawBody };
|
|
72
|
+
try {
|
|
73
|
+
return { data: JSON.parse(rawBody), rawBody };
|
|
74
|
+
}
|
|
75
|
+
catch {
|
|
76
|
+
return { data: null, rawBody };
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
export function responseBodyPreview(rawBody, maxLength = 500) {
|
|
80
|
+
const normalized = rawBody.replace(/\s+/g, ' ').trim();
|
|
81
|
+
if (!normalized)
|
|
82
|
+
return '';
|
|
83
|
+
return normalized.length > maxLength
|
|
84
|
+
? `${normalized.slice(0, maxLength)}...`
|
|
85
|
+
: normalized;
|
|
86
|
+
}
|
|
37
87
|
export function buildExecEnv(skillDir) {
|
|
38
88
|
const proxyUrl = process.env.CCV_PROXY_URL || undefined;
|
|
39
89
|
const env = proxyUrl
|
|
@@ -57,7 +107,9 @@ export function timeoutExecResult(timeoutMs, durationMs) {
|
|
|
57
107
|
outputTokens: 0,
|
|
58
108
|
cacheReadTokens: 0,
|
|
59
109
|
cacheCreationTokens: 0,
|
|
110
|
+
tokenUsageReportedByExecutor: false,
|
|
60
111
|
costUSD: 0,
|
|
112
|
+
costReportedByExecutor: false,
|
|
61
113
|
output: null,
|
|
62
114
|
stopReason: 'timeout',
|
|
63
115
|
numTurns: 0,
|
|
@@ -76,7 +128,9 @@ export function interruptedExecResult(durationMs) {
|
|
|
76
128
|
outputTokens: 0,
|
|
77
129
|
cacheReadTokens: 0,
|
|
78
130
|
cacheCreationTokens: 0,
|
|
131
|
+
tokenUsageReportedByExecutor: false,
|
|
79
132
|
costUSD: 0,
|
|
133
|
+
costReportedByExecutor: false,
|
|
80
134
|
output: null,
|
|
81
135
|
stopReason: 'interrupted',
|
|
82
136
|
numTurns: 0,
|
|
@@ -97,6 +151,7 @@ export function interruptedExecResult(durationMs) {
|
|
|
97
151
|
// 立即退出(软关 → 硬关两段式)
|
|
98
152
|
// - 同时支持 timeout 跟 abortSignal 两条 kill 路径,跟 SIGINT 共用 grace 逻辑
|
|
99
153
|
const activeChildren = new Set();
|
|
154
|
+
const sigintSubscribers = new Set();
|
|
100
155
|
let sigintListenerInstalled = false;
|
|
101
156
|
let shuttingDown = false;
|
|
102
157
|
const SIGTERM_GRACE_MS = 500;
|
|
@@ -104,6 +159,12 @@ function broadcastShutdown() {
|
|
|
104
159
|
if (shuttingDown)
|
|
105
160
|
return;
|
|
106
161
|
shuttingDown = true;
|
|
162
|
+
for (const subscriber of sigintSubscribers) {
|
|
163
|
+
try {
|
|
164
|
+
subscriber();
|
|
165
|
+
}
|
|
166
|
+
catch { /* cancellation handlers must not block shutdown */ }
|
|
167
|
+
}
|
|
107
168
|
for (const child of activeChildren) {
|
|
108
169
|
try {
|
|
109
170
|
child.kill('SIGTERM');
|
|
@@ -119,7 +180,14 @@ function broadcastShutdown() {
|
|
|
119
180
|
}
|
|
120
181
|
// 卸载自己,re-raise SIGINT 让 host listener / default action(exit code 130)接管
|
|
121
182
|
process.removeListener('SIGINT', sigintHandler);
|
|
183
|
+
sigintListenerInstalled = false;
|
|
122
184
|
process.kill(process.pid, 'SIGINT');
|
|
185
|
+
// A host may intentionally intercept the re-raised signal. In that case the
|
|
186
|
+
// process remains usable and a later child/SDK registration must reinstall
|
|
187
|
+
// a functional coordinator instead of inheriting a permanently latched flag.
|
|
188
|
+
setImmediate(() => {
|
|
189
|
+
shuttingDown = false;
|
|
190
|
+
}).unref();
|
|
123
191
|
}
|
|
124
192
|
function sigintHandler() {
|
|
125
193
|
broadcastShutdown();
|
|
@@ -130,9 +198,21 @@ function ensureSigintListener() {
|
|
|
130
198
|
sigintListenerInstalled = true;
|
|
131
199
|
process.on('SIGINT', sigintHandler);
|
|
132
200
|
}
|
|
201
|
+
/**
|
|
202
|
+
* Register an in-process runtime (for example an SDK-owned child) with the
|
|
203
|
+
* same SIGINT coordinator used by spawned executors.
|
|
204
|
+
*/
|
|
205
|
+
export function registerSigintSubscriber(subscriber) {
|
|
206
|
+
ensureSigintListener();
|
|
207
|
+
sigintSubscribers.add(subscriber);
|
|
208
|
+
return () => {
|
|
209
|
+
sigintSubscribers.delete(subscriber);
|
|
210
|
+
};
|
|
211
|
+
}
|
|
133
212
|
// test-only:重置模块级状态,让 vitest 之间互不污染
|
|
134
213
|
export function __resetSigintRegistryForTest() {
|
|
135
214
|
activeChildren.clear();
|
|
215
|
+
sigintSubscribers.clear();
|
|
136
216
|
if (sigintListenerInstalled) {
|
|
137
217
|
process.removeListener('SIGINT', sigintHandler);
|
|
138
218
|
sigintListenerInstalled = false;
|
|
@@ -157,7 +237,9 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
|
|
|
157
237
|
activeChildren.add(child);
|
|
158
238
|
let stdout = '';
|
|
159
239
|
let stderr = '';
|
|
160
|
-
let
|
|
240
|
+
let stdoutBytes = 0;
|
|
241
|
+
let stderrBytes = 0;
|
|
242
|
+
let bufferOverflowStream = null;
|
|
161
243
|
let killedByTimeout = false;
|
|
162
244
|
let killedBySignalReason = null;
|
|
163
245
|
let graceTimer = null;
|
|
@@ -192,16 +274,31 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
|
|
|
192
274
|
if (abortSignal) {
|
|
193
275
|
abortListener = () => killWithGrace('abort');
|
|
194
276
|
abortSignal.addEventListener('abort', abortListener, { once: true });
|
|
277
|
+
if (abortSignal.aborted) {
|
|
278
|
+
// Defer until `done` has installed the child close/error listeners below.
|
|
279
|
+
queueMicrotask(abortListener);
|
|
280
|
+
}
|
|
195
281
|
}
|
|
196
282
|
child.stdout?.on('data', (chunk) => {
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
283
|
+
if (bufferOverflowStream)
|
|
284
|
+
return;
|
|
285
|
+
stdoutBytes += chunk.byteLength;
|
|
286
|
+
if (stdoutBytes > maxBuffer) {
|
|
287
|
+
bufferOverflowStream = 'stdout';
|
|
201
288
|
killWithGrace('buffer');
|
|
289
|
+
return;
|
|
202
290
|
}
|
|
291
|
+
stdout += chunk.toString();
|
|
203
292
|
});
|
|
204
293
|
child.stderr?.on('data', (chunk) => {
|
|
294
|
+
if (bufferOverflowStream)
|
|
295
|
+
return;
|
|
296
|
+
stderrBytes += chunk.byteLength;
|
|
297
|
+
if (stderrBytes > maxBuffer) {
|
|
298
|
+
bufferOverflowStream = 'stderr';
|
|
299
|
+
killWithGrace('buffer');
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
205
302
|
stderr += chunk.toString();
|
|
206
303
|
});
|
|
207
304
|
function cleanup() {
|
|
@@ -230,25 +327,26 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
|
|
|
230
327
|
killedByTimeout,
|
|
231
328
|
killedBySignal: killedSig,
|
|
232
329
|
};
|
|
233
|
-
//
|
|
234
|
-
if
|
|
235
|
-
|
|
330
|
+
// Buffer overflow is authoritative: truncated output is not valid evidence,
|
|
331
|
+
// even if the child catches SIGTERM and later exits 0.
|
|
332
|
+
if (bufferOverflowStream) {
|
|
333
|
+
const e = Object.assign(new Error(`${bufferOverflowStream} maxBuffer (${maxBuffer} bytes) exceeded`), result);
|
|
236
334
|
reject(e);
|
|
237
335
|
return;
|
|
238
336
|
}
|
|
239
|
-
//
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
if (code === 0) {
|
|
243
|
-
resolve(result);
|
|
244
|
-
return;
|
|
245
|
-
}
|
|
337
|
+
// Deadline / cancellation are caller-side facts. A child may catch
|
|
338
|
+
// SIGTERM, flush partial output and exit 0, but that cannot retroactively
|
|
339
|
+
// turn an over-budget or cancelled evaluation into a successful sample.
|
|
246
340
|
if (killedByTimeout) {
|
|
247
341
|
const tSec = timeoutMs ? (timeoutMs / 1000).toFixed(0) : '?';
|
|
248
342
|
const e = Object.assign(new Error(`execution timed out after ${tSec}s`), result);
|
|
249
343
|
reject(e);
|
|
250
344
|
return;
|
|
251
345
|
}
|
|
346
|
+
if (code === 0) {
|
|
347
|
+
resolve(result);
|
|
348
|
+
return;
|
|
349
|
+
}
|
|
252
350
|
if (killedSig) {
|
|
253
351
|
const e = Object.assign(new Error(`execution interrupted by signal ${killedSig}`), result);
|
|
254
352
|
reject(e);
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { Assertion, AssertionResults, ExecutorFn, Sample, ToolCallInfo } from '../types/index.js';
|
|
2
|
-
export
|
|
2
|
+
export { ASYNC_ASSERTION_TYPES } from '../shared/assertion-types.js';
|
|
3
3
|
export interface AsyncAssertionContext {
|
|
4
4
|
executor: ExecutorFn;
|
|
5
5
|
judgeModel: string;
|
|
@@ -2,18 +2,12 @@ import { resolve } from 'node:path';
|
|
|
2
2
|
import _Ajv from 'ajv';
|
|
3
3
|
import { ASSERTION_LAYER } from './layered-scores.js';
|
|
4
4
|
import { buildSemanticSimilarityPrompt, SEMANTIC_SIMILARITY_SYSTEM, buildRagJudgePrompt } from '../shared/llm-prompts/judge-prompts.js';
|
|
5
|
+
import { assertionContractValidationError, } from '../shared/sample-contract.js';
|
|
6
|
+
import { ASYNC_ASSERTION_TYPES, SYNC_ASSERTION_TYPES, } from '../shared/assertion-types.js';
|
|
7
|
+
export { ASYNC_ASSERTION_TYPES } from '../shared/assertion-types.js';
|
|
5
8
|
const Ajv = _Ajv.default ?? _Ajv;
|
|
6
9
|
const ajv = new Ajv();
|
|
7
10
|
const CUSTOM_ASSERTION_TIMEOUT_MS = 30_000;
|
|
8
|
-
export const ASYNC_ASSERTION_TYPES = new Set([
|
|
9
|
-
'semantic_similarity',
|
|
10
|
-
'custom',
|
|
11
|
-
// RAG-specific judge metrics. All three go through the LLM-judge path
|
|
12
|
-
// and inherit the same length-debias instruction as the main rubric judge.
|
|
13
|
-
'faithfulness',
|
|
14
|
-
'answer_relevancy',
|
|
15
|
-
'context_recall',
|
|
16
|
-
]);
|
|
17
11
|
function getErrorMessage(err) {
|
|
18
12
|
return err instanceof Error ? err.message : String(err);
|
|
19
13
|
}
|
|
@@ -293,6 +287,14 @@ function resolveAssertSetLayer(assertion) {
|
|
|
293
287
|
return only === 'unknown' ? undefined : only;
|
|
294
288
|
}
|
|
295
289
|
export function runAssertions(output, assertions, context = {}) {
|
|
290
|
+
assertions.forEach((assertion, index) => {
|
|
291
|
+
const error = assertionContractValidationError(assertion);
|
|
292
|
+
if (error)
|
|
293
|
+
throw new TypeError(`assertions[${index}]: ${error}`);
|
|
294
|
+
if (!SYNC_ASSERTION_TYPES.has(assertion.type)) {
|
|
295
|
+
throw new TypeError(`assertions[${index}]: async assertion ${JSON.stringify(assertion.type)} requires runAsyncAssertions()`);
|
|
296
|
+
}
|
|
297
|
+
});
|
|
296
298
|
const outputLower = output.toLowerCase();
|
|
297
299
|
const toolCalls = context.toolCalls || [];
|
|
298
300
|
const toolNames = toolCalls.map((tc) => tc.tool.toLowerCase());
|
|
@@ -324,6 +326,14 @@ export function runAssertions(output, assertions, context = {}) {
|
|
|
324
326
|
};
|
|
325
327
|
}
|
|
326
328
|
export async function runAsyncAssertions(output, assertions, { executor, judgeModel, sample, samplesDir }) {
|
|
329
|
+
assertions.forEach((assertion, index) => {
|
|
330
|
+
const error = assertionContractValidationError(assertion);
|
|
331
|
+
if (error)
|
|
332
|
+
throw new TypeError(`assertions[${index}]: ${error}`);
|
|
333
|
+
if (!ASYNC_ASSERTION_TYPES.has(assertion.type)) {
|
|
334
|
+
throw new TypeError(`assertions[${index}]: sync assertion ${JSON.stringify(assertion.type)} requires runAssertions()`);
|
|
335
|
+
}
|
|
336
|
+
});
|
|
327
337
|
const details = [];
|
|
328
338
|
let asyncCostUSD = 0;
|
|
329
339
|
let anyCostUnreported = false;
|
|
@@ -9,10 +9,17 @@
|
|
|
9
9
|
*
|
|
10
10
|
* 触发条件:sample 至少有 1 条 failed assertion(全过的 sample 不需要诊断,省成本)。
|
|
11
11
|
*
|
|
12
|
-
*
|
|
12
|
+
* 运行目标跟随首位 judge;没有 judge 配置时跟随被测 executor/model。
|
|
13
|
+
* provider-specific 的廉价默认值只在 CLI runtime resolution 中决定。
|
|
13
14
|
*/
|
|
14
|
-
import type { ExecutorFn, ToolCallInfo, TurnInfo, Sample } from '../types/index.js';
|
|
15
|
+
import type { ExecutorFn, ToolCallInfo, TurnInfo, JudgeConfig, Sample } from '../types/index.js';
|
|
15
16
|
import type { AssertionDetail, DiagnosticResult, WorkflowCheck, FailureMode } from '../types/judge.js';
|
|
17
|
+
export interface DiagnosticTarget {
|
|
18
|
+
executor: string;
|
|
19
|
+
model: string;
|
|
20
|
+
}
|
|
21
|
+
export declare function resolveDiagnosticTarget(judgeModels: JudgeConfig[], mainExecutor: string, mainModel: string): DiagnosticTarget;
|
|
22
|
+
export declare function getDiagnosticPromptHash(): string;
|
|
16
23
|
interface RunDiagnosticOptions {
|
|
17
24
|
sample: Sample;
|
|
18
25
|
skillContent: string | null;
|
|
@@ -9,9 +9,12 @@
|
|
|
9
9
|
*
|
|
10
10
|
* 触发条件:sample 至少有 1 条 failed assertion(全过的 sample 不需要诊断,省成本)。
|
|
11
11
|
*
|
|
12
|
-
*
|
|
12
|
+
* 运行目标跟随首位 judge;没有 judge 配置时跟随被测 executor/model。
|
|
13
|
+
* provider-specific 的廉价默认值只在 CLI runtime resolution 中决定。
|
|
13
14
|
*/
|
|
15
|
+
import { createHash } from 'node:crypto';
|
|
14
16
|
import { FAILURE_MODES } from '../types/judge.js';
|
|
17
|
+
import { toolCallStatus } from '../shared/tool-call-status.js';
|
|
15
18
|
const SYSTEM_PROMPT = `你是 skill 评测诊断助手。基于失败用例的 expected/actual 差异,给 skill 作者具体可操作的改进建议。
|
|
16
19
|
|
|
17
20
|
**诱错样本(tripwire)规则 — 只看 sample 的显式标记,不要自己识别**:
|
|
@@ -65,6 +68,15 @@ ${FAILURE_MODES.map((m, i) => ` ${i + 1}. ${m}`).join('\n')}
|
|
|
65
68
|
全过的 sample 给空数组 []。
|
|
66
69
|
|
|
67
70
|
只返回 JSON,不要其他内容。`;
|
|
71
|
+
export function resolveDiagnosticTarget(judgeModels, mainExecutor, mainModel) {
|
|
72
|
+
const primaryJudge = judgeModels[0];
|
|
73
|
+
return primaryJudge
|
|
74
|
+
? { executor: primaryJudge.executor, model: primaryJudge.model }
|
|
75
|
+
: { executor: mainExecutor, model: mainModel };
|
|
76
|
+
}
|
|
77
|
+
export function getDiagnosticPromptHash() {
|
|
78
|
+
return createHash('sha256').update(SYSTEM_PROMPT).digest('hex').slice(0, 12);
|
|
79
|
+
}
|
|
68
80
|
const TOOL_INPUT_PREVIEW_MAX = 350;
|
|
69
81
|
const SKILL_CONTENT_MAX = 12000;
|
|
70
82
|
const FULL_OUTPUT_MAX = 1500;
|
|
@@ -83,7 +95,11 @@ function previewToolCall(tc, idx) {
|
|
|
83
95
|
const truncated = inputRepr.length > TOOL_INPUT_PREVIEW_MAX
|
|
84
96
|
? inputRepr.slice(0, TOOL_INPUT_PREVIEW_MAX) + '…'
|
|
85
97
|
: inputRepr;
|
|
86
|
-
const
|
|
98
|
+
const resultStatus = toolCallStatus(tc);
|
|
99
|
+
const status = resultStatus === 'failure'
|
|
100
|
+
? ' [失败]'
|
|
101
|
+
: resultStatus === 'cancelled' ? ' [取消]'
|
|
102
|
+
: resultStatus === 'unknown' ? ' [状态未知]' : '';
|
|
87
103
|
return `[${idx + 1}] ${tc.tool}${status} → ${truncated}`;
|
|
88
104
|
}
|
|
89
105
|
function formatExpectedAssertion(a) {
|
|
@@ -327,6 +343,9 @@ export async function runDiagnostic(opts) {
|
|
|
327
343
|
timeoutMs: opts.timeoutMs ?? 180_000,
|
|
328
344
|
lean: true, // 诊断也是纯文本生成,不需要工具循环
|
|
329
345
|
});
|
|
346
|
+
const costReporting = result.costReportedByExecutor === false
|
|
347
|
+
? { costReportedByExecutor: false }
|
|
348
|
+
: {};
|
|
330
349
|
if (!result.ok) {
|
|
331
350
|
return {
|
|
332
351
|
ok: false,
|
|
@@ -337,6 +356,7 @@ export async function runDiagnostic(opts) {
|
|
|
337
356
|
rootCause: [],
|
|
338
357
|
suggestion: { skill: '', sample: '', none: '' },
|
|
339
358
|
costUSD: result.costUSD,
|
|
359
|
+
...costReporting,
|
|
340
360
|
};
|
|
341
361
|
}
|
|
342
362
|
const raw = (result.output || '').trim();
|
|
@@ -375,6 +395,7 @@ export async function runDiagnostic(opts) {
|
|
|
375
395
|
failureModes: salvaged.failureModes ?? [],
|
|
376
396
|
suggestion: salvaged.suggestion ?? { skill: '', sample: '', none: '' },
|
|
377
397
|
costUSD: result.costUSD,
|
|
398
|
+
...costReporting,
|
|
378
399
|
};
|
|
379
400
|
}
|
|
380
401
|
return {
|
|
@@ -386,6 +407,7 @@ export async function runDiagnostic(opts) {
|
|
|
386
407
|
rootCause: [],
|
|
387
408
|
suggestion: { skill: '', sample: '', none: '' },
|
|
388
409
|
costUSD: result.costUSD,
|
|
410
|
+
...costReporting,
|
|
389
411
|
};
|
|
390
412
|
}
|
|
391
413
|
const obj = parsed;
|
|
@@ -410,6 +432,7 @@ export async function runDiagnostic(opts) {
|
|
|
410
432
|
none: String(sug.none ?? ''),
|
|
411
433
|
},
|
|
412
434
|
costUSD: result.costUSD,
|
|
435
|
+
...costReporting,
|
|
413
436
|
};
|
|
414
437
|
}
|
|
415
438
|
function parseWorkflowChecks(raw) {
|
package/dist/grading/index.js
CHANGED
|
@@ -2,12 +2,18 @@
|
|
|
2
2
|
* Mixed grading: deterministic assertions + LLM judge + multi-dimensional scoring.
|
|
3
3
|
*/
|
|
4
4
|
import { ASYNC_ASSERTION_TYPES, ratioToScore, runAssertions, runAsyncAssertions } from './assertions.js';
|
|
5
|
+
import { setOwnRecordValue } from '../shared/record-count.js';
|
|
5
6
|
import { buildTraceSummary, llmJudgeEnsemble, llmJudgeRepeat } from './judge.js';
|
|
6
7
|
import { computeLayeredScores } from './layered-scores.js';
|
|
8
|
+
import { ownRecordValue } from '../shared/record-count.js';
|
|
9
|
+
import { sampleContractValidationError } from '../shared/sample-contract.js';
|
|
7
10
|
/**
|
|
8
11
|
* Grade a model output against a sample's criteria.
|
|
9
12
|
*/
|
|
10
13
|
export async function grade({ output, sample, judgeModels, judgeExecutors, allowLlmJudge = true, execMetrics = {}, samplesDir = '.', judgeRepeat = 1, lengthDebias = true }) {
|
|
14
|
+
const sampleError = sampleContractValidationError(sample);
|
|
15
|
+
if (sampleError)
|
|
16
|
+
throw new TypeError(`grade(): invalid sample contract: ${sampleError}`);
|
|
11
17
|
if (!judgeModels || judgeModels.length === 0) {
|
|
12
18
|
throw new Error('grade(): judgeModels must be non-empty');
|
|
13
19
|
}
|
|
@@ -20,14 +26,14 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
|
|
|
20
26
|
// code path that needs the executor. Sync-only samples never invoke these
|
|
21
27
|
// helpers, so they can pass `judgeExecutors: {}` (or omit it entirely).
|
|
22
28
|
const requirePrimaryExecutor = () => {
|
|
23
|
-
const exec = judgeExecutorMap
|
|
29
|
+
const exec = ownRecordValue(judgeExecutorMap, primaryJudge.executor);
|
|
24
30
|
if (!exec) {
|
|
25
31
|
throw new Error(`grade(): judgeExecutors missing entry for primary judge "${primaryJudge.executor}"`);
|
|
26
32
|
}
|
|
27
33
|
return exec;
|
|
28
34
|
};
|
|
29
35
|
const executorByName = (name) => {
|
|
30
|
-
const exec = judgeExecutorMap
|
|
36
|
+
const exec = ownRecordValue(judgeExecutorMap, name);
|
|
31
37
|
if (!exec) {
|
|
32
38
|
throw new Error(`No executor registered for "${name}"; pipeline must populate judgeExecutors for every judge`);
|
|
33
39
|
}
|
|
@@ -86,9 +92,9 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
|
|
|
86
92
|
results.dimensions = {};
|
|
87
93
|
for (const [dim, rubric] of Object.entries(sample.dimensions)) {
|
|
88
94
|
const dimOptions = { output, rubric, prompt: sample.prompt, executor: requirePrimaryExecutor(), model: judgeModel, traceSummary, lengthDebias };
|
|
89
|
-
results.dimensions
|
|
95
|
+
setOwnRecordValue(results.dimensions, dim, useEnsemble
|
|
90
96
|
? await llmJudgeEnsemble(dimOptions, judgeModels, executorByName, judgeRepeat)
|
|
91
|
-
: await llmJudgeRepeat(dimOptions, judgeRepeat);
|
|
97
|
+
: await llmJudgeRepeat(dimOptions, judgeRepeat));
|
|
92
98
|
}
|
|
93
99
|
const dimValues = Object.values(results.dimensions);
|
|
94
100
|
const dimScores = dimValues.map((d) => d.score).filter((s) => s > 0);
|
package/dist/grading/judge.js
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { buildJudgePrompt, JUDGE_SYSTEM_PROMPT } from '../shared/llm-prompts/judge-prompts.js';
|
|
2
|
+
import { isToolCallCancelled, isToolCallFailure, isToolCallSuccess, isToolCallUnknown, toolCallStatus, } from '../shared/tool-call-status.js';
|
|
3
|
+
import { incrementRecordCount } from '../shared/record-count.js';
|
|
2
4
|
// 评分类 prompt 已收口到 shared/llm-prompts/judge-prompts.ts(单一来源 + prompt-registry 冻结)。
|
|
3
5
|
// 这里 re-export 保对外 API 不破:既有消费方仍从 grading/judge.js import 这两个符号。
|
|
4
6
|
export { buildJudgePrompt, getJudgePromptHash } from '../shared/llm-prompts/judge-prompts.js';
|
|
@@ -80,17 +82,23 @@ export function buildTraceSummary(turns, toolCalls) {
|
|
|
80
82
|
const lines = [];
|
|
81
83
|
if (toolCalls && toolCalls.length > 0) {
|
|
82
84
|
lines.push(`共调用 ${toolCalls.length} 个工具:`);
|
|
83
|
-
const successCount = toolCalls.filter(
|
|
84
|
-
const failureCount = toolCalls.length
|
|
85
|
+
const successCount = toolCalls.filter(isToolCallSuccess).length;
|
|
86
|
+
const failureCount = toolCalls.filter(isToolCallFailure).length;
|
|
87
|
+
const cancelledCount = toolCalls.filter(isToolCallCancelled).length;
|
|
88
|
+
const unknownCount = toolCalls.filter(isToolCallUnknown).length;
|
|
85
89
|
lines.push(` 成功 ${successCount}/${toolCalls.length}`);
|
|
86
90
|
if (failureCount > 0)
|
|
87
91
|
lines.push(` 失败 ${failureCount}/${toolCalls.length}`);
|
|
92
|
+
if (cancelledCount > 0)
|
|
93
|
+
lines.push(` 取消 ${cancelledCount}/${toolCalls.length}`);
|
|
94
|
+
if (unknownCount > 0)
|
|
95
|
+
lines.push(` 状态未知 ${unknownCount}/${toolCalls.length}`);
|
|
88
96
|
const dist = {};
|
|
89
97
|
for (const tc of toolCalls) {
|
|
90
|
-
dist
|
|
98
|
+
incrementRecordCount(dist, tc.tool);
|
|
91
99
|
}
|
|
92
100
|
lines.push(` 工具分布:${Object.entries(dist).map(([k, v]) => `${k}(${v})`).join(', ')}`);
|
|
93
|
-
const failedTools = toolCalls.filter(
|
|
101
|
+
const failedTools = toolCalls.filter(isToolCallFailure).map((tc) => tc.tool);
|
|
94
102
|
if (failedTools.length > 0) {
|
|
95
103
|
lines.push(` 失败工具:${[...new Set(failedTools)].join(', ')}`);
|
|
96
104
|
}
|
|
@@ -100,7 +108,11 @@ export function buildTraceSummary(turns, toolCalls) {
|
|
|
100
108
|
const detailCap = Math.min(toolCalls.length, TOOL_DETAIL_MAX_CALLS);
|
|
101
109
|
for (let i = 0; i < detailCap; i++) {
|
|
102
110
|
const tc = toolCalls[i];
|
|
103
|
-
const
|
|
111
|
+
const resultStatus = toolCallStatus(tc);
|
|
112
|
+
const status = resultStatus === 'failure'
|
|
113
|
+
? ' [失败]'
|
|
114
|
+
: resultStatus === 'cancelled' ? ' [取消]'
|
|
115
|
+
: resultStatus === 'unknown' ? ' [状态未知]' : '';
|
|
104
116
|
lines.push(` [${i + 1}] ${tc.tool}${status} → ${previewToolInput(tc)}`);
|
|
105
117
|
}
|
|
106
118
|
if (toolCalls.length > TOOL_DETAIL_MAX_CALLS) {
|
|
@@ -110,9 +122,10 @@ export function buildTraceSummary(turns, toolCalls) {
|
|
|
110
122
|
if (turns && turns.length > 0) {
|
|
111
123
|
lines.push('');
|
|
112
124
|
lines.push('执行轨迹摘要:');
|
|
125
|
+
const userTurns = turns.filter((turn) => turn.role === 'user').length;
|
|
113
126
|
const assistantTurns = turns.filter((turn) => turn.role === 'assistant').length;
|
|
114
127
|
const toolTurns = turns.filter((turn) => turn.role === 'tool').length;
|
|
115
|
-
lines.push(` 共 ${turns.length} 步(assistant ${assistantTurns} / tool ${toolTurns})`);
|
|
128
|
+
lines.push(` 共 ${turns.length} 步(user ${userTurns} / assistant ${assistantTurns} / tool ${toolTurns})`);
|
|
116
129
|
const maxTurns = Math.min(turns.length, 10);
|
|
117
130
|
for (let i = 0; i < maxTurns; i++) {
|
|
118
131
|
const t = turns[i];
|
|
@@ -9,9 +9,8 @@ import type { AssertionDetail, LayeredScores } from '../types/index.js';
|
|
|
9
9
|
* 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
|
|
10
10
|
* bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
|
|
11
11
|
* 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
|
|
12
|
-
* 解析(同层→归层、混层→不计),结果落 detail.layer
|
|
13
|
-
*
|
|
14
|
-
* 映射、又不是已知组合器,即 CI 失败。
|
|
12
|
+
* 解析(同层→归层、混层→不计),结果落 detail.layer。共享注册表 `shared/assertion-types.ts` 是支持类型的真源;
|
|
13
|
+
* `test/grading/layered-scores-exhaustiveness.test.ts` 守住新增类型必须在本映射或组合器集合中显式处理。
|
|
15
14
|
*/
|
|
16
15
|
export declare const ASSERTION_LAYER: Record<string, 'fact' | 'behavior'>;
|
|
17
16
|
interface CompositeInput {
|
|
@@ -8,9 +8,8 @@
|
|
|
8
8
|
* 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
|
|
9
9
|
* bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
|
|
10
10
|
* 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
|
|
11
|
-
* 解析(同层→归层、混层→不计),结果落 detail.layer
|
|
12
|
-
*
|
|
13
|
-
* 映射、又不是已知组合器,即 CI 失败。
|
|
11
|
+
* 解析(同层→归层、混层→不计),结果落 detail.layer。共享注册表 `shared/assertion-types.ts` 是支持类型的真源;
|
|
12
|
+
* `test/grading/layered-scores-exhaustiveness.test.ts` 守住新增类型必须在本映射或组合器集合中显式处理。
|
|
14
13
|
*/
|
|
15
14
|
export const ASSERTION_LAYER = {
|
|
16
15
|
contains: 'fact',
|
|
@@ -1,5 +1,4 @@
|
|
|
1
|
-
import type { Sample } from '../types/index.js';
|
|
2
|
-
import type { DependencyRequirements } from '../eval-core/dependency-checker.js';
|
|
1
|
+
import type { DependencyRequirements, Sample } from '../types/index.js';
|
|
3
2
|
export declare function parseYaml(text: string): unknown;
|
|
4
3
|
export interface LoadSamplesResult {
|
|
5
4
|
samples: Sample[];
|