oh-my-knowledge 0.48.0 → 0.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/README.md +50 -18
  2. package/README.zh.md +55 -23
  3. package/dist/analysis/coverage-analyzer.d.ts +1 -0
  4. package/dist/analysis/coverage-analyzer.js +125 -62
  5. package/dist/analysis/failure-clusterer.js +2 -1
  6. package/dist/analysis/gap-analyzer.d.ts +2 -2
  7. package/dist/analysis/gap-analyzer.js +13 -3
  8. package/dist/analysis/hedging-classifier.d.ts +2 -2
  9. package/dist/analysis/hedging-classifier.js +3 -4
  10. package/dist/analysis/report-diagnostics.js +9 -7
  11. package/dist/analysis/sample-diagnostics.js +6 -6
  12. package/dist/artifact-graph/doctor.js +15 -7
  13. package/dist/assets/agent-skills/omk/SKILL.md +27 -7
  14. package/dist/assets/agent-skills/omk/references/commands.md +18 -17
  15. package/dist/authoring/evolver.d.ts +10 -6
  16. package/dist/authoring/evolver.js +496 -83
  17. package/dist/authoring/generator.d.ts +3 -3
  18. package/dist/authoring/generator.js +5 -10
  19. package/dist/authoring/sample-fixer.d.ts +8 -6
  20. package/dist/authoring/sample-fixer.js +76 -5
  21. package/dist/cli/commands/doctor.js +31 -14
  22. package/dist/cli/commands/eval/index.d.ts +3 -0
  23. package/dist/cli/commands/eval/index.js +163 -17
  24. package/dist/cli/commands/evolve.d.ts +4 -4
  25. package/dist/cli/commands/evolve.js +27 -13
  26. package/dist/cli/commands/init.js +16 -3
  27. package/dist/cli/commands/observe/inbox.js +28 -21
  28. package/dist/cli/commands/observe/index.js +20 -11
  29. package/dist/cli/commands/observe/ingest.d.ts +3 -0
  30. package/dist/cli/commands/observe/ingest.js +30 -2
  31. package/dist/cli/commands/sample.d.ts +6 -3
  32. package/dist/cli/commands/sample.js +72 -68
  33. package/dist/cli/lib/codex-model-hint.d.ts +9 -0
  34. package/dist/cli/lib/codex-model-hint.js +45 -0
  35. package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
  36. package/dist/cli/lib/generation-failure-hint.js +61 -0
  37. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  38. package/dist/cli/lib/i18n-dict/common.js +4 -0
  39. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  40. package/dist/cli/lib/i18n-dict/gen.js +38 -6
  41. package/dist/cli/lib/i18n-dict/help.js +6 -6
  42. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  43. package/dist/cli/lib/i18n-dict/init.js +13 -9
  44. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  45. package/dist/cli/lib/i18n-dict/run.js +34 -2
  46. package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
  47. package/dist/cli/lib/llm-failure-classifier.js +8 -0
  48. package/dist/cli/lib/parse-run-config.d.ts +6 -5
  49. package/dist/cli/lib/parse-run-config.js +16 -9
  50. package/dist/cli/lib/runtime-defaults.d.ts +21 -0
  51. package/dist/cli/lib/runtime-defaults.js +79 -0
  52. package/dist/diagnosis/observe-mapper.js +14 -15
  53. package/dist/diagnosis/observe-producer.js +3 -1
  54. package/dist/diagnosis/studio-projection.js +14 -7
  55. package/dist/diagnosis/types.d.ts +2 -0
  56. package/dist/diagnosis/types.js +12 -0
  57. package/dist/doctor/endpoint-rule.js +2 -1
  58. package/dist/eval-core/artifact-file-names.js +18 -1
  59. package/dist/eval-core/artifact-index.d.ts +7 -11
  60. package/dist/eval-core/artifact-index.js +139 -80
  61. package/dist/eval-core/cache.d.ts +12 -3
  62. package/dist/eval-core/cache.js +89 -29
  63. package/dist/eval-core/comparability.js +10 -6
  64. package/dist/eval-core/evaluation-execution.d.ts +2 -1
  65. package/dist/eval-core/evaluation-execution.js +122 -37
  66. package/dist/eval-core/evaluation-job.d.ts +4 -1
  67. package/dist/eval-core/evaluation-job.js +4 -1
  68. package/dist/eval-core/evaluation-reporting.d.ts +15 -13
  69. package/dist/eval-core/evaluation-reporting.js +54 -52
  70. package/dist/eval-core/execution-strategy.d.ts +2 -0
  71. package/dist/eval-core/execution-strategy.js +11 -9
  72. package/dist/eval-core/fact-checker.js +15 -7
  73. package/dist/eval-core/holdout.js +3 -2
  74. package/dist/eval-core/judge-independence.d.ts +2 -2
  75. package/dist/eval-core/mock-hook.cjs +23 -6
  76. package/dist/eval-core/mocks-runtime.js +30 -8
  77. package/dist/eval-core/report-document.d.ts +12 -0
  78. package/dist/eval-core/report-document.js +1151 -0
  79. package/dist/eval-core/report-extensions.d.ts +4 -0
  80. package/dist/eval-core/report-extensions.js +500 -0
  81. package/dist/eval-core/report-file-migration.js +7 -2
  82. package/dist/eval-core/resume-compatibility.d.ts +31 -0
  83. package/dist/eval-core/resume-compatibility.js +141 -0
  84. package/dist/eval-core/sample-fingerprint.d.ts +12 -0
  85. package/dist/eval-core/sample-fingerprint.js +193 -0
  86. package/dist/eval-core/schema.js +86 -31
  87. package/dist/eval-core/verdict.d.ts +8 -4
  88. package/dist/eval-core/verdict.js +24 -10
  89. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
  90. package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
  91. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
  92. package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
  93. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
  94. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
  95. package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
  96. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
  97. package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
  98. package/dist/eval-workflows/evaluation-pipeline.js +12 -7
  99. package/dist/eval-workflows/run-evaluation.d.ts +9 -7
  100. package/dist/eval-workflows/run-evaluation.js +79 -51
  101. package/dist/executors/anthropic-api.js +65 -9
  102. package/dist/executors/claude-cli.js +16 -79
  103. package/dist/executors/claude-protocol.d.ts +28 -0
  104. package/dist/executors/claude-protocol.js +180 -0
  105. package/dist/executors/claude-sdk-trace.js +56 -28
  106. package/dist/executors/claude-sdk.d.ts +1 -0
  107. package/dist/executors/claude-sdk.js +39 -93
  108. package/dist/executors/codex-cli-trace.js +166 -31
  109. package/dist/executors/codex-cli.d.ts +6 -8
  110. package/dist/executors/codex-cli.js +49 -151
  111. package/dist/executors/codex-protocol.d.ts +24 -0
  112. package/dist/executors/codex-protocol.js +234 -0
  113. package/dist/executors/codex-sdk.js +68 -120
  114. package/dist/executors/gemini.js +88 -13
  115. package/dist/executors/index.d.ts +2 -3
  116. package/dist/executors/index.js +5 -3
  117. package/dist/executors/openai-api.js +70 -9
  118. package/dist/executors/runtime-fingerprint.js +88 -11
  119. package/dist/executors/script-command.d.ts +8 -0
  120. package/dist/executors/script-command.js +87 -0
  121. package/dist/executors/script.js +202 -29
  122. package/dist/executors/shared.d.ts +35 -3
  123. package/dist/executors/shared.js +113 -15
  124. package/dist/grading/assertions.d.ts +1 -1
  125. package/dist/grading/assertions.js +19 -9
  126. package/dist/grading/diagnostic.d.ts +9 -2
  127. package/dist/grading/diagnostic.js +25 -2
  128. package/dist/grading/index.js +10 -4
  129. package/dist/grading/judge.js +19 -6
  130. package/dist/grading/layered-scores.d.ts +2 -3
  131. package/dist/grading/layered-scores.js +2 -3
  132. package/dist/inputs/load-samples.d.ts +1 -2
  133. package/dist/inputs/load-samples.js +23 -1
  134. package/dist/inputs/mcp-resolver.js +6 -3
  135. package/dist/inputs/sample-document.d.ts +11 -0
  136. package/dist/inputs/sample-document.js +96 -0
  137. package/dist/managed/evidence.d.ts +1 -0
  138. package/dist/managed/evidence.js +1 -1
  139. package/dist/managed/store.js +200 -91
  140. package/dist/observability/codex-trace-adapter.d.ts +5 -0
  141. package/dist/observability/codex-trace-adapter.js +850 -0
  142. package/dist/observability/experience.d.ts +32 -6
  143. package/dist/observability/experience.js +2695 -459
  144. package/dist/observability/feedback-matchers.js +16 -1
  145. package/dist/observability/inbox-view-model.d.ts +1 -1
  146. package/dist/observability/inbox-view-model.js +19 -14
  147. package/dist/observability/inbox.d.ts +7 -1
  148. package/dist/observability/inbox.js +632 -124
  149. package/dist/observability/problem-patterns.js +2 -0
  150. package/dist/observability/review-state.d.ts +6 -0
  151. package/dist/observability/review-state.js +235 -63
  152. package/dist/observability/skill-chain-advisories.js +1 -1
  153. package/dist/observability/skill-chain.js +17 -4
  154. package/dist/observability/skill-health-analyzer.d.ts +32 -7
  155. package/dist/observability/skill-health-analyzer.js +194 -121
  156. package/dist/observability/skill-health-report.d.ts +10 -0
  157. package/dist/observability/skill-health-report.js +620 -0
  158. package/dist/observability/soft-standards/constants.d.ts +0 -1
  159. package/dist/observability/soft-standards/constants.js +0 -1
  160. package/dist/observability/soft-standards/index.d.ts +1 -1
  161. package/dist/observability/soft-standards/index.js +1 -1
  162. package/dist/observability/soft-standards/llm-extractor.js +8 -10
  163. package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
  164. package/dist/observability/soft-standards/skill-standards-store.js +59 -18
  165. package/dist/observability/soft-standards/types.d.ts +2 -2
  166. package/dist/observability/trace-adapter.d.ts +12 -7
  167. package/dist/observability/trace-adapter.js +11 -9
  168. package/dist/observability/trace-attribution.d.ts +13 -5
  169. package/dist/observability/trace-attribution.js +315 -21
  170. package/dist/observability/trace-ingestion.d.ts +9 -0
  171. package/dist/observability/trace-ingestion.js +80 -0
  172. package/dist/observability/trace-ir.d.ts +113 -0
  173. package/dist/observability/trace-ir.js +87 -0
  174. package/dist/observability/trace-segmenter.d.ts +19 -6
  175. package/dist/observability/trace-segmenter.js +377 -196
  176. package/dist/observability/trace-session-index.d.ts +19 -0
  177. package/dist/observability/trace-session-index.js +68 -0
  178. package/dist/observability/trace-source.d.ts +12 -4
  179. package/dist/observability/trace-source.js +939 -215
  180. package/dist/renderer/html-renderer.js +37 -6
  181. package/dist/renderer/icons.js +3 -0
  182. package/dist/renderer/observation-inbox-renderer.js +208 -90
  183. package/dist/renderer/skill-detail-renderer.js +452 -109
  184. package/dist/renderer/skill-health-renderer.js +69 -12
  185. package/dist/renderer/summary.js +28 -7
  186. package/dist/renderer/table.js +21 -4
  187. package/dist/renderer/test-view.d.ts +1 -0
  188. package/dist/renderer/test-view.js +44 -9
  189. package/dist/server/indexed-report-store.js +14 -18
  190. package/dist/server/job-store.js +64 -26
  191. package/dist/server/report-server.js +190 -78
  192. package/dist/server/report-store.js +57 -80
  193. package/dist/server/skill-index.js +143 -49
  194. package/dist/server/skill-insights.js +44 -5
  195. package/dist/shared/artifact-graph.d.ts +3 -0
  196. package/dist/shared/artifact-graph.js +224 -0
  197. package/dist/shared/assertion-types.d.ts +8 -0
  198. package/dist/shared/assertion-types.js +46 -0
  199. package/dist/shared/atomic-json.d.ts +8 -0
  200. package/dist/shared/atomic-json.js +33 -0
  201. package/dist/shared/diagnosis-schema.d.ts +9 -0
  202. package/dist/shared/diagnosis-schema.js +181 -0
  203. package/dist/shared/doctor-report.d.ts +3 -0
  204. package/dist/shared/doctor-report.js +103 -0
  205. package/dist/shared/evaluation-job.d.ts +6 -0
  206. package/dist/shared/evaluation-job.js +217 -0
  207. package/dist/shared/executor-result.d.ts +17 -0
  208. package/dist/shared/executor-result.js +221 -0
  209. package/dist/shared/file-lock.d.ts +12 -0
  210. package/dist/shared/file-lock.js +129 -0
  211. package/dist/shared/json-value.d.ts +5 -0
  212. package/dist/shared/json-value.js +36 -0
  213. package/dist/shared/keyed-mutex.d.ts +7 -0
  214. package/dist/shared/keyed-mutex.js +24 -0
  215. package/dist/shared/record-count.d.ts +8 -0
  216. package/dist/shared/record-count.js +43 -0
  217. package/dist/shared/sample-contract.d.ts +3 -0
  218. package/dist/shared/sample-contract.js +332 -0
  219. package/dist/shared/timestamp.d.ts +6 -0
  220. package/dist/shared/timestamp.js +64 -0
  221. package/dist/shared/token-usage.d.ts +19 -0
  222. package/dist/shared/token-usage.js +50 -0
  223. package/dist/shared/tool-call-status.d.ts +8 -0
  224. package/dist/shared/tool-call-status.js +28 -0
  225. package/dist/shared/tool-identity.d.ts +21 -0
  226. package/dist/shared/tool-identity.js +84 -0
  227. package/dist/shared/tool-search.js +73 -16
  228. package/dist/shared/trace-projection.d.ts +5 -0
  229. package/dist/shared/trace-projection.js +20 -0
  230. package/dist/shared/trace-source-kind.d.ts +3 -0
  231. package/dist/shared/trace-source-kind.js +12 -0
  232. package/dist/types/diagnosis.d.ts +2 -0
  233. package/dist/types/eval.d.ts +4 -0
  234. package/dist/types/executor.d.ts +32 -5
  235. package/dist/types/index.d.ts +1 -0
  236. package/dist/types/index.js +1 -0
  237. package/dist/types/judge.d.ts +2 -0
  238. package/dist/types/observability.d.ts +116 -9
  239. package/dist/types/report.d.ts +58 -6
  240. package/dist/types/skill-index.d.ts +7 -0
  241. package/dist/types/trace.d.ts +2 -0
  242. package/dist/types/trace.js +1 -0
  243. package/package.json +9 -5
@@ -20,7 +20,7 @@ export interface ClaudeCliResponse {
20
20
  usage?: TokenUsage;
21
21
  total_cost_usd?: number;
22
22
  result?: string;
23
- stop_reason?: string;
23
+ stop_reason?: string | null;
24
24
  num_turns?: number;
25
25
  }
26
26
  export interface OpenAiUsage {
@@ -34,7 +34,8 @@ export interface OpenAiResponse {
34
34
  usage?: OpenAiUsage;
35
35
  choices?: Array<{
36
36
  message?: {
37
- content?: string;
37
+ content?: string | null;
38
+ refusal?: string | null;
38
39
  };
39
40
  finish_reason?: string;
40
41
  }>;
@@ -52,6 +53,7 @@ export interface GeminiResponse {
52
53
  export interface AnthropicResponse {
53
54
  usage?: TokenUsage;
54
55
  content?: Array<{
56
+ type?: string;
55
57
  text?: string;
56
58
  }>;
57
59
  stop_reason?: string;
@@ -99,6 +101,13 @@ export interface ClaudeSdkResultMessage extends ClaudeSdkBaseMessage {
99
101
  duration_api_ms?: number;
100
102
  duration_ms?: number;
101
103
  num_turns?: number;
104
+ stop_reason?: string | null;
105
+ modelUsage?: Record<string, {
106
+ inputTokens?: number;
107
+ outputTokens?: number;
108
+ cacheReadInputTokens?: number;
109
+ cacheCreationInputTokens?: number;
110
+ }>;
102
111
  subtype?: string;
103
112
  errors?: string[];
104
113
  }
@@ -130,18 +139,30 @@ export interface CodexEvent {
130
139
  results?: unknown[];
131
140
  changes?: Array<{
132
141
  path?: string;
142
+ changeKind?: string;
133
143
  }>;
134
144
  server?: string;
135
145
  tool?: string;
146
+ name?: string;
136
147
  arguments?: unknown;
137
148
  result?: unknown;
138
149
  message?: string;
150
+ error?: {
151
+ message?: string;
152
+ };
139
153
  };
140
154
  error?: {
141
155
  message?: string;
142
156
  };
157
+ message?: string;
143
158
  ts?: number;
144
159
  }
160
+ /**
161
+ * Translate Codex's external event shape into omk's internal protocol model.
162
+ * Codex currently calls file-change discriminators `kind`; omk reserves bare
163
+ * `kind` for ArtifactKind, so the raw field is qualified at the boundary.
164
+ */
165
+ export declare function normalizeCodexProtocolEvent(value: unknown): CodexEvent | null;
145
166
  export interface ExecutorErrorLike {
146
167
  message?: string;
147
168
  name?: string;
@@ -151,9 +172,20 @@ export interface ExecutorErrorLike {
151
172
  export declare function asErrorLike(err: unknown): ExecutorErrorLike;
152
173
  export declare function errorMessage(err: unknown, fallback?: string): string;
153
174
  export declare function parseJson<T>(content: string): T;
175
+ export interface JsonResponseBody<T> {
176
+ data: T | null;
177
+ rawBody: string;
178
+ }
179
+ export declare function readJsonResponse<T>(response: Response): Promise<JsonResponseBody<T>>;
180
+ export declare function responseBodyPreview(rawBody: string, maxLength?: number): string;
154
181
  export declare function buildExecEnv(skillDir?: string | null): NodeJS.ProcessEnv;
155
182
  export declare function timeoutExecResult(timeoutMs: number, durationMs: number): ExecResult;
156
183
  export declare function interruptedExecResult(durationMs: number): ExecResult;
184
+ /**
185
+ * Register an in-process runtime (for example an SDK-owned child) with the
186
+ * same SIGINT coordinator used by spawned executors.
187
+ */
188
+ export declare function registerSigintSubscriber(subscriber: () => void): () => void;
157
189
  export declare function __resetSigintRegistryForTest(): void;
158
190
  export interface SpawnHelperResult {
159
191
  stdout: string;
@@ -176,7 +208,7 @@ export interface SpawnHelperOptions {
176
208
  env?: NodeJS.ProcessEnv;
177
209
  /** kill child after this many ms; reject with killedByTimeout=true */
178
210
  timeoutMs?: number;
179
- /** stdout overflow threshold; reject when累计超限 */
211
+ /** per-stream stdout/stderr byte limit; reject when either stream exceeds it */
180
212
  maxBuffer?: number;
181
213
  /** external abort signal; abort() 走跟 SIGINT 同一 grace 路径 */
182
214
  abortSignal?: AbortSignal;
@@ -24,6 +24,37 @@ const EXECUTOR_VENDOR = {
24
24
  export function executorVendor(executor) {
25
25
  return EXECUTOR_VENDOR[executor] ?? 'unknown';
26
26
  }
27
+ /**
28
+ * Translate Codex's external event shape into omk's internal protocol model.
29
+ * Codex currently calls file-change discriminators `kind`; omk reserves bare
30
+ * `kind` for ArtifactKind, so the raw field is qualified at the boundary.
31
+ */
32
+ export function normalizeCodexProtocolEvent(value) {
33
+ if (typeof value !== 'object' || value === null || Array.isArray(value))
34
+ return null;
35
+ const event = value;
36
+ const rawItem = event.item;
37
+ if (typeof rawItem !== 'object' || rawItem === null || Array.isArray(rawItem)) {
38
+ return event;
39
+ }
40
+ const item = rawItem;
41
+ const rawChanges = item.changes;
42
+ const normalizedItem = {
43
+ ...item,
44
+ ...(Array.isArray(rawChanges) && {
45
+ changes: rawChanges.flatMap((change) => {
46
+ if (typeof change !== 'object' || change === null || Array.isArray(change))
47
+ return [];
48
+ const rawChange = change;
49
+ return [{
50
+ ...(typeof rawChange.path === 'string' && { path: rawChange.path }),
51
+ ...(typeof rawChange.kind === 'string' && { changeKind: rawChange.kind }),
52
+ }];
53
+ }),
54
+ }),
55
+ };
56
+ return { ...event, item: normalizedItem };
57
+ }
27
58
  export function asErrorLike(err) {
28
59
  return typeof err === 'object' && err !== null ? err : {};
29
60
  }
@@ -34,6 +65,25 @@ export function errorMessage(err, fallback = 'unknown error') {
34
65
  export function parseJson(content) {
35
66
  return JSON.parse(content);
36
67
  }
68
+ export async function readJsonResponse(response) {
69
+ const rawBody = await response.text();
70
+ if (!rawBody.trim())
71
+ return { data: null, rawBody };
72
+ try {
73
+ return { data: JSON.parse(rawBody), rawBody };
74
+ }
75
+ catch {
76
+ return { data: null, rawBody };
77
+ }
78
+ }
79
+ export function responseBodyPreview(rawBody, maxLength = 500) {
80
+ const normalized = rawBody.replace(/\s+/g, ' ').trim();
81
+ if (!normalized)
82
+ return '';
83
+ return normalized.length > maxLength
84
+ ? `${normalized.slice(0, maxLength)}...`
85
+ : normalized;
86
+ }
37
87
  export function buildExecEnv(skillDir) {
38
88
  const proxyUrl = process.env.CCV_PROXY_URL || undefined;
39
89
  const env = proxyUrl
@@ -57,7 +107,9 @@ export function timeoutExecResult(timeoutMs, durationMs) {
57
107
  outputTokens: 0,
58
108
  cacheReadTokens: 0,
59
109
  cacheCreationTokens: 0,
110
+ tokenUsageReportedByExecutor: false,
60
111
  costUSD: 0,
112
+ costReportedByExecutor: false,
61
113
  output: null,
62
114
  stopReason: 'timeout',
63
115
  numTurns: 0,
@@ -76,7 +128,9 @@ export function interruptedExecResult(durationMs) {
76
128
  outputTokens: 0,
77
129
  cacheReadTokens: 0,
78
130
  cacheCreationTokens: 0,
131
+ tokenUsageReportedByExecutor: false,
79
132
  costUSD: 0,
133
+ costReportedByExecutor: false,
80
134
  output: null,
81
135
  stopReason: 'interrupted',
82
136
  numTurns: 0,
@@ -97,6 +151,7 @@ export function interruptedExecResult(durationMs) {
97
151
  // 立即退出(软关 → 硬关两段式)
98
152
  // - 同时支持 timeout 跟 abortSignal 两条 kill 路径,跟 SIGINT 共用 grace 逻辑
99
153
  const activeChildren = new Set();
154
+ const sigintSubscribers = new Set();
100
155
  let sigintListenerInstalled = false;
101
156
  let shuttingDown = false;
102
157
  const SIGTERM_GRACE_MS = 500;
@@ -104,6 +159,12 @@ function broadcastShutdown() {
104
159
  if (shuttingDown)
105
160
  return;
106
161
  shuttingDown = true;
162
+ for (const subscriber of sigintSubscribers) {
163
+ try {
164
+ subscriber();
165
+ }
166
+ catch { /* cancellation handlers must not block shutdown */ }
167
+ }
107
168
  for (const child of activeChildren) {
108
169
  try {
109
170
  child.kill('SIGTERM');
@@ -119,7 +180,14 @@ function broadcastShutdown() {
119
180
  }
120
181
  // 卸载自己,re-raise SIGINT 让 host listener / default action(exit code 130)接管
121
182
  process.removeListener('SIGINT', sigintHandler);
183
+ sigintListenerInstalled = false;
122
184
  process.kill(process.pid, 'SIGINT');
185
+ // A host may intentionally intercept the re-raised signal. In that case the
186
+ // process remains usable and a later child/SDK registration must reinstall
187
+ // a functional coordinator instead of inheriting a permanently latched flag.
188
+ setImmediate(() => {
189
+ shuttingDown = false;
190
+ }).unref();
123
191
  }
124
192
  function sigintHandler() {
125
193
  broadcastShutdown();
@@ -130,9 +198,21 @@ function ensureSigintListener() {
130
198
  sigintListenerInstalled = true;
131
199
  process.on('SIGINT', sigintHandler);
132
200
  }
201
+ /**
202
+ * Register an in-process runtime (for example an SDK-owned child) with the
203
+ * same SIGINT coordinator used by spawned executors.
204
+ */
205
+ export function registerSigintSubscriber(subscriber) {
206
+ ensureSigintListener();
207
+ sigintSubscribers.add(subscriber);
208
+ return () => {
209
+ sigintSubscribers.delete(subscriber);
210
+ };
211
+ }
133
212
  // test-only:重置模块级状态,让 vitest 之间互不污染
134
213
  export function __resetSigintRegistryForTest() {
135
214
  activeChildren.clear();
215
+ sigintSubscribers.clear();
136
216
  if (sigintListenerInstalled) {
137
217
  process.removeListener('SIGINT', sigintHandler);
138
218
  sigintListenerInstalled = false;
@@ -157,7 +237,9 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
157
237
  activeChildren.add(child);
158
238
  let stdout = '';
159
239
  let stderr = '';
160
- let bufferOverflow = false;
240
+ let stdoutBytes = 0;
241
+ let stderrBytes = 0;
242
+ let bufferOverflowStream = null;
161
243
  let killedByTimeout = false;
162
244
  let killedBySignalReason = null;
163
245
  let graceTimer = null;
@@ -192,16 +274,31 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
192
274
  if (abortSignal) {
193
275
  abortListener = () => killWithGrace('abort');
194
276
  abortSignal.addEventListener('abort', abortListener, { once: true });
277
+ if (abortSignal.aborted) {
278
+ // Defer until `done` has installed the child close/error listeners below.
279
+ queueMicrotask(abortListener);
280
+ }
195
281
  }
196
282
  child.stdout?.on('data', (chunk) => {
197
- stdout += chunk.toString();
198
- if (stdout.length > maxBuffer && !bufferOverflow) {
199
- bufferOverflow = true;
200
- // 走 killWithGrace 保证 SIGTERM trap child 也会被 500ms 后 SIGKILL 兜底
283
+ if (bufferOverflowStream)
284
+ return;
285
+ stdoutBytes += chunk.byteLength;
286
+ if (stdoutBytes > maxBuffer) {
287
+ bufferOverflowStream = 'stdout';
201
288
  killWithGrace('buffer');
289
+ return;
202
290
  }
291
+ stdout += chunk.toString();
203
292
  });
204
293
  child.stderr?.on('data', (chunk) => {
294
+ if (bufferOverflowStream)
295
+ return;
296
+ stderrBytes += chunk.byteLength;
297
+ if (stderrBytes > maxBuffer) {
298
+ bufferOverflowStream = 'stderr';
299
+ killWithGrace('buffer');
300
+ return;
301
+ }
205
302
  stderr += chunk.toString();
206
303
  });
207
304
  function cleanup() {
@@ -230,25 +327,26 @@ export function spawnWithSigintPropagation(command, args, options = {}) {
230
327
  killedByTimeout,
231
328
  killedBySignal: killedSig,
232
329
  };
233
- // bufferOverflow 优先 — 数据已截断不可信,即使 child 后来 exit 0 也不能用
234
- if (bufferOverflow) {
235
- const e = Object.assign(new Error(`stdout maxBuffer (${maxBuffer}) exceeded`), result);
330
+ // Buffer overflow is authoritative: truncated output is not valid evidence,
331
+ // even if the child catches SIGTERM and later exits 0.
332
+ if (bufferOverflowStream) {
333
+ const e = Object.assign(new Error(`${bufferOverflowStream} maxBuffer (${maxBuffer} bytes) exceeded`), result);
236
334
  reject(e);
237
335
  return;
238
336
  }
239
- // **code === 0 时 child 是干净完成的**:即使我们前面发了 SIGTERM(timeout / abort),
240
- // child 可能 trap 信号并 graceful exit 0 完成数据写入。这种情况 stdout 是完整的,
241
- // 不应该当 timeout / signal 错误 reject。优先级:exit 0 > 任何 kill reason。
242
- if (code === 0) {
243
- resolve(result);
244
- return;
245
- }
337
+ // Deadline / cancellation are caller-side facts. A child may catch
338
+ // SIGTERM, flush partial output and exit 0, but that cannot retroactively
339
+ // turn an over-budget or cancelled evaluation into a successful sample.
246
340
  if (killedByTimeout) {
247
341
  const tSec = timeoutMs ? (timeoutMs / 1000).toFixed(0) : '?';
248
342
  const e = Object.assign(new Error(`execution timed out after ${tSec}s`), result);
249
343
  reject(e);
250
344
  return;
251
345
  }
346
+ if (code === 0) {
347
+ resolve(result);
348
+ return;
349
+ }
252
350
  if (killedSig) {
253
351
  const e = Object.assign(new Error(`execution interrupted by signal ${killedSig}`), result);
254
352
  reject(e);
@@ -1,5 +1,5 @@
1
1
  import type { Assertion, AssertionResults, ExecutorFn, Sample, ToolCallInfo } from '../types/index.js';
2
- export declare const ASYNC_ASSERTION_TYPES: Set<string>;
2
+ export { ASYNC_ASSERTION_TYPES } from '../shared/assertion-types.js';
3
3
  export interface AsyncAssertionContext {
4
4
  executor: ExecutorFn;
5
5
  judgeModel: string;
@@ -2,18 +2,12 @@ import { resolve } from 'node:path';
2
2
  import _Ajv from 'ajv';
3
3
  import { ASSERTION_LAYER } from './layered-scores.js';
4
4
  import { buildSemanticSimilarityPrompt, SEMANTIC_SIMILARITY_SYSTEM, buildRagJudgePrompt } from '../shared/llm-prompts/judge-prompts.js';
5
+ import { assertionContractValidationError, } from '../shared/sample-contract.js';
6
+ import { ASYNC_ASSERTION_TYPES, SYNC_ASSERTION_TYPES, } from '../shared/assertion-types.js';
7
+ export { ASYNC_ASSERTION_TYPES } from '../shared/assertion-types.js';
5
8
  const Ajv = _Ajv.default ?? _Ajv;
6
9
  const ajv = new Ajv();
7
10
  const CUSTOM_ASSERTION_TIMEOUT_MS = 30_000;
8
- export const ASYNC_ASSERTION_TYPES = new Set([
9
- 'semantic_similarity',
10
- 'custom',
11
- // RAG-specific judge metrics. All three go through the LLM-judge path
12
- // and inherit the same length-debias instruction as the main rubric judge.
13
- 'faithfulness',
14
- 'answer_relevancy',
15
- 'context_recall',
16
- ]);
17
11
  function getErrorMessage(err) {
18
12
  return err instanceof Error ? err.message : String(err);
19
13
  }
@@ -293,6 +287,14 @@ function resolveAssertSetLayer(assertion) {
293
287
  return only === 'unknown' ? undefined : only;
294
288
  }
295
289
  export function runAssertions(output, assertions, context = {}) {
290
+ assertions.forEach((assertion, index) => {
291
+ const error = assertionContractValidationError(assertion);
292
+ if (error)
293
+ throw new TypeError(`assertions[${index}]: ${error}`);
294
+ if (!SYNC_ASSERTION_TYPES.has(assertion.type)) {
295
+ throw new TypeError(`assertions[${index}]: async assertion ${JSON.stringify(assertion.type)} requires runAsyncAssertions()`);
296
+ }
297
+ });
296
298
  const outputLower = output.toLowerCase();
297
299
  const toolCalls = context.toolCalls || [];
298
300
  const toolNames = toolCalls.map((tc) => tc.tool.toLowerCase());
@@ -324,6 +326,14 @@ export function runAssertions(output, assertions, context = {}) {
324
326
  };
325
327
  }
326
328
  export async function runAsyncAssertions(output, assertions, { executor, judgeModel, sample, samplesDir }) {
329
+ assertions.forEach((assertion, index) => {
330
+ const error = assertionContractValidationError(assertion);
331
+ if (error)
332
+ throw new TypeError(`assertions[${index}]: ${error}`);
333
+ if (!ASYNC_ASSERTION_TYPES.has(assertion.type)) {
334
+ throw new TypeError(`assertions[${index}]: sync assertion ${JSON.stringify(assertion.type)} requires runAssertions()`);
335
+ }
336
+ });
327
337
  const details = [];
328
338
  let asyncCostUSD = 0;
329
339
  let anyCostUnreported = false;
@@ -9,10 +9,17 @@
9
9
  *
10
10
  * 触发条件:sample 至少有 1 条 failed assertion(全过的 sample 不需要诊断,省成本)。
11
11
  *
12
- * 默认走 haiku(便宜 + 够用),典型成本 ~$0.005/失败样本。
12
+ * 运行目标跟随首位 judge;没有 judge 配置时跟随被测 executor/model。
13
+ * provider-specific 的廉价默认值只在 CLI runtime resolution 中决定。
13
14
  */
14
- import type { ExecutorFn, ToolCallInfo, TurnInfo, Sample } from '../types/index.js';
15
+ import type { ExecutorFn, ToolCallInfo, TurnInfo, JudgeConfig, Sample } from '../types/index.js';
15
16
  import type { AssertionDetail, DiagnosticResult, WorkflowCheck, FailureMode } from '../types/judge.js';
17
+ export interface DiagnosticTarget {
18
+ executor: string;
19
+ model: string;
20
+ }
21
+ export declare function resolveDiagnosticTarget(judgeModels: JudgeConfig[], mainExecutor: string, mainModel: string): DiagnosticTarget;
22
+ export declare function getDiagnosticPromptHash(): string;
16
23
  interface RunDiagnosticOptions {
17
24
  sample: Sample;
18
25
  skillContent: string | null;
@@ -9,9 +9,12 @@
9
9
  *
10
10
  * 触发条件:sample 至少有 1 条 failed assertion(全过的 sample 不需要诊断,省成本)。
11
11
  *
12
- * 默认走 haiku(便宜 + 够用),典型成本 ~$0.005/失败样本。
12
+ * 运行目标跟随首位 judge;没有 judge 配置时跟随被测 executor/model。
13
+ * provider-specific 的廉价默认值只在 CLI runtime resolution 中决定。
13
14
  */
15
+ import { createHash } from 'node:crypto';
14
16
  import { FAILURE_MODES } from '../types/judge.js';
17
+ import { toolCallStatus } from '../shared/tool-call-status.js';
15
18
  const SYSTEM_PROMPT = `你是 skill 评测诊断助手。基于失败用例的 expected/actual 差异,给 skill 作者具体可操作的改进建议。
16
19
 
17
20
  **诱错样本(tripwire)规则 — 只看 sample 的显式标记,不要自己识别**:
@@ -65,6 +68,15 @@ ${FAILURE_MODES.map((m, i) => ` ${i + 1}. ${m}`).join('\n')}
65
68
  全过的 sample 给空数组 []。
66
69
 
67
70
  只返回 JSON,不要其他内容。`;
71
+ export function resolveDiagnosticTarget(judgeModels, mainExecutor, mainModel) {
72
+ const primaryJudge = judgeModels[0];
73
+ return primaryJudge
74
+ ? { executor: primaryJudge.executor, model: primaryJudge.model }
75
+ : { executor: mainExecutor, model: mainModel };
76
+ }
77
+ export function getDiagnosticPromptHash() {
78
+ return createHash('sha256').update(SYSTEM_PROMPT).digest('hex').slice(0, 12);
79
+ }
68
80
  const TOOL_INPUT_PREVIEW_MAX = 350;
69
81
  const SKILL_CONTENT_MAX = 12000;
70
82
  const FULL_OUTPUT_MAX = 1500;
@@ -83,7 +95,11 @@ function previewToolCall(tc, idx) {
83
95
  const truncated = inputRepr.length > TOOL_INPUT_PREVIEW_MAX
84
96
  ? inputRepr.slice(0, TOOL_INPUT_PREVIEW_MAX) + '…'
85
97
  : inputRepr;
86
- const status = tc.success ? '' : ' [失败]';
98
+ const resultStatus = toolCallStatus(tc);
99
+ const status = resultStatus === 'failure'
100
+ ? ' [失败]'
101
+ : resultStatus === 'cancelled' ? ' [取消]'
102
+ : resultStatus === 'unknown' ? ' [状态未知]' : '';
87
103
  return `[${idx + 1}] ${tc.tool}${status} → ${truncated}`;
88
104
  }
89
105
  function formatExpectedAssertion(a) {
@@ -327,6 +343,9 @@ export async function runDiagnostic(opts) {
327
343
  timeoutMs: opts.timeoutMs ?? 180_000,
328
344
  lean: true, // 诊断也是纯文本生成,不需要工具循环
329
345
  });
346
+ const costReporting = result.costReportedByExecutor === false
347
+ ? { costReportedByExecutor: false }
348
+ : {};
330
349
  if (!result.ok) {
331
350
  return {
332
351
  ok: false,
@@ -337,6 +356,7 @@ export async function runDiagnostic(opts) {
337
356
  rootCause: [],
338
357
  suggestion: { skill: '', sample: '', none: '' },
339
358
  costUSD: result.costUSD,
359
+ ...costReporting,
340
360
  };
341
361
  }
342
362
  const raw = (result.output || '').trim();
@@ -375,6 +395,7 @@ export async function runDiagnostic(opts) {
375
395
  failureModes: salvaged.failureModes ?? [],
376
396
  suggestion: salvaged.suggestion ?? { skill: '', sample: '', none: '' },
377
397
  costUSD: result.costUSD,
398
+ ...costReporting,
378
399
  };
379
400
  }
380
401
  return {
@@ -386,6 +407,7 @@ export async function runDiagnostic(opts) {
386
407
  rootCause: [],
387
408
  suggestion: { skill: '', sample: '', none: '' },
388
409
  costUSD: result.costUSD,
410
+ ...costReporting,
389
411
  };
390
412
  }
391
413
  const obj = parsed;
@@ -410,6 +432,7 @@ export async function runDiagnostic(opts) {
410
432
  none: String(sug.none ?? ''),
411
433
  },
412
434
  costUSD: result.costUSD,
435
+ ...costReporting,
413
436
  };
414
437
  }
415
438
  function parseWorkflowChecks(raw) {
@@ -2,12 +2,18 @@
2
2
  * Mixed grading: deterministic assertions + LLM judge + multi-dimensional scoring.
3
3
  */
4
4
  import { ASYNC_ASSERTION_TYPES, ratioToScore, runAssertions, runAsyncAssertions } from './assertions.js';
5
+ import { setOwnRecordValue } from '../shared/record-count.js';
5
6
  import { buildTraceSummary, llmJudgeEnsemble, llmJudgeRepeat } from './judge.js';
6
7
  import { computeLayeredScores } from './layered-scores.js';
8
+ import { ownRecordValue } from '../shared/record-count.js';
9
+ import { sampleContractValidationError } from '../shared/sample-contract.js';
7
10
  /**
8
11
  * Grade a model output against a sample's criteria.
9
12
  */
10
13
  export async function grade({ output, sample, judgeModels, judgeExecutors, allowLlmJudge = true, execMetrics = {}, samplesDir = '.', judgeRepeat = 1, lengthDebias = true }) {
14
+ const sampleError = sampleContractValidationError(sample);
15
+ if (sampleError)
16
+ throw new TypeError(`grade(): invalid sample contract: ${sampleError}`);
11
17
  if (!judgeModels || judgeModels.length === 0) {
12
18
  throw new Error('grade(): judgeModels must be non-empty');
13
19
  }
@@ -20,14 +26,14 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
20
26
  // code path that needs the executor. Sync-only samples never invoke these
21
27
  // helpers, so they can pass `judgeExecutors: {}` (or omit it entirely).
22
28
  const requirePrimaryExecutor = () => {
23
- const exec = judgeExecutorMap[primaryJudge.executor];
29
+ const exec = ownRecordValue(judgeExecutorMap, primaryJudge.executor);
24
30
  if (!exec) {
25
31
  throw new Error(`grade(): judgeExecutors missing entry for primary judge "${primaryJudge.executor}"`);
26
32
  }
27
33
  return exec;
28
34
  };
29
35
  const executorByName = (name) => {
30
- const exec = judgeExecutorMap[name];
36
+ const exec = ownRecordValue(judgeExecutorMap, name);
31
37
  if (!exec) {
32
38
  throw new Error(`No executor registered for "${name}"; pipeline must populate judgeExecutors for every judge`);
33
39
  }
@@ -86,9 +92,9 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
86
92
  results.dimensions = {};
87
93
  for (const [dim, rubric] of Object.entries(sample.dimensions)) {
88
94
  const dimOptions = { output, rubric, prompt: sample.prompt, executor: requirePrimaryExecutor(), model: judgeModel, traceSummary, lengthDebias };
89
- results.dimensions[dim] = useEnsemble
95
+ setOwnRecordValue(results.dimensions, dim, useEnsemble
90
96
  ? await llmJudgeEnsemble(dimOptions, judgeModels, executorByName, judgeRepeat)
91
- : await llmJudgeRepeat(dimOptions, judgeRepeat);
97
+ : await llmJudgeRepeat(dimOptions, judgeRepeat));
92
98
  }
93
99
  const dimValues = Object.values(results.dimensions);
94
100
  const dimScores = dimValues.map((d) => d.score).filter((s) => s > 0);
@@ -1,4 +1,6 @@
1
1
  import { buildJudgePrompt, JUDGE_SYSTEM_PROMPT } from '../shared/llm-prompts/judge-prompts.js';
2
+ import { isToolCallCancelled, isToolCallFailure, isToolCallSuccess, isToolCallUnknown, toolCallStatus, } from '../shared/tool-call-status.js';
3
+ import { incrementRecordCount } from '../shared/record-count.js';
2
4
  // 评分类 prompt 已收口到 shared/llm-prompts/judge-prompts.ts(单一来源 + prompt-registry 冻结)。
3
5
  // 这里 re-export 保对外 API 不破:既有消费方仍从 grading/judge.js import 这两个符号。
4
6
  export { buildJudgePrompt, getJudgePromptHash } from '../shared/llm-prompts/judge-prompts.js';
@@ -80,17 +82,23 @@ export function buildTraceSummary(turns, toolCalls) {
80
82
  const lines = [];
81
83
  if (toolCalls && toolCalls.length > 0) {
82
84
  lines.push(`共调用 ${toolCalls.length} 个工具:`);
83
- const successCount = toolCalls.filter((tc) => tc.success).length;
84
- const failureCount = toolCalls.length - successCount;
85
+ const successCount = toolCalls.filter(isToolCallSuccess).length;
86
+ const failureCount = toolCalls.filter(isToolCallFailure).length;
87
+ const cancelledCount = toolCalls.filter(isToolCallCancelled).length;
88
+ const unknownCount = toolCalls.filter(isToolCallUnknown).length;
85
89
  lines.push(` 成功 ${successCount}/${toolCalls.length}`);
86
90
  if (failureCount > 0)
87
91
  lines.push(` 失败 ${failureCount}/${toolCalls.length}`);
92
+ if (cancelledCount > 0)
93
+ lines.push(` 取消 ${cancelledCount}/${toolCalls.length}`);
94
+ if (unknownCount > 0)
95
+ lines.push(` 状态未知 ${unknownCount}/${toolCalls.length}`);
88
96
  const dist = {};
89
97
  for (const tc of toolCalls) {
90
- dist[tc.tool] = (dist[tc.tool] || 0) + 1;
98
+ incrementRecordCount(dist, tc.tool);
91
99
  }
92
100
  lines.push(` 工具分布:${Object.entries(dist).map(([k, v]) => `${k}(${v})`).join(', ')}`);
93
- const failedTools = toolCalls.filter((tc) => !tc.success).map((tc) => tc.tool);
101
+ const failedTools = toolCalls.filter(isToolCallFailure).map((tc) => tc.tool);
94
102
  if (failedTools.length > 0) {
95
103
  lines.push(` 失败工具:${[...new Set(failedTools)].join(', ')}`);
96
104
  }
@@ -100,7 +108,11 @@ export function buildTraceSummary(turns, toolCalls) {
100
108
  const detailCap = Math.min(toolCalls.length, TOOL_DETAIL_MAX_CALLS);
101
109
  for (let i = 0; i < detailCap; i++) {
102
110
  const tc = toolCalls[i];
103
- const status = tc.success ? '' : ' [失败]';
111
+ const resultStatus = toolCallStatus(tc);
112
+ const status = resultStatus === 'failure'
113
+ ? ' [失败]'
114
+ : resultStatus === 'cancelled' ? ' [取消]'
115
+ : resultStatus === 'unknown' ? ' [状态未知]' : '';
104
116
  lines.push(` [${i + 1}] ${tc.tool}${status} → ${previewToolInput(tc)}`);
105
117
  }
106
118
  if (toolCalls.length > TOOL_DETAIL_MAX_CALLS) {
@@ -110,9 +122,10 @@ export function buildTraceSummary(turns, toolCalls) {
110
122
  if (turns && turns.length > 0) {
111
123
  lines.push('');
112
124
  lines.push('执行轨迹摘要:');
125
+ const userTurns = turns.filter((turn) => turn.role === 'user').length;
113
126
  const assistantTurns = turns.filter((turn) => turn.role === 'assistant').length;
114
127
  const toolTurns = turns.filter((turn) => turn.role === 'tool').length;
115
- lines.push(` 共 ${turns.length} 步(assistant ${assistantTurns} / tool ${toolTurns})`);
128
+ lines.push(` 共 ${turns.length} 步(user ${userTurns} / assistant ${assistantTurns} / tool ${toolTurns})`);
116
129
  const maxTurns = Math.min(turns.length, 10);
117
130
  for (let i = 0; i < maxTurns; i++) {
118
131
  const t = turns[i];
@@ -9,9 +9,8 @@ import type { AssertionDetail, LayeredScores } from '../types/index.js';
9
9
  * 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
10
10
  * bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
11
11
  * 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
12
- * 解析(同层→归层、混层→不计),结果落 detail.layer。`test/grading/layered-scores-exhaustiveness.test.ts` 扫
13
- * runner 源(evalAssertion 的 case ∪ `assertion.type ===` 组合器 ∪ ASYNC_ASSERTION_TYPES)守住:新增类型既不在本
14
- * 映射、又不是已知组合器,即 CI 失败。
12
+ * 解析(同层→归层、混层→不计),结果落 detail.layer。共享注册表 `shared/assertion-types.ts` 是支持类型的真源;
13
+ * `test/grading/layered-scores-exhaustiveness.test.ts` 守住新增类型必须在本映射或组合器集合中显式处理。
15
14
  */
16
15
  export declare const ASSERTION_LAYER: Record<string, 'fact' | 'behavior'>;
17
16
  interface CompositeInput {
@@ -8,9 +8,8 @@
8
8
  * 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
9
9
  * bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
10
10
  * 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
11
- * 解析(同层→归层、混层→不计),结果落 detail.layer。`test/grading/layered-scores-exhaustiveness.test.ts` 扫
12
- * runner 源(evalAssertion 的 case ∪ `assertion.type ===` 组合器 ∪ ASYNC_ASSERTION_TYPES)守住:新增类型既不在本
13
- * 映射、又不是已知组合器,即 CI 失败。
11
+ * 解析(同层→归层、混层→不计),结果落 detail.layer。共享注册表 `shared/assertion-types.ts` 是支持类型的真源;
12
+ * `test/grading/layered-scores-exhaustiveness.test.ts` 守住新增类型必须在本映射或组合器集合中显式处理。
14
13
  */
15
14
  export const ASSERTION_LAYER = {
16
15
  contains: 'fact',
@@ -1,5 +1,4 @@
1
- import type { Sample } from '../types/index.js';
2
- import type { DependencyRequirements } from '../eval-core/dependency-checker.js';
1
+ import type { DependencyRequirements, Sample } from '../types/index.js';
3
2
  export declare function parseYaml(text: string): unknown;
4
3
  export interface LoadSamplesResult {
5
4
  samples: Sample[];