oh-my-knowledge 0.51.0 → 0.51.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/dist/assets/agent-skills/omk/SKILL.md +2 -0
  2. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  3. package/dist/authoring/generator.d.ts +15 -5
  4. package/dist/authoring/generator.js +181 -18
  5. package/dist/authoring/sample-fixer.d.ts +2 -0
  6. package/dist/authoring/sample-fixer.js +19 -5
  7. package/dist/cli/commands/sample.js +11 -4
  8. package/dist/eval-core/dependency-checker.js +30 -18
  9. package/dist/eval-core/evaluation-execution.js +11 -2
  10. package/dist/eval-core/fact-checker.d.ts +15 -2
  11. package/dist/eval-core/fact-checker.js +75 -20
  12. package/dist/eval-core/mock-hook.cjs +40 -1
  13. package/dist/eval-core/mocks-runtime.js +39 -2
  14. package/dist/eval-core/task-planner.d.ts +2 -2
  15. package/dist/eval-core/task-planner.js +7 -7
  16. package/dist/eval-workflows/evaluation-pipeline.js +2 -0
  17. package/dist/eval-workflows/run-evaluation.js +2 -1
  18. package/dist/executors/capabilities.d.ts +15 -0
  19. package/dist/executors/capabilities.js +64 -0
  20. package/dist/executors/index.d.ts +1 -0
  21. package/dist/executors/index.js +4 -1
  22. package/dist/observability/codex-exec-command.d.ts +7 -0
  23. package/dist/observability/codex-exec-command.js +134 -0
  24. package/dist/observability/codex-trace-adapter.js +16 -6
  25. package/dist/observability/trace-attribution.js +11 -107
  26. package/dist/server/skill-insights.js +1 -1
  27. package/dist/shared/sample-contract.d.ts +1 -0
  28. package/dist/shared/sample-contract.js +35 -0
  29. package/dist/shared/tool-identity.d.ts +8 -0
  30. package/dist/shared/tool-identity.js +13 -0
  31. package/dist/types/eval.d.ts +7 -6
  32. package/dist/types/executor.d.ts +3 -2
  33. package/package.json +4 -4
@@ -1,4 +1,5 @@
1
1
  /** Skill attribution rules for trace records. */
2
+ import { extractCodexExecCommands } from './codex-exec-command.js';
2
3
  export function extractMarkdownLogSkill(text) {
3
4
  const patterns = [
4
5
  /\b(?:prefer|use|call|invoke)\s+`?([a-zA-Z0-9][\w.-]*)`?\s+skill\b/i,
@@ -192,108 +193,6 @@ function extractInstalledSkillReadRef(text) {
192
193
  }
193
194
  return null;
194
195
  }
195
- function skipJsString(source, start) {
196
- const quote = source[start];
197
- let index = start + 1;
198
- while (index < source.length) {
199
- if (source[index] === '\\') {
200
- index += 2;
201
- continue;
202
- }
203
- if (source[index] === quote)
204
- return index + 1;
205
- index += 1;
206
- }
207
- return source.length;
208
- }
209
- function skipJsTrivia(source, start) {
210
- let index = start;
211
- while (index < source.length) {
212
- if (/\s/.test(source[index])) {
213
- index += 1;
214
- continue;
215
- }
216
- if (source.startsWith('//', index)) {
217
- const newline = source.indexOf('\n', index + 2);
218
- return newline < 0 ? source.length : skipJsTrivia(source, newline + 1);
219
- }
220
- if (source.startsWith('/*', index)) {
221
- const end = source.indexOf('*/', index + 2);
222
- return end < 0 ? source.length : skipJsTrivia(source, end + 2);
223
- }
224
- break;
225
- }
226
- return index;
227
- }
228
- function extractExecCommandLiteral(source, callStart) {
229
- let index = skipJsTrivia(source, callStart + 'tools.exec_command'.length);
230
- if (source[index] !== '(')
231
- return null;
232
- index = skipJsTrivia(source, index + 1);
233
- if (source[index] !== '{')
234
- return null;
235
- let depth = 1;
236
- index += 1;
237
- while (index < source.length && depth > 0) {
238
- index = skipJsTrivia(source, index);
239
- const char = source[index];
240
- if (char === '"' || char === "'" || char === '`') {
241
- index = skipJsString(source, index);
242
- continue;
243
- }
244
- if (char === '{') {
245
- depth += 1;
246
- index += 1;
247
- continue;
248
- }
249
- if (char === '}') {
250
- depth -= 1;
251
- index += 1;
252
- continue;
253
- }
254
- if (depth === 1
255
- && source.startsWith('cmd', index)
256
- && !/[\w$]/.test(source[index - 1] ?? '')
257
- && !/[\w$]/.test(source[index + 3] ?? '')) {
258
- let valueStart = skipJsTrivia(source, index + 3);
259
- if (source[valueStart] !== ':') {
260
- index += 3;
261
- continue;
262
- }
263
- valueStart = skipJsTrivia(source, valueStart + 1);
264
- const quote = source[valueStart];
265
- if (quote !== '"' && quote !== "'" && quote !== '`')
266
- return null;
267
- const end = skipJsString(source, valueStart);
268
- return source.slice(valueStart + 1, Math.max(valueStart + 1, end - 1));
269
- }
270
- index += 1;
271
- }
272
- return null;
273
- }
274
- function extractExecCommandLiterals(source) {
275
- const commands = [];
276
- let index = 0;
277
- while (index < source.length) {
278
- index = skipJsTrivia(source, index);
279
- const char = source[index];
280
- if (char === '"' || char === "'" || char === '`') {
281
- index = skipJsString(source, index);
282
- continue;
283
- }
284
- if (source.startsWith('tools.exec_command', index)
285
- && !/[\w$.]/.test(source[index - 1] ?? '')
286
- && !/[\w$]/.test(source[index + 'tools.exec_command'.length] ?? '')) {
287
- const command = extractExecCommandLiteral(source, index);
288
- if (command)
289
- commands.push(command);
290
- index += 'tools.exec_command'.length;
291
- continue;
292
- }
293
- index += 1;
294
- }
295
- return commands;
296
- }
297
196
  function splitShellCommandSegments(command) {
298
197
  const segments = [];
299
198
  let start = 0;
@@ -375,7 +274,7 @@ export function extractSkillReadFileRef(record) {
375
274
  ? []
376
275
  : part.name === 'Bash'
377
276
  ? [rawCommand]
378
- : extractExecCommandLiterals(rawCommand);
277
+ : extractCodexExecCommands(rawCommand);
379
278
  for (const command of commands) {
380
279
  const skillRef = extractShellSkillReadRef(command);
381
280
  if (skillRef)
@@ -408,7 +307,7 @@ export function extractSkillScriptCommandRef(record) {
408
307
  : part.input?.code ?? part.input?.command;
409
308
  if (typeof rawCommand === 'string') {
410
309
  texts.push(...(part.name?.toLowerCase() === 'exec'
411
- ? extractExecCommandLiterals(rawCommand)
310
+ ? extractCodexExecCommands(rawCommand)
412
311
  : [rawCommand]));
413
312
  }
414
313
  }
@@ -465,9 +364,14 @@ export function extractSkillReadFileRefFromEvent(event) {
465
364
  || sourceName === 'js'
466
365
  || event.tool.name === 'node_repl.js'
467
366
  || event.tool.name.toLowerCase() === 'js';
468
- const commands = isOrchestrationWrapper
469
- ? extractExecCommandLiterals(rawCommand)
470
- : [rawCommand];
367
+ const normalizedCommands = Array.isArray(event.input.commands)
368
+ ? event.input.commands.filter((value) => typeof value === 'string')
369
+ : [];
370
+ const commands = normalizedCommands.length > 0
371
+ ? normalizedCommands
372
+ : isOrchestrationWrapper
373
+ ? extractCodexExecCommands(rawCommand)
374
+ : [rawCommand];
471
375
  for (const command of commands) {
472
376
  const skillRef = extractShellSkillReadRef(command);
473
377
  if (skillRef)
@@ -250,7 +250,7 @@ function detectSkillDocGap(doctor, evalReport, observe, variantName) {
250
250
  const recs = [];
251
251
  if (depRule) {
252
252
  recs.push({
253
- action: `在 sample.environment.files_available 加上文件路径,告诉 LLM "这些文件已就绪,无需探测"`,
253
+ action: '把已知路径写进 sample.context;只有纯题设前提才放 environment.files_available,且不要把它当作已物化 fixture',
254
254
  priority: severity,
255
255
  patch: {
256
256
  target: 'sample-environment',
@@ -1,3 +1,4 @@
1
1
  export declare function assertionContractValidationError(value: unknown, depth?: number, insideAssertSet?: boolean): string | undefined;
2
+ export declare function sampleMockReferenceKeys(value: unknown): ReadonlySet<string>;
2
3
  export declare function dependencyRequirementsValidationError(value: unknown): string | undefined;
3
4
  export declare function sampleContractValidationError(value: unknown, expectedId?: string): string | undefined;
@@ -219,6 +219,38 @@ function mockValidationError(value) {
219
219
  return '"match.input" must be a JSON object when present';
220
220
  return undefined;
221
221
  }
222
+ export function sampleMockReferenceKeys(value) {
223
+ const keys = new Set();
224
+ if (!Array.isArray(value))
225
+ return keys;
226
+ const countByTool = new Map();
227
+ for (const mock of value) {
228
+ if (!isRecord(mock) || !isNonEmptyString(mock.tool))
229
+ continue;
230
+ const ordinal = (countByTool.get(mock.tool) ?? 0) + 1;
231
+ countByTool.set(mock.tool, ordinal);
232
+ keys.add(`${mock.tool}:${ordinal}`);
233
+ }
234
+ return keys;
235
+ }
236
+ function mockHitReferenceValidationError(assertions, mockKeys) {
237
+ if (!Array.isArray(assertions))
238
+ return undefined;
239
+ for (const assertion of assertions) {
240
+ if (!isRecord(assertion))
241
+ continue;
242
+ if (assertion.type === 'mock_hit'
243
+ && typeof assertion.value === 'string'
244
+ && !mockKeys.has(assertion.value)) {
245
+ const available = mockKeys.size > 0 ? [...mockKeys].join(', ') : '(none)';
246
+ return `"mock_hit" references missing mock ${JSON.stringify(assertion.value)}; available mock keys: ${available}`;
247
+ }
248
+ const childError = mockHitReferenceValidationError(assertion.children, mockKeys);
249
+ if (childError)
250
+ return childError;
251
+ }
252
+ return undefined;
253
+ }
222
254
  export function dependencyRequirementsValidationError(value) {
223
255
  if (!isRecord(value))
224
256
  return '"requires" must be an object';
@@ -276,6 +308,9 @@ export function sampleContractValidationError(value, expectedId) {
276
308
  return `"mocks[${index}]": ${error}`;
277
309
  }
278
310
  }
311
+ const mockHitError = mockHitReferenceValidationError(value.assertions, sampleMockReferenceKeys(value.mocks));
312
+ if (mockHitError)
313
+ return mockHitError;
279
314
  if (value.mocksStrict !== undefined && typeof value.mocksStrict !== 'boolean') {
280
315
  return '"mocksStrict" must be boolean when present';
281
316
  }
@@ -19,3 +19,11 @@ export interface ToolIdentityInput {
19
19
  * retaining source identity for audit and future protocol migrations.
20
20
  */
21
21
  export declare function normalizeToolIdentity(input: ToolIdentityInput): NormalizedToolIdentity;
22
+ /**
23
+ * Match a source-neutral tool identity against a runtime-native tool name.
24
+ *
25
+ * Exact matching remains first for legacy/custom tools. Normalization then lets
26
+ * one mock identity work across adapters such as `Bash` ↔ `exec_command` and
27
+ * `Edit` ↔ `apply_patch`.
28
+ */
29
+ export declare function toolIdentityMatches(expectedName: string, runtimeName: string): boolean;
@@ -62,6 +62,19 @@ export function normalizeToolIdentity(input) {
62
62
  ...(sourceName !== name ? { sourceName } : {}),
63
63
  };
64
64
  }
65
+ /**
66
+ * Match a source-neutral tool identity against a runtime-native tool name.
67
+ *
68
+ * Exact matching remains first for legacy/custom tools. Normalization then lets
69
+ * one mock identity work across adapters such as `Bash` ↔ `exec_command` and
70
+ * `Edit` ↔ `apply_patch`.
71
+ */
72
+ export function toolIdentityMatches(expectedName, runtimeName) {
73
+ if (expectedName === runtimeName)
74
+ return true;
75
+ return normalizeToolIdentity({ sourceName: expectedName }).name
76
+ === normalizeToolIdentity({ sourceName: runtimeName }).name;
77
+ }
65
78
  function inferredMcpNamespace(sourceName) {
66
79
  const parts = sourceName.split('__').filter(Boolean);
67
80
  return parts[0] === 'mcp' && parts.length > 2
@@ -35,12 +35,12 @@ export interface SampleCoverageTarget {
35
35
  /** 稳定引用:路径或 ID。reference/script 用 skill 根相对路径,workflow_node 用 workflowId.nodeId。 */
36
36
  ref: string;
37
37
  }
38
- /** Sample 评测环境前置:声明性"已就绪"清单,LLM 看到后跳过环境探测,直接进入工作流。
39
- * 类比 unit test 的 fixture / setup —— 评测是测 skill 工作流,不是测环境探测能力。 */
38
+ /** Sample 题设环境声明。仅注入 prompt,不会修改 PATH、物化文件或改变 runtime。 */
40
39
  export interface SampleEnvironment {
41
- /** 假定已在 PATH 上的 CLI,LLM 不再 which / find / type / command -v 探测。 */
40
+ /** 题设声明可用的 CLI,LLM 不再 which / find / type / command -v 探测。 */
42
41
  cli_available?: string[];
43
- /** 假定存在的文件/脚本(支持 ~ / $SKILL_DIR / 绝对路径),LLM 不再 Glob / Read / test -f 探测。 */
42
+ /** 题设声明存在的文件/脚本(支持 ~ / $SKILL_DIR / 绝对路径),LLM 不再 Glob / Read / test -f 探测。
43
+ * 这是 prompt context,不会在 cwd 物化文件;需要真实读取时必须提供 sample.cwd 中的 fixture。 */
44
44
  files_available?: string[];
45
45
  /** 自由文本兜底,场景特殊说明(如"凭证已配""设备 SN xxx 已租"等)。 */
46
46
  notes?: string;
@@ -80,7 +80,8 @@ export type MockReturn = {
80
80
  } | string;
81
81
  /** 单条 Mock 规则。runtime 拦到匹配的 tool 调用即返回 mocked 结果,不放出去。 */
82
82
  export interface Mock {
83
- /** 拦截的工具名,如 "Read" / "Bash" / "WebFetch" / "Edit" / "Write" / "Grep" / "Glob"。
83
+ /** source-neutral 工具身份,如 "Read" / "Bash" / "WebFetch" / "Edit" / "Write" / "Grep" / "Glob"。
84
+ * executor adapter 会把 runtime-native 名称(如 exec_command / apply_patch)映射后匹配。
84
85
  * 特殊值 `"*"`:通配,匹配任何工具名(配合 match.input_contains 做 intent-level mock)。 */
85
86
  tool: string;
86
87
  /** 命中规则。所有字段 AND,字段未填即不限制。 */
@@ -136,7 +137,7 @@ export interface Sample {
136
137
  * - true:未命中即 deny(防意外真调外部接口/CLI/MCP/写状态)。
137
138
  * 全 mock 评测场景建议 true,部分 mock 探索场景留 false。 */
138
139
  mocksStrict?: boolean;
139
- /** 环境前置:声明性"已就绪"清单,LLM 跳过探测直接干活。详见 SampleEnvironment。 */
140
+ /** 题设环境声明,仅作 prompt 上下文。详见 SampleEnvironment。 */
140
141
  environment?: SampleEnvironment;
141
142
  [key: string]: unknown;
142
143
  }
@@ -87,8 +87,9 @@ export interface ExecutorInput {
87
87
  * - claude-sdk:转 in-process HookCallback 装到 SDK options.hooks.PreToolUse
88
88
  * - claude-cli:物化为临时 settings.json + on-disk hook 脚本,跑完清理
89
89
  * - script(自定义脚本):同样物化临时 settings,通过 env(OMK_MOCK_SETTINGS_FILE /
90
- * OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本若包 Claude Code 兼容
91
- * CLI 可透传 --settings 复用同一 mock hook,否则忽略(不支持的 CLI 静默无 mock) */
90
+ * OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本负责消费该协议
91
+ * - codex / codex-sdk / gemini / *-api:不支持,executor capability gate 会拒绝,
92
+ * 绝不静默忽略后把 mock_hit 记成模型失败 */
92
93
  mocks?: import('./eval.js').Mock[];
93
94
  /** 解析 mock.return_file 的相对路径锚点(默认 sample 文件所在目录)。 */
94
95
  mocksBaseDir?: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "oh-my-knowledge",
3
- "version": "0.51.0",
3
+ "version": "0.51.2",
4
4
  "packageManager": "yarn@4.16.0",
5
5
  "description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
6
6
  "type": "module",
@@ -96,11 +96,11 @@
96
96
  "license": "MIT",
97
97
  "dependencies": {
98
98
  "@anthropic-ai/claude-agent-sdk": "^0.3.143",
99
- "@anthropic-ai/sdk": "^0.112.3",
99
+ "@anthropic-ai/sdk": "^0.115.0",
100
100
  "@inquirer/prompts": "^8.4.3",
101
101
  "@modelcontextprotocol/sdk": "^1.29.0",
102
102
  "@oclif/core": "^4",
103
- "@openai/codex-sdk": "0.144.6",
103
+ "@openai/codex-sdk": "0.145.0",
104
104
  "ajv": "^8.18.0",
105
105
  "chart.js": "^4.5.1",
106
106
  "es-module-lexer": "^2.0.0",
@@ -116,7 +116,7 @@
116
116
  "@types/node": "^25.5.0",
117
117
  "eslint": "^10.1.0",
118
118
  "husky": "^9.1.7",
119
- "lint-staged": "17.1.0",
119
+ "lint-staged": "17.2.0",
120
120
  "npm-run-all2": "^9.0.1",
121
121
  "typescript": "^6.0.2",
122
122
  "typescript-eslint": "^8.58.0",