oh-my-knowledge 0.51.0 → 0.51.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/agent-skills/omk/SKILL.md +2 -0
- package/dist/assets/agent-skills/omk/references/commands.md +2 -2
- package/dist/authoring/generator.d.ts +15 -5
- package/dist/authoring/generator.js +181 -18
- package/dist/authoring/sample-fixer.d.ts +2 -0
- package/dist/authoring/sample-fixer.js +19 -5
- package/dist/cli/commands/sample.js +11 -4
- package/dist/eval-core/dependency-checker.js +30 -18
- package/dist/eval-core/evaluation-execution.js +11 -2
- package/dist/eval-core/fact-checker.d.ts +15 -2
- package/dist/eval-core/fact-checker.js +75 -20
- package/dist/eval-core/mock-hook.cjs +40 -1
- package/dist/eval-core/mocks-runtime.js +39 -2
- package/dist/eval-core/task-planner.d.ts +2 -2
- package/dist/eval-core/task-planner.js +7 -7
- package/dist/eval-workflows/evaluation-pipeline.js +2 -0
- package/dist/eval-workflows/run-evaluation.js +2 -1
- package/dist/executors/capabilities.d.ts +15 -0
- package/dist/executors/capabilities.js +64 -0
- package/dist/executors/index.d.ts +1 -0
- package/dist/executors/index.js +4 -1
- package/dist/observability/codex-exec-command.d.ts +7 -0
- package/dist/observability/codex-exec-command.js +134 -0
- package/dist/observability/codex-trace-adapter.js +16 -6
- package/dist/observability/trace-attribution.js +11 -107
- package/dist/server/skill-insights.js +1 -1
- package/dist/shared/sample-contract.d.ts +1 -0
- package/dist/shared/sample-contract.js +35 -0
- package/dist/shared/tool-identity.d.ts +8 -0
- package/dist/shared/tool-identity.js +13 -0
- package/dist/types/eval.d.ts +7 -6
- package/dist/types/executor.d.ts +3 -2
- package/package.json +4 -4
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
/** Skill attribution rules for trace records. */
|
|
2
|
+
import { extractCodexExecCommands } from './codex-exec-command.js';
|
|
2
3
|
export function extractMarkdownLogSkill(text) {
|
|
3
4
|
const patterns = [
|
|
4
5
|
/\b(?:prefer|use|call|invoke)\s+`?([a-zA-Z0-9][\w.-]*)`?\s+skill\b/i,
|
|
@@ -192,108 +193,6 @@ function extractInstalledSkillReadRef(text) {
|
|
|
192
193
|
}
|
|
193
194
|
return null;
|
|
194
195
|
}
|
|
195
|
-
function skipJsString(source, start) {
|
|
196
|
-
const quote = source[start];
|
|
197
|
-
let index = start + 1;
|
|
198
|
-
while (index < source.length) {
|
|
199
|
-
if (source[index] === '\\') {
|
|
200
|
-
index += 2;
|
|
201
|
-
continue;
|
|
202
|
-
}
|
|
203
|
-
if (source[index] === quote)
|
|
204
|
-
return index + 1;
|
|
205
|
-
index += 1;
|
|
206
|
-
}
|
|
207
|
-
return source.length;
|
|
208
|
-
}
|
|
209
|
-
function skipJsTrivia(source, start) {
|
|
210
|
-
let index = start;
|
|
211
|
-
while (index < source.length) {
|
|
212
|
-
if (/\s/.test(source[index])) {
|
|
213
|
-
index += 1;
|
|
214
|
-
continue;
|
|
215
|
-
}
|
|
216
|
-
if (source.startsWith('//', index)) {
|
|
217
|
-
const newline = source.indexOf('\n', index + 2);
|
|
218
|
-
return newline < 0 ? source.length : skipJsTrivia(source, newline + 1);
|
|
219
|
-
}
|
|
220
|
-
if (source.startsWith('/*', index)) {
|
|
221
|
-
const end = source.indexOf('*/', index + 2);
|
|
222
|
-
return end < 0 ? source.length : skipJsTrivia(source, end + 2);
|
|
223
|
-
}
|
|
224
|
-
break;
|
|
225
|
-
}
|
|
226
|
-
return index;
|
|
227
|
-
}
|
|
228
|
-
function extractExecCommandLiteral(source, callStart) {
|
|
229
|
-
let index = skipJsTrivia(source, callStart + 'tools.exec_command'.length);
|
|
230
|
-
if (source[index] !== '(')
|
|
231
|
-
return null;
|
|
232
|
-
index = skipJsTrivia(source, index + 1);
|
|
233
|
-
if (source[index] !== '{')
|
|
234
|
-
return null;
|
|
235
|
-
let depth = 1;
|
|
236
|
-
index += 1;
|
|
237
|
-
while (index < source.length && depth > 0) {
|
|
238
|
-
index = skipJsTrivia(source, index);
|
|
239
|
-
const char = source[index];
|
|
240
|
-
if (char === '"' || char === "'" || char === '`') {
|
|
241
|
-
index = skipJsString(source, index);
|
|
242
|
-
continue;
|
|
243
|
-
}
|
|
244
|
-
if (char === '{') {
|
|
245
|
-
depth += 1;
|
|
246
|
-
index += 1;
|
|
247
|
-
continue;
|
|
248
|
-
}
|
|
249
|
-
if (char === '}') {
|
|
250
|
-
depth -= 1;
|
|
251
|
-
index += 1;
|
|
252
|
-
continue;
|
|
253
|
-
}
|
|
254
|
-
if (depth === 1
|
|
255
|
-
&& source.startsWith('cmd', index)
|
|
256
|
-
&& !/[\w$]/.test(source[index - 1] ?? '')
|
|
257
|
-
&& !/[\w$]/.test(source[index + 3] ?? '')) {
|
|
258
|
-
let valueStart = skipJsTrivia(source, index + 3);
|
|
259
|
-
if (source[valueStart] !== ':') {
|
|
260
|
-
index += 3;
|
|
261
|
-
continue;
|
|
262
|
-
}
|
|
263
|
-
valueStart = skipJsTrivia(source, valueStart + 1);
|
|
264
|
-
const quote = source[valueStart];
|
|
265
|
-
if (quote !== '"' && quote !== "'" && quote !== '`')
|
|
266
|
-
return null;
|
|
267
|
-
const end = skipJsString(source, valueStart);
|
|
268
|
-
return source.slice(valueStart + 1, Math.max(valueStart + 1, end - 1));
|
|
269
|
-
}
|
|
270
|
-
index += 1;
|
|
271
|
-
}
|
|
272
|
-
return null;
|
|
273
|
-
}
|
|
274
|
-
function extractExecCommandLiterals(source) {
|
|
275
|
-
const commands = [];
|
|
276
|
-
let index = 0;
|
|
277
|
-
while (index < source.length) {
|
|
278
|
-
index = skipJsTrivia(source, index);
|
|
279
|
-
const char = source[index];
|
|
280
|
-
if (char === '"' || char === "'" || char === '`') {
|
|
281
|
-
index = skipJsString(source, index);
|
|
282
|
-
continue;
|
|
283
|
-
}
|
|
284
|
-
if (source.startsWith('tools.exec_command', index)
|
|
285
|
-
&& !/[\w$.]/.test(source[index - 1] ?? '')
|
|
286
|
-
&& !/[\w$]/.test(source[index + 'tools.exec_command'.length] ?? '')) {
|
|
287
|
-
const command = extractExecCommandLiteral(source, index);
|
|
288
|
-
if (command)
|
|
289
|
-
commands.push(command);
|
|
290
|
-
index += 'tools.exec_command'.length;
|
|
291
|
-
continue;
|
|
292
|
-
}
|
|
293
|
-
index += 1;
|
|
294
|
-
}
|
|
295
|
-
return commands;
|
|
296
|
-
}
|
|
297
196
|
function splitShellCommandSegments(command) {
|
|
298
197
|
const segments = [];
|
|
299
198
|
let start = 0;
|
|
@@ -375,7 +274,7 @@ export function extractSkillReadFileRef(record) {
|
|
|
375
274
|
? []
|
|
376
275
|
: part.name === 'Bash'
|
|
377
276
|
? [rawCommand]
|
|
378
|
-
:
|
|
277
|
+
: extractCodexExecCommands(rawCommand);
|
|
379
278
|
for (const command of commands) {
|
|
380
279
|
const skillRef = extractShellSkillReadRef(command);
|
|
381
280
|
if (skillRef)
|
|
@@ -408,7 +307,7 @@ export function extractSkillScriptCommandRef(record) {
|
|
|
408
307
|
: part.input?.code ?? part.input?.command;
|
|
409
308
|
if (typeof rawCommand === 'string') {
|
|
410
309
|
texts.push(...(part.name?.toLowerCase() === 'exec'
|
|
411
|
-
?
|
|
310
|
+
? extractCodexExecCommands(rawCommand)
|
|
412
311
|
: [rawCommand]));
|
|
413
312
|
}
|
|
414
313
|
}
|
|
@@ -465,9 +364,14 @@ export function extractSkillReadFileRefFromEvent(event) {
|
|
|
465
364
|
|| sourceName === 'js'
|
|
466
365
|
|| event.tool.name === 'node_repl.js'
|
|
467
366
|
|| event.tool.name.toLowerCase() === 'js';
|
|
468
|
-
const
|
|
469
|
-
?
|
|
470
|
-
: [
|
|
367
|
+
const normalizedCommands = Array.isArray(event.input.commands)
|
|
368
|
+
? event.input.commands.filter((value) => typeof value === 'string')
|
|
369
|
+
: [];
|
|
370
|
+
const commands = normalizedCommands.length > 0
|
|
371
|
+
? normalizedCommands
|
|
372
|
+
: isOrchestrationWrapper
|
|
373
|
+
? extractCodexExecCommands(rawCommand)
|
|
374
|
+
: [rawCommand];
|
|
471
375
|
for (const command of commands) {
|
|
472
376
|
const skillRef = extractShellSkillReadRef(command);
|
|
473
377
|
if (skillRef)
|
|
@@ -250,7 +250,7 @@ function detectSkillDocGap(doctor, evalReport, observe, variantName) {
|
|
|
250
250
|
const recs = [];
|
|
251
251
|
if (depRule) {
|
|
252
252
|
recs.push({
|
|
253
|
-
action:
|
|
253
|
+
action: '把已知路径写进 sample.context;只有纯题设前提才放 environment.files_available,且不要把它当作已物化 fixture',
|
|
254
254
|
priority: severity,
|
|
255
255
|
patch: {
|
|
256
256
|
target: 'sample-environment',
|
|
@@ -1,3 +1,4 @@
|
|
|
1
1
|
export declare function assertionContractValidationError(value: unknown, depth?: number, insideAssertSet?: boolean): string | undefined;
|
|
2
|
+
export declare function sampleMockReferenceKeys(value: unknown): ReadonlySet<string>;
|
|
2
3
|
export declare function dependencyRequirementsValidationError(value: unknown): string | undefined;
|
|
3
4
|
export declare function sampleContractValidationError(value: unknown, expectedId?: string): string | undefined;
|
|
@@ -219,6 +219,38 @@ function mockValidationError(value) {
|
|
|
219
219
|
return '"match.input" must be a JSON object when present';
|
|
220
220
|
return undefined;
|
|
221
221
|
}
|
|
222
|
+
export function sampleMockReferenceKeys(value) {
|
|
223
|
+
const keys = new Set();
|
|
224
|
+
if (!Array.isArray(value))
|
|
225
|
+
return keys;
|
|
226
|
+
const countByTool = new Map();
|
|
227
|
+
for (const mock of value) {
|
|
228
|
+
if (!isRecord(mock) || !isNonEmptyString(mock.tool))
|
|
229
|
+
continue;
|
|
230
|
+
const ordinal = (countByTool.get(mock.tool) ?? 0) + 1;
|
|
231
|
+
countByTool.set(mock.tool, ordinal);
|
|
232
|
+
keys.add(`${mock.tool}:${ordinal}`);
|
|
233
|
+
}
|
|
234
|
+
return keys;
|
|
235
|
+
}
|
|
236
|
+
function mockHitReferenceValidationError(assertions, mockKeys) {
|
|
237
|
+
if (!Array.isArray(assertions))
|
|
238
|
+
return undefined;
|
|
239
|
+
for (const assertion of assertions) {
|
|
240
|
+
if (!isRecord(assertion))
|
|
241
|
+
continue;
|
|
242
|
+
if (assertion.type === 'mock_hit'
|
|
243
|
+
&& typeof assertion.value === 'string'
|
|
244
|
+
&& !mockKeys.has(assertion.value)) {
|
|
245
|
+
const available = mockKeys.size > 0 ? [...mockKeys].join(', ') : '(none)';
|
|
246
|
+
return `"mock_hit" references missing mock ${JSON.stringify(assertion.value)}; available mock keys: ${available}`;
|
|
247
|
+
}
|
|
248
|
+
const childError = mockHitReferenceValidationError(assertion.children, mockKeys);
|
|
249
|
+
if (childError)
|
|
250
|
+
return childError;
|
|
251
|
+
}
|
|
252
|
+
return undefined;
|
|
253
|
+
}
|
|
222
254
|
export function dependencyRequirementsValidationError(value) {
|
|
223
255
|
if (!isRecord(value))
|
|
224
256
|
return '"requires" must be an object';
|
|
@@ -276,6 +308,9 @@ export function sampleContractValidationError(value, expectedId) {
|
|
|
276
308
|
return `"mocks[${index}]": ${error}`;
|
|
277
309
|
}
|
|
278
310
|
}
|
|
311
|
+
const mockHitError = mockHitReferenceValidationError(value.assertions, sampleMockReferenceKeys(value.mocks));
|
|
312
|
+
if (mockHitError)
|
|
313
|
+
return mockHitError;
|
|
279
314
|
if (value.mocksStrict !== undefined && typeof value.mocksStrict !== 'boolean') {
|
|
280
315
|
return '"mocksStrict" must be boolean when present';
|
|
281
316
|
}
|
|
@@ -19,3 +19,11 @@ export interface ToolIdentityInput {
|
|
|
19
19
|
* retaining source identity for audit and future protocol migrations.
|
|
20
20
|
*/
|
|
21
21
|
export declare function normalizeToolIdentity(input: ToolIdentityInput): NormalizedToolIdentity;
|
|
22
|
+
/**
|
|
23
|
+
* Match a source-neutral tool identity against a runtime-native tool name.
|
|
24
|
+
*
|
|
25
|
+
* Exact matching remains first for legacy/custom tools. Normalization then lets
|
|
26
|
+
* one mock identity work across adapters such as `Bash` ↔ `exec_command` and
|
|
27
|
+
* `Edit` ↔ `apply_patch`.
|
|
28
|
+
*/
|
|
29
|
+
export declare function toolIdentityMatches(expectedName: string, runtimeName: string): boolean;
|
|
@@ -62,6 +62,19 @@ export function normalizeToolIdentity(input) {
|
|
|
62
62
|
...(sourceName !== name ? { sourceName } : {}),
|
|
63
63
|
};
|
|
64
64
|
}
|
|
65
|
+
/**
|
|
66
|
+
* Match a source-neutral tool identity against a runtime-native tool name.
|
|
67
|
+
*
|
|
68
|
+
* Exact matching remains first for legacy/custom tools. Normalization then lets
|
|
69
|
+
* one mock identity work across adapters such as `Bash` ↔ `exec_command` and
|
|
70
|
+
* `Edit` ↔ `apply_patch`.
|
|
71
|
+
*/
|
|
72
|
+
export function toolIdentityMatches(expectedName, runtimeName) {
|
|
73
|
+
if (expectedName === runtimeName)
|
|
74
|
+
return true;
|
|
75
|
+
return normalizeToolIdentity({ sourceName: expectedName }).name
|
|
76
|
+
=== normalizeToolIdentity({ sourceName: runtimeName }).name;
|
|
77
|
+
}
|
|
65
78
|
function inferredMcpNamespace(sourceName) {
|
|
66
79
|
const parts = sourceName.split('__').filter(Boolean);
|
|
67
80
|
return parts[0] === 'mcp' && parts.length > 2
|
package/dist/types/eval.d.ts
CHANGED
|
@@ -35,12 +35,12 @@ export interface SampleCoverageTarget {
|
|
|
35
35
|
/** 稳定引用:路径或 ID。reference/script 用 skill 根相对路径,workflow_node 用 workflowId.nodeId。 */
|
|
36
36
|
ref: string;
|
|
37
37
|
}
|
|
38
|
-
/** Sample
|
|
39
|
-
* 类比 unit test 的 fixture / setup —— 评测是测 skill 工作流,不是测环境探测能力。 */
|
|
38
|
+
/** Sample 题设环境声明。仅注入 prompt,不会修改 PATH、物化文件或改变 runtime。 */
|
|
40
39
|
export interface SampleEnvironment {
|
|
41
|
-
/**
|
|
40
|
+
/** 题设声明可用的 CLI,LLM 不再 which / find / type / command -v 探测。 */
|
|
42
41
|
cli_available?: string[];
|
|
43
|
-
/**
|
|
42
|
+
/** 题设声明存在的文件/脚本(支持 ~ / $SKILL_DIR / 绝对路径),LLM 不再 Glob / Read / test -f 探测。
|
|
43
|
+
* 这是 prompt context,不会在 cwd 物化文件;需要真实读取时必须提供 sample.cwd 中的 fixture。 */
|
|
44
44
|
files_available?: string[];
|
|
45
45
|
/** 自由文本兜底,场景特殊说明(如"凭证已配""设备 SN xxx 已租"等)。 */
|
|
46
46
|
notes?: string;
|
|
@@ -80,7 +80,8 @@ export type MockReturn = {
|
|
|
80
80
|
} | string;
|
|
81
81
|
/** 单条 Mock 规则。runtime 拦到匹配的 tool 调用即返回 mocked 结果,不放出去。 */
|
|
82
82
|
export interface Mock {
|
|
83
|
-
/**
|
|
83
|
+
/** source-neutral 工具身份,如 "Read" / "Bash" / "WebFetch" / "Edit" / "Write" / "Grep" / "Glob"。
|
|
84
|
+
* executor adapter 会把 runtime-native 名称(如 exec_command / apply_patch)映射后匹配。
|
|
84
85
|
* 特殊值 `"*"`:通配,匹配任何工具名(配合 match.input_contains 做 intent-level mock)。 */
|
|
85
86
|
tool: string;
|
|
86
87
|
/** 命中规则。所有字段 AND,字段未填即不限制。 */
|
|
@@ -136,7 +137,7 @@ export interface Sample {
|
|
|
136
137
|
* - true:未命中即 deny(防意外真调外部接口/CLI/MCP/写状态)。
|
|
137
138
|
* 全 mock 评测场景建议 true,部分 mock 探索场景留 false。 */
|
|
138
139
|
mocksStrict?: boolean;
|
|
139
|
-
/**
|
|
140
|
+
/** 题设环境声明,仅作 prompt 上下文。详见 SampleEnvironment。 */
|
|
140
141
|
environment?: SampleEnvironment;
|
|
141
142
|
[key: string]: unknown;
|
|
142
143
|
}
|
package/dist/types/executor.d.ts
CHANGED
|
@@ -87,8 +87,9 @@ export interface ExecutorInput {
|
|
|
87
87
|
* - claude-sdk:转 in-process HookCallback 装到 SDK options.hooks.PreToolUse
|
|
88
88
|
* - claude-cli:物化为临时 settings.json + on-disk hook 脚本,跑完清理
|
|
89
89
|
* - script(自定义脚本):同样物化临时 settings,通过 env(OMK_MOCK_SETTINGS_FILE /
|
|
90
|
-
* OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)
|
|
91
|
-
*
|
|
90
|
+
* OMK_MOCK_MCP_CONFIG_FILE / OMK_MOCKS_FILE)暴露给脚本;脚本负责消费该协议
|
|
91
|
+
* - codex / codex-sdk / gemini / *-api:不支持,executor capability gate 会拒绝,
|
|
92
|
+
* 绝不静默忽略后把 mock_hit 记成模型失败 */
|
|
92
93
|
mocks?: import('./eval.js').Mock[];
|
|
93
94
|
/** 解析 mock.return_file 的相对路径锚点(默认 sample 文件所在目录)。 */
|
|
94
95
|
mocksBaseDir?: string;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "oh-my-knowledge",
|
|
3
|
-
"version": "0.51.
|
|
3
|
+
"version": "0.51.2",
|
|
4
4
|
"packageManager": "yarn@4.16.0",
|
|
5
5
|
"description": "Evaluation framework for LLM knowledge inputs — prompts, RAG corpora, skills, agent workflows. Fix the model, vary the artifact. Built-in statistical rigor: bootstrap CI, Krippendorff α, length-debias, saturation curves.",
|
|
6
6
|
"type": "module",
|
|
@@ -96,11 +96,11 @@
|
|
|
96
96
|
"license": "MIT",
|
|
97
97
|
"dependencies": {
|
|
98
98
|
"@anthropic-ai/claude-agent-sdk": "^0.3.143",
|
|
99
|
-
"@anthropic-ai/sdk": "^0.
|
|
99
|
+
"@anthropic-ai/sdk": "^0.115.0",
|
|
100
100
|
"@inquirer/prompts": "^8.4.3",
|
|
101
101
|
"@modelcontextprotocol/sdk": "^1.29.0",
|
|
102
102
|
"@oclif/core": "^4",
|
|
103
|
-
"@openai/codex-sdk": "0.
|
|
103
|
+
"@openai/codex-sdk": "0.145.0",
|
|
104
104
|
"ajv": "^8.18.0",
|
|
105
105
|
"chart.js": "^4.5.1",
|
|
106
106
|
"es-module-lexer": "^2.0.0",
|
|
@@ -116,7 +116,7 @@
|
|
|
116
116
|
"@types/node": "^25.5.0",
|
|
117
117
|
"eslint": "^10.1.0",
|
|
118
118
|
"husky": "^9.1.7",
|
|
119
|
-
"lint-staged": "17.
|
|
119
|
+
"lint-staged": "17.2.0",
|
|
120
120
|
"npm-run-all2": "^9.0.1",
|
|
121
121
|
"typescript": "^6.0.2",
|
|
122
122
|
"typescript-eslint": "^8.58.0",
|