oh-my-knowledge 0.27.0 → 0.29.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -34
- package/README.zh.md +70 -34
- package/dist/src/analysis/failure-clusterer.js +1 -1
- package/dist/src/analysis/failure-clusterer.js.map +1 -1
- package/dist/src/analysis/gap-analyzer.d.ts +8 -1
- package/dist/src/analysis/gap-analyzer.d.ts.map +1 -1
- package/dist/src/analysis/gap-analyzer.js +65 -33
- package/dist/src/analysis/gap-analyzer.js.map +1 -1
- package/dist/src/analysis/report-diagnostics.d.ts +2 -2
- package/dist/src/analysis/report-diagnostics.js +2 -2
- package/dist/src/analysis/sample-diagnostics.d.ts +3 -3
- package/dist/src/analysis/sample-diagnostics.js +17 -17
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/authoring/generator.js +1 -1
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/commands/doctor.d.ts.map +1 -1
- package/dist/src/cli/commands/doctor.js +39 -2
- package/dist/src/cli/commands/doctor.js.map +1 -1
- package/dist/src/cli/commands/eval.d.ts.map +1 -1
- package/dist/src/cli/commands/eval.js +0 -5
- package/dist/src/cli/commands/eval.js.map +1 -1
- package/dist/src/cli/commands/{export.d.ts → evolve.d.ts} +1 -1
- package/dist/src/cli/commands/evolve.d.ts.map +1 -0
- package/dist/src/cli/commands/{improve-skill.js → evolve.js} +3 -3
- package/dist/src/cli/commands/evolve.js.map +1 -0
- package/dist/src/cli/commands/observe.d.ts.map +1 -1
- package/dist/src/cli/commands/observe.js +169 -0
- package/dist/src/cli/commands/observe.js.map +1 -1
- package/dist/src/cli/commands/registry.d.ts.map +1 -1
- package/dist/src/cli/commands/registry.js +10 -20
- package/dist/src/cli/commands/registry.js.map +1 -1
- package/dist/src/cli/commands/{improve.d.ts → sample.d.ts} +1 -1
- package/dist/src/cli/commands/sample.d.ts.map +1 -0
- package/dist/src/cli/commands/{improve-samples.js → sample.js} +5 -4
- package/dist/src/cli/commands/sample.js.map +1 -0
- package/dist/src/cli/commands/studio.d.ts.map +1 -1
- package/dist/src/cli/commands/studio.js +8 -0
- package/dist/src/cli/commands/studio.js.map +1 -1
- package/dist/src/cli/i18n-dict.d.ts +2 -4
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +255 -349
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts +1 -1
- package/dist/src/cli/parse-run-config.js +1 -1
- package/dist/src/doctor/health/builtin-dimensions.d.ts +11 -0
- package/dist/src/doctor/health/builtin-dimensions.d.ts.map +1 -0
- package/dist/src/doctor/health/builtin-dimensions.js +94 -0
- package/dist/src/doctor/health/builtin-dimensions.js.map +1 -0
- package/dist/src/doctor/health/composer.d.ts +19 -0
- package/dist/src/doctor/health/composer.d.ts.map +1 -0
- package/dist/src/doctor/health/composer.js +289 -0
- package/dist/src/doctor/health/composer.js.map +1 -0
- package/dist/src/doctor/health/dimension-registry.d.ts +13 -0
- package/dist/src/doctor/health/dimension-registry.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-registry.js +28 -0
- package/dist/src/doctor/health/dimension-registry.js.map +1 -0
- package/dist/src/doctor/health/dimension-spec.d.ts +46 -0
- package/dist/src/doctor/health/dimension-spec.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-spec.js +12 -0
- package/dist/src/doctor/health/dimension-spec.js.map +1 -0
- package/dist/src/doctor/health/parser.d.ts +27 -0
- package/dist/src/doctor/health/parser.d.ts.map +1 -0
- package/dist/src/doctor/health/parser.js +190 -0
- package/dist/src/doctor/health/parser.js.map +1 -0
- package/dist/src/doctor/health/prompt-builder.d.ts +22 -0
- package/dist/src/doctor/health/prompt-builder.d.ts.map +1 -0
- package/dist/src/doctor/health/prompt-builder.js +162 -0
- package/dist/src/doctor/health/prompt-builder.js.map +1 -0
- package/dist/src/doctor/health/register.d.ts +13 -0
- package/dist/src/doctor/health/register.d.ts.map +1 -0
- package/dist/src/doctor/health/register.js +20 -0
- package/dist/src/doctor/health/register.js.map +1 -0
- package/dist/src/doctor/html-renderer.d.ts +20 -0
- package/dist/src/doctor/html-renderer.d.ts.map +1 -0
- package/dist/src/doctor/html-renderer.js +366 -0
- package/dist/src/doctor/html-renderer.js.map +1 -0
- package/dist/src/doctor/index.d.ts.map +1 -1
- package/dist/src/doctor/index.js +55 -6
- package/dist/src/doctor/index.js.map +1 -1
- package/dist/src/doctor/renderer.d.ts +4 -0
- package/dist/src/doctor/renderer.d.ts.map +1 -1
- package/dist/src/doctor/renderer.js +72 -18
- package/dist/src/doctor/renderer.js.map +1 -1
- package/dist/src/doctor/rules.d.ts +6 -5
- package/dist/src/doctor/rules.d.ts.map +1 -1
- package/dist/src/doctor/rules.js +4 -3
- package/dist/src/doctor/rules.js.map +1 -1
- package/dist/src/grading/debias-validate.js +4 -4
- package/dist/src/grading/debias-validate.js.map +1 -1
- package/dist/src/grading/gold-cli.js +2 -2
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/observability/inbox-view-model.d.ts +17 -0
- package/dist/src/observability/inbox-view-model.d.ts.map +1 -0
- package/dist/src/observability/inbox-view-model.js +62 -0
- package/dist/src/observability/inbox-view-model.js.map +1 -0
- package/dist/src/observability/inbox.d.ts +100 -0
- package/dist/src/observability/inbox.d.ts.map +1 -0
- package/dist/src/observability/inbox.js +703 -0
- package/dist/src/observability/inbox.js.map +1 -0
- package/dist/src/observability/trace-adapter.d.ts +12 -64
- package/dist/src/observability/trace-adapter.d.ts.map +1 -1
- package/dist/src/observability/trace-adapter.js +9 -355
- package/dist/src/observability/trace-adapter.js.map +1 -1
- package/dist/src/observability/trace-attribution.d.ts +28 -0
- package/dist/src/observability/trace-attribution.d.ts.map +1 -0
- package/dist/src/observability/trace-attribution.js +104 -0
- package/dist/src/observability/trace-attribution.js.map +1 -0
- package/dist/src/observability/trace-segmenter.d.ts +42 -0
- package/dist/src/observability/trace-segmenter.d.ts.map +1 -0
- package/dist/src/observability/trace-segmenter.js +231 -0
- package/dist/src/observability/trace-segmenter.js.map +1 -0
- package/dist/src/observability/trace-source.d.ts +72 -0
- package/dist/src/observability/trace-source.d.ts.map +1 -0
- package/dist/src/observability/trace-source.js +142 -0
- package/dist/src/observability/trace-source.js.map +1 -0
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +107 -59
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +58 -21
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/observation-inbox-renderer.d.ts +4 -0
- package/dist/src/renderer/observation-inbox-renderer.d.ts.map +1 -0
- package/dist/src/renderer/observation-inbox-renderer.js +900 -0
- package/dist/src/renderer/observation-inbox-renderer.js.map +1 -0
- package/dist/src/renderer/summary.d.ts +17 -0
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +277 -33
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +3 -1
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/report-server.d.ts +3 -1
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +35 -4
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/shared/tool-search.d.ts +9 -0
- package/dist/src/shared/tool-search.d.ts.map +1 -0
- package/dist/src/shared/tool-search.js +104 -0
- package/dist/src/shared/tool-search.js.map +1 -0
- package/dist/src/types/doctor.d.ts +41 -3
- package/dist/src/types/doctor.d.ts.map +1 -1
- package/dist/src/types/doctor.js +3 -0
- package/dist/src/types/doctor.js.map +1 -1
- package/dist/src/types/eval.d.ts +1 -1
- package/dist/src/types/executor.d.ts +4 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +2 -2
- package/dist/src/types/report.d.ts.map +1 -1
- package/package.json +1 -1
- package/dist/src/cli/commands/eval-debias.d.ts +0 -2
- package/dist/src/cli/commands/eval-debias.d.ts.map +0 -1
- package/dist/src/cli/commands/eval-debias.js +0 -88
- package/dist/src/cli/commands/eval-debias.js.map +0 -1
- package/dist/src/cli/commands/export-diff.d.ts +0 -2
- package/dist/src/cli/commands/export-diff.d.ts.map +0 -1
- package/dist/src/cli/commands/export-diff.js +0 -177
- package/dist/src/cli/commands/export-diff.js.map +0 -1
- package/dist/src/cli/commands/export-saturation.d.ts +0 -2
- package/dist/src/cli/commands/export-saturation.d.ts.map +0 -1
- package/dist/src/cli/commands/export-saturation.js +0 -59
- package/dist/src/cli/commands/export-saturation.js.map +0 -1
- package/dist/src/cli/commands/export-verdict.d.ts +0 -2
- package/dist/src/cli/commands/export-verdict.d.ts.map +0 -1
- package/dist/src/cli/commands/export-verdict.js +0 -45
- package/dist/src/cli/commands/export-verdict.js.map +0 -1
- package/dist/src/cli/commands/export.d.ts.map +0 -1
- package/dist/src/cli/commands/export.js +0 -150
- package/dist/src/cli/commands/export.js.map +0 -1
- package/dist/src/cli/commands/improve-failures.d.ts +0 -2
- package/dist/src/cli/commands/improve-failures.d.ts.map +0 -1
- package/dist/src/cli/commands/improve-failures.js +0 -59
- package/dist/src/cli/commands/improve-failures.js.map +0 -1
- package/dist/src/cli/commands/improve-plan.d.ts +0 -2
- package/dist/src/cli/commands/improve-plan.d.ts.map +0 -1
- package/dist/src/cli/commands/improve-plan.js +0 -75
- package/dist/src/cli/commands/improve-plan.js.map +0 -1
- package/dist/src/cli/commands/improve-samples.d.ts +0 -2
- package/dist/src/cli/commands/improve-samples.d.ts.map +0 -1
- package/dist/src/cli/commands/improve-samples.js.map +0 -1
- package/dist/src/cli/commands/improve-skill.d.ts +0 -2
- package/dist/src/cli/commands/improve-skill.d.ts.map +0 -1
- package/dist/src/cli/commands/improve-skill.js.map +0 -1
- package/dist/src/cli/commands/improve.d.ts.map +0 -1
- package/dist/src/cli/commands/improve.js +0 -34
- package/dist/src/cli/commands/improve.js.map +0 -1
- package/dist/src/cli/coverage-renderer.d.ts +0 -15
- package/dist/src/cli/coverage-renderer.d.ts.map +0 -1
- package/dist/src/cli/coverage-renderer.js +0 -74
- package/dist/src/cli/coverage-renderer.js.map +0 -1
|
@@ -16,9 +16,7 @@
|
|
|
16
16
|
* 2. **保留原文的白名单 (产品术语 / 命令 / 文件名)**
|
|
17
17
|
* 以下 token 在两种语言里都保留原文, 不翻译:
|
|
18
18
|
* - 产品名: omk, oh-my-knowledge, Claude, npm
|
|
19
|
-
* -
|
|
20
|
-
* studio, samples, skill, plan, failures, gold, debias, diff, verdict,
|
|
21
|
-
* saturation
|
|
19
|
+
* - 命令名: init, doctor, eval, observe, evolve, sample, studio, gold
|
|
22
20
|
* - omk 核心业务术语: skill, variant, sample, judge, executor (出现在产品
|
|
23
21
|
* UI 里时首字母可大写如 "Skill 评测", 描述句中保持小写)
|
|
24
22
|
* - 技术参数: --lang, --control, --treatment, --bootstrap, --judge-repeat,
|
|
@@ -73,7 +71,7 @@ export const CLI_DICT = {
|
|
|
73
71
|
en: ' 3. Run: omk eval --control code-review-v1 --treatment code-review-v2',
|
|
74
72
|
},
|
|
75
73
|
'cli.init.note_codex_executor': {
|
|
76
|
-
zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt
|
|
74
|
+
zh: '\n注: omk 评测时把 SKILL.md 整文(含 frontmatter)作为 system prompt 注入——跨 executor 一致(claude / codex / openai-api / gemini 都走同一条路径,不依赖任何 executor 的 native skill auto-discovery 或 Skill 工具机制)。frontmatter 在 prompt 头部对 model 行为无显著影响。\n模板带 Claude Code 兼容的 frontmatter(name + description)是为了让同一份 directory-skill 也能 deploy 到 Claude Code:把整个目录复制到 ~/.claude/skills/code-review-v1/(整目录,不是单个 SKILL.md),Claude SDK 才能识别。这是 omk 评测之外的 bonus,一份文件双向 dogfood。',
|
|
77
75
|
en: '\nNote: during omk evaluation the full SKILL.md (frontmatter included) is injected as the system prompt — uniformly across executors (claude / codex / openai-api / gemini all take the same path; omk does not rely on any executor\'s native skill auto-discovery or Skill tool). Frontmatter has no measurable impact on model behavior in this position.\nThe template ships with Claude Code-compatible frontmatter (name + description) so the same directory-skill can also be deployed to Claude Code: copy the whole directory to ~/.claude/skills/code-review-v1/ (the directory, not just SKILL.md) so Claude SDK can recognize it. That is a bonus beyond omk evaluation — one source, two-way dogfood.',
|
|
78
76
|
},
|
|
79
77
|
'cli.update.new_version_available': {
|
|
@@ -213,8 +211,8 @@ export const CLI_DICT = {
|
|
|
213
211
|
en: '\n💡 Non-interactive environment, skipping report server\n',
|
|
214
212
|
},
|
|
215
213
|
'cli.run.no_serve_view_hint': {
|
|
216
|
-
zh: '
|
|
217
|
-
en: '
|
|
214
|
+
zh: ' 查看报告:omk studio --reports-dir {dir}(报告 ID:{id})\n',
|
|
215
|
+
en: ' View report: omk studio --reports-dir {dir} (report id: {id})\n',
|
|
218
216
|
},
|
|
219
217
|
'cli.run.gold_load_failed': {
|
|
220
218
|
zh: '\n⚠ gold dataset 加载失败 ({dir}):\n',
|
|
@@ -260,18 +258,6 @@ export const CLI_DICT = {
|
|
|
260
258
|
zh: '⚠ 加载 samples 文件失败 ({path}): {message}\n',
|
|
261
259
|
en: '⚠ Failed to load samples file ({path}): {message}\n',
|
|
262
260
|
},
|
|
263
|
-
'cli.export.unsupported_format': {
|
|
264
|
-
zh: '不支持的导出格式:{format}。可用格式:html / markdown / github-summary。',
|
|
265
|
-
en: 'Unsupported export format: {format}. Available: html / markdown / github-summary.',
|
|
266
|
-
},
|
|
267
|
-
'cli.export.html_done': {
|
|
268
|
-
zh: '已导出 HTML:{path}',
|
|
269
|
-
en: 'HTML exported to: {path}',
|
|
270
|
-
},
|
|
271
|
-
'cli.export.done': {
|
|
272
|
-
zh: '已导出:{path}',
|
|
273
|
-
en: 'Exported to: {path}',
|
|
274
|
-
},
|
|
275
261
|
'cli.studio.started': {
|
|
276
262
|
zh: 'studio 已启动:{url}',
|
|
277
263
|
en: 'Studio running at {url}',
|
|
@@ -309,8 +295,8 @@ export const CLI_DICT = {
|
|
|
309
295
|
en: '\nGenerated {n} eval-samples files. Review them, then run: omk eval --batch',
|
|
310
296
|
},
|
|
311
297
|
'cli.gen.specify_skill_path': {
|
|
312
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
313
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
298
|
+
zh: '请指定 skill 文件路径, 例如: omk sample skills/my-skill.md',
|
|
299
|
+
en: 'Please specify a skill file path, e.g.: omk sample skills/my-skill.md',
|
|
314
300
|
},
|
|
315
301
|
'cli.gen.samples_already_exists': {
|
|
316
302
|
zh: 'eval-samples.json 已存在。如需覆盖请先删除该文件。',
|
|
@@ -333,8 +319,8 @@ export const CLI_DICT = {
|
|
|
333
319
|
en: 'Generation failed: {message}',
|
|
334
320
|
},
|
|
335
321
|
'cli.evolve.specify_skill_path': {
|
|
336
|
-
zh: '请指定 skill 文件路径, 例如: omk
|
|
337
|
-
en: 'Please specify a skill file path, e.g.: omk
|
|
322
|
+
zh: '请指定 skill 文件路径, 例如: omk evolve skills/my-skill.md',
|
|
323
|
+
en: 'Please specify a skill file path, e.g.: omk evolve skills/my-skill.md',
|
|
338
324
|
},
|
|
339
325
|
'cli.evolve.section_header': {
|
|
340
326
|
zh: '\n=== Improve skill: {path} ===\n',
|
|
@@ -365,28 +351,27 @@ export const CLI_DICT = {
|
|
|
365
351
|
en: 'All versions saved at: {dir}/\n',
|
|
366
352
|
},
|
|
367
353
|
'cli.evolve.report_link': {
|
|
368
|
-
zh: '📊
|
|
369
|
-
en: '📊
|
|
354
|
+
zh: '📊 查看报告:omk studio(报告 ID:{id})\n',
|
|
355
|
+
en: '📊 View report: omk studio (report id: {id})\n',
|
|
370
356
|
},
|
|
371
357
|
'cli.help.product_main': {
|
|
372
358
|
zh: `
|
|
373
|
-
oh-my-knowledge
|
|
359
|
+
oh-my-knowledge——知识载体工作台
|
|
374
360
|
|
|
375
361
|
用法:
|
|
376
362
|
omk init [dir] 初始化一个 skill 评测项目
|
|
377
|
-
omk doctor [path]
|
|
363
|
+
omk doctor [path] LLM 健康度审计(7 内置维度 + 可扩展);--static-only 切离线静态模式
|
|
378
364
|
omk eval [options] 离线评测:比较版本,输出 verdict + report
|
|
379
365
|
omk observe <sessions-dir> 线上观测:真实 session、gap、失败率、inbox
|
|
380
|
-
omk
|
|
381
|
-
omk
|
|
382
|
-
omk studio
|
|
366
|
+
omk evolve <skill> 多轮自动迭代改进 skill
|
|
367
|
+
omk sample <skill> 生成或补齐 eval-samples 评测用例(或 --batch 批量模式)
|
|
368
|
+
omk studio 打开本地工作台浏览报告
|
|
383
369
|
|
|
384
370
|
主路径:
|
|
385
371
|
omk doctor
|
|
386
372
|
omk eval --control code-review-v1 --treatment code-review-v2
|
|
387
373
|
omk observe ~/.claude/projects/<project>
|
|
388
|
-
omk
|
|
389
|
-
omk export <report-id> --format github-summary
|
|
374
|
+
omk evolve skills/code-review-v2/SKILL.md
|
|
390
375
|
omk studio
|
|
391
376
|
|
|
392
377
|
通用选项:
|
|
@@ -399,19 +384,18 @@ oh-my-knowledge — Knowledge Artifact Workbench
|
|
|
399
384
|
|
|
400
385
|
Usage:
|
|
401
386
|
omk init [dir] Scaffold a skill evaluation project
|
|
402
|
-
omk doctor [path]
|
|
387
|
+
omk doctor [path] LLM health audit (7 builtin dimensions, extensible); --static-only for offline static checks
|
|
403
388
|
omk eval [options] Offline evaluation: compare versions, emit verdict + report
|
|
404
389
|
omk observe <sessions-dir> Production observation: sessions, gaps, failure rate, inbox
|
|
405
|
-
omk
|
|
406
|
-
omk
|
|
407
|
-
omk studio Open the local workbench
|
|
390
|
+
omk evolve <skill> Auto-iterate a skill through multi-round eval loops
|
|
391
|
+
omk sample <skill> Generate or fill eval-samples test cases (or --batch for all skills)
|
|
392
|
+
omk studio Open the local workbench to browse reports
|
|
408
393
|
|
|
409
394
|
Main workflow:
|
|
410
395
|
omk doctor
|
|
411
396
|
omk eval --control code-review-v1 --treatment code-review-v2
|
|
412
397
|
omk observe ~/.claude/projects/<project>
|
|
413
|
-
omk
|
|
414
|
-
omk export <report-id> --format github-summary
|
|
398
|
+
omk evolve skills/code-review-v2/SKILL.md
|
|
415
399
|
omk studio
|
|
416
400
|
|
|
417
401
|
Common options:
|
|
@@ -422,7 +406,7 @@ Run 'omk <command> --help' for command-specific options.
|
|
|
422
406
|
},
|
|
423
407
|
'cli.help.init_usage': {
|
|
424
408
|
zh: `
|
|
425
|
-
omk init
|
|
409
|
+
omk init——初始化 skill 评测项目
|
|
426
410
|
|
|
427
411
|
用法:
|
|
428
412
|
omk init [dir]
|
|
@@ -456,12 +440,11 @@ Next steps:
|
|
|
456
440
|
},
|
|
457
441
|
'cli.help.eval': {
|
|
458
442
|
zh: `
|
|
459
|
-
omk eval
|
|
443
|
+
omk eval——离线评测 skill 版本,并给出 ship/no-ship verdict
|
|
460
444
|
|
|
461
445
|
用法:
|
|
462
446
|
omk eval --control <variant> --treatment <variant> [options]
|
|
463
447
|
omk eval gold <init|validate|compare> ...
|
|
464
|
-
omk eval debias length <report-id> ...
|
|
465
448
|
|
|
466
449
|
常用选项:
|
|
467
450
|
--samples <path> 用例文件(默认:eval-samples.json)
|
|
@@ -482,7 +465,8 @@ omk eval — 离线评测 skill 版本,并给出 ship/no-ship verdict
|
|
|
482
465
|
--no-serve 评测后不自动启动报告 server
|
|
483
466
|
|
|
484
467
|
示例:
|
|
485
|
-
omk eval --control
|
|
468
|
+
omk eval --control baseline --treatment my-skill # 单 skill 必要性测试(baseline 是保留 variant 名,代表「不注入 skill 的裸基线」)
|
|
469
|
+
omk eval --control code-review-v1 --treatment code-review-v2 # 多版本 A/B
|
|
486
470
|
omk eval --config eval.yaml
|
|
487
471
|
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
488
472
|
`,
|
|
@@ -492,7 +476,6 @@ omk eval — run offline skill evaluation and emit a ship/no-ship verdict
|
|
|
492
476
|
Usage:
|
|
493
477
|
omk eval --control <variant> --treatment <variant> [options]
|
|
494
478
|
omk eval gold <init|validate|compare> ...
|
|
495
|
-
omk eval debias length <report-id> ...
|
|
496
479
|
|
|
497
480
|
Common options:
|
|
498
481
|
--samples <path> Sample file (default: eval-samples.json)
|
|
@@ -513,14 +496,15 @@ Common options:
|
|
|
513
496
|
--no-serve Do not auto-start report server after evaluation
|
|
514
497
|
|
|
515
498
|
Examples:
|
|
516
|
-
omk eval --control
|
|
499
|
+
omk eval --control baseline --treatment my-skill # Single-skill necessity test (baseline is a reserved variant — "no skill injected")
|
|
500
|
+
omk eval --control code-review-v1 --treatment code-review-v2 # Multi-variant A/B
|
|
517
501
|
omk eval --config eval.yaml
|
|
518
502
|
omk eval gold compare v1-vs-v2-20260505-1200 --gold-dir gold-dataset
|
|
519
503
|
`,
|
|
520
504
|
},
|
|
521
505
|
'cli.help.eval_gold': {
|
|
522
506
|
zh: `
|
|
523
|
-
omk eval gold
|
|
507
|
+
omk eval gold——管理 human-gold 标注集
|
|
524
508
|
|
|
525
509
|
用法:
|
|
526
510
|
omk eval gold init [--out <dir>] [--annotator <name>]
|
|
@@ -544,42 +528,17 @@ Options:
|
|
|
544
528
|
--reports-dir <path> Reports directory for compare (default: ~/.oh-my-knowledge/reports)
|
|
545
529
|
--variant <name> Variant in the report to compare
|
|
546
530
|
--bootstrap-samples <n> Bootstrap resamples for compare
|
|
547
|
-
`,
|
|
548
|
-
},
|
|
549
|
-
'cli.help.eval_debias': {
|
|
550
|
-
zh: `
|
|
551
|
-
omk eval debias — 验证 length-debias 是否降低评委长度偏差
|
|
552
|
-
|
|
553
|
-
用法:
|
|
554
|
-
omk eval debias length <reportId> [options]
|
|
555
|
-
|
|
556
|
-
选项:
|
|
557
|
-
--samples <path> 用例文件;默认从 report.meta.request 读取
|
|
558
|
-
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
559
|
-
--variant <name> 只验证指定 variant
|
|
560
|
-
--judge-models <executor:model> 指定单评委
|
|
561
|
-
--bootstrap-samples <n> bootstrap 重采样次数
|
|
562
|
-
`,
|
|
563
|
-
en: `
|
|
564
|
-
omk eval debias — validate whether length-debias reduces judge length bias
|
|
565
|
-
|
|
566
|
-
Usage:
|
|
567
|
-
omk eval debias length <reportId> [options]
|
|
568
|
-
|
|
569
|
-
Options:
|
|
570
|
-
--samples <path> Sample file; defaults to report.meta.request
|
|
571
|
-
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
572
|
-
--variant <name> Validate only one variant
|
|
573
|
-
--judge-models <executor:model> Single judge to use
|
|
574
|
-
--bootstrap-samples <n> Bootstrap resamples
|
|
575
531
|
`,
|
|
576
532
|
},
|
|
577
533
|
'cli.help.observe': {
|
|
578
534
|
zh: `
|
|
579
|
-
omk observe
|
|
535
|
+
omk observe——分析真实 session trace,生成 skill 健康度日报
|
|
580
536
|
|
|
581
537
|
用法:
|
|
582
538
|
omk observe <sessions-dir> [options]
|
|
539
|
+
omk observe ingest <sessions-dir> [options]
|
|
540
|
+
omk observe inbox [options]
|
|
541
|
+
omk observe show <inbox_id> [options]
|
|
583
542
|
|
|
584
543
|
选项:
|
|
585
544
|
--kb <path> 知识库根路径(默认:从 trace cwd 推断)
|
|
@@ -588,12 +547,26 @@ omk observe — 分析真实 session trace,生成 skill 健康度日报
|
|
|
588
547
|
--to <iso> 窗口终点,优先级高于 --last
|
|
589
548
|
--skills <n1,n2,...> 只分析指定 skill
|
|
590
549
|
--output-dir <path> 输出目录(默认:~/.oh-my-knowledge/analyses)
|
|
550
|
+
|
|
551
|
+
Inbox:
|
|
552
|
+
ingest --output-dir <path> 写入 observe inbox 数据(默认:.omk/observations;读取时兜底到 ~/.oh-my-knowledge/observations)
|
|
553
|
+
inbox --input-dir <path> 读取 observe inbox 数据
|
|
554
|
+
inbox --limit <n> 展示 top N(默认:20)
|
|
555
|
+
inbox --skill <name> 只看指定 skill
|
|
556
|
+
inbox --explore <n> 从最近 50 条 medium/low 问题里抽样查看长尾
|
|
557
|
+
inbox --include-noise --explore 时显式包含 noise 桶
|
|
558
|
+
inbox --by-skill 按 skill 输出资产看板
|
|
559
|
+
inbox --json 输出 JSON
|
|
560
|
+
show <inbox_id> --input-dir <path> 查看单条 observation 的前后上下文
|
|
591
561
|
`,
|
|
592
562
|
en: `
|
|
593
563
|
omk observe — analyze production session traces and produce skill health reports
|
|
594
564
|
|
|
595
565
|
Usage:
|
|
596
566
|
omk observe <sessions-dir> [options]
|
|
567
|
+
omk observe ingest <sessions-dir> [options]
|
|
568
|
+
omk observe inbox [options]
|
|
569
|
+
omk observe show <inbox_id> [options]
|
|
597
570
|
|
|
598
571
|
Options:
|
|
599
572
|
--kb <path> Knowledge base root (default: infer from trace cwd)
|
|
@@ -602,303 +575,185 @@ Options:
|
|
|
602
575
|
--to <iso> Window end, overrides --last
|
|
603
576
|
--skills <n1,n2,...> Only analyze selected skills
|
|
604
577
|
--output-dir <path> Output directory (default: ~/.oh-my-knowledge/analyses)
|
|
605
|
-
`,
|
|
606
|
-
},
|
|
607
|
-
'cli.help.improve': {
|
|
608
|
-
zh: `
|
|
609
|
-
omk improve — 从报告或 trace 中得到下一步改进建议
|
|
610
|
-
|
|
611
|
-
用法:
|
|
612
|
-
omk improve <report-id> 输出样本质量诊断和改进计划
|
|
613
|
-
omk improve plan <report-id> 同上,显式 plan 子命令
|
|
614
|
-
omk improve failures <report-id> 聚类失败用例,生成根因和修复方向
|
|
615
|
-
omk improve samples [skill] 为 skill 生成或补齐 eval samples
|
|
616
|
-
omk improve skill <skill> 基于评测循环尝试改进 skill
|
|
617
578
|
|
|
618
|
-
|
|
619
|
-
omk
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
omk improve plan <report-id> Same as above, explicit plan subcommand
|
|
629
|
-
omk improve failures <report-id> Cluster failed cases into root causes and fixes
|
|
630
|
-
omk improve samples [skill] Generate or fill eval samples for a skill
|
|
631
|
-
omk improve skill <skill> Try to improve a skill through evaluation loops
|
|
632
|
-
|
|
633
|
-
Examples:
|
|
634
|
-
omk improve v1-vs-v2-20260505-1200
|
|
635
|
-
omk improve samples skills/code-review/SKILL.md
|
|
636
|
-
omk improve failures v1-vs-v2-20260505-1200
|
|
579
|
+
Inbox:
|
|
580
|
+
ingest --output-dir <path> Write observe inbox data (default: .omk/observations; read fallback: ~/.oh-my-knowledge/observations)
|
|
581
|
+
inbox --input-dir <path> Read observe inbox data
|
|
582
|
+
inbox --limit <n> Show top N (default: 20)
|
|
583
|
+
inbox --skill <name> Only show one skill
|
|
584
|
+
inbox --explore <n> Sample long-tail issues from the latest 50 medium/low items
|
|
585
|
+
inbox --include-noise Explicitly include the noise bucket with --explore
|
|
586
|
+
inbox --by-skill Print the skill-level asset board
|
|
587
|
+
inbox --json Print JSON
|
|
588
|
+
show <inbox_id> --input-dir <path> Show context around one observation
|
|
637
589
|
`,
|
|
638
590
|
},
|
|
639
|
-
'cli.help.
|
|
640
|
-
zh: [
|
|
641
|
-
'',
|
|
642
|
-
'用法: omk improve <reportId> [options]',
|
|
643
|
-
' omk improve plan <reportId> [options]',
|
|
644
|
-
'',
|
|
645
|
-
'诊断用例集本身的质量问题: 区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
646
|
-
'回答 "评测结论是否被坏用例污染"。',
|
|
647
|
-
'',
|
|
648
|
-
'选项:',
|
|
649
|
-
' --reports-dir <dir> 报告存储目录',
|
|
650
|
-
' --samples <path> 用例文件路径 (用于 near-duplicate 检测; 默认从 report.meta.request 读)',
|
|
651
|
-
' --top <n> 每类只显示前 N 个 (默认 10, 0=全部)',
|
|
652
|
-
' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
|
|
653
|
-
' --ambiguous-stddev <num> 歧义阈值, judge stddev (默认 1.0, 需要 --judge-repeat ≥ 2 数据)',
|
|
654
|
-
' --cost-k <num> 成本异常倍数 vs 中位数 (默认 3)',
|
|
655
|
-
' --latency-k <num> 耗时异常倍数 vs 中位数 (默认 3)',
|
|
656
|
-
' --flat <num> flat_scores 分差阈值 (默认 0.5)',
|
|
657
|
-
'',
|
|
658
|
-
].join('\n'),
|
|
659
|
-
en: [
|
|
660
|
-
'',
|
|
661
|
-
'Usage: omk improve <reportId> [options]',
|
|
662
|
-
' omk improve plan <reportId> [options]',
|
|
663
|
-
'',
|
|
664
|
-
'Diagnose quality issues in the sample set itself: low discrimination /',
|
|
665
|
-
'duplicates / ambiguity / cost anomalies / all-fail. Answers "is the verdict',
|
|
666
|
-
'tainted by bad samples?".',
|
|
667
|
-
'',
|
|
668
|
-
'Options:',
|
|
669
|
-
' --reports-dir <dir> report store dir',
|
|
670
|
-
' --samples <path> sample file path (for near-duplicate detection; defaults to report.meta.request)',
|
|
671
|
-
' --top <n> top N per category (default 10, 0=all)',
|
|
672
|
-
' --duplicate-rouge <num> near-duplicate ROUGE-1 threshold (default 0.7)',
|
|
673
|
-
' --ambiguous-stddev <num> ambiguity threshold, judge stddev (default 1.0, requires --judge-repeat ≥ 2)',
|
|
674
|
-
' --cost-k <num> cost-outlier multiplier vs median (default 3)',
|
|
675
|
-
' --latency-k <num> latency-outlier multiplier vs median (default 3)',
|
|
676
|
-
' --flat <num> flat_scores spread threshold (default 0.5)',
|
|
677
|
-
'',
|
|
678
|
-
].join('\n'),
|
|
679
|
-
},
|
|
680
|
-
'cli.help.improve_failures': {
|
|
681
|
-
zh: [
|
|
682
|
-
'',
|
|
683
|
-
'用法: omk improve failures <reportId> [options]',
|
|
684
|
-
'',
|
|
685
|
-
'把已有 report 的失败用例喂给一次 LLM 调用, 自动聚类并给出修复建议。',
|
|
686
|
-
'失败定义: compositeScore < threshold 或 ok=false。',
|
|
687
|
-
'',
|
|
688
|
-
'选项:',
|
|
689
|
-
' --reports-dir <dir> 报告存储目录',
|
|
690
|
-
' --judge-models <executor:model> 评委 (默认: 沿用 report.meta.judgeModels[0]; failures 仅支持单评委)',
|
|
691
|
-
' --max-clusters <n> 最多聚成几类 (默认 5)',
|
|
692
|
-
' --threshold <num> compositeScore < threshold 算失败 (默认 3)',
|
|
693
|
-
' --max-feed <n> 最多喂给 LLM 多少条 (默认 50, 超出取最差)',
|
|
694
|
-
'',
|
|
695
|
-
].join('\n'),
|
|
696
|
-
en: [
|
|
697
|
-
'',
|
|
698
|
-
'Usage: omk improve failures <reportId> [options]',
|
|
699
|
-
'',
|
|
700
|
-
'Feed failing samples from an existing report to a single LLM call, auto-cluster',
|
|
701
|
-
'them, and produce per-cluster fix suggestions.',
|
|
702
|
-
'Failure definition: compositeScore < threshold or ok=false.',
|
|
703
|
-
'',
|
|
704
|
-
'Options:',
|
|
705
|
-
' --reports-dir <dir> report store dir',
|
|
706
|
-
' --judge-models <executor:model> Judge (default: from report.meta.judgeModels[0]; failures is single-judge only)',
|
|
707
|
-
' --max-clusters <n> max number of clusters (default 5)',
|
|
708
|
-
' --threshold <num> compositeScore < threshold counts as failure (default 3)',
|
|
709
|
-
' --max-feed <n> max samples to feed the LLM (default 50, takes the worst)',
|
|
710
|
-
'',
|
|
711
|
-
].join('\n'),
|
|
712
|
-
},
|
|
713
|
-
'cli.help.improve_samples': {
|
|
591
|
+
'cli.help.observe_ingest': {
|
|
714
592
|
zh: `
|
|
715
|
-
omk
|
|
593
|
+
omk observe ingest — 读取真实 session trace,写入 observe inbox 数据
|
|
716
594
|
|
|
717
595
|
用法:
|
|
718
|
-
omk
|
|
719
|
-
omk improve samples --batch [--skill-dir <dir>] [options]
|
|
596
|
+
omk observe ingest <sessions-dir-or-file> [options]
|
|
720
597
|
|
|
721
598
|
选项:
|
|
722
|
-
--
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
599
|
+
--output-dir <path> 输出目录(默认:.omk/observations;读取时兜底到 ~/.oh-my-knowledge/observations)
|
|
600
|
+
|
|
601
|
+
支持:
|
|
602
|
+
Claude Code JSONL
|
|
603
|
+
Markdown 对话日志(.log)
|
|
726
604
|
`,
|
|
727
605
|
en: `
|
|
728
|
-
omk
|
|
606
|
+
omk observe ingest — read real session traces and write observe inbox data
|
|
729
607
|
|
|
730
608
|
Usage:
|
|
731
|
-
omk
|
|
732
|
-
omk improve samples --batch [--skill-dir <dir>] [options]
|
|
609
|
+
omk observe ingest <sessions-dir-or-file> [options]
|
|
733
610
|
|
|
734
611
|
Options:
|
|
735
|
-
--
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
612
|
+
--output-dir <path> Output directory (default: .omk/observations; read fallback: ~/.oh-my-knowledge/observations)
|
|
613
|
+
|
|
614
|
+
Supported:
|
|
615
|
+
Claude Code JSONL
|
|
616
|
+
Markdown conversation logs (.log)
|
|
739
617
|
`,
|
|
740
618
|
},
|
|
741
|
-
'cli.help.
|
|
619
|
+
'cli.help.observe_inbox': {
|
|
742
620
|
zh: `
|
|
743
|
-
omk
|
|
621
|
+
omk observe inbox — 查看已写入的 observe inbox 问题列表
|
|
744
622
|
|
|
745
623
|
用法:
|
|
746
|
-
omk
|
|
624
|
+
omk observe inbox [options]
|
|
747
625
|
|
|
748
626
|
选项:
|
|
749
|
-
--
|
|
750
|
-
--
|
|
751
|
-
--
|
|
752
|
-
--
|
|
627
|
+
--input-dir <path> 读取目录(默认:.omk/observations;兜底到 ~/.oh-my-knowledge/observations)
|
|
628
|
+
--limit <n> 展示 top N(默认:20)
|
|
629
|
+
--skill <name> 只看指定 skill
|
|
630
|
+
--explore <n> 从最近 50 条 medium/low 问题里抽样查看长尾
|
|
631
|
+
--include-noise --explore 时显式包含 noise 桶
|
|
632
|
+
--by-skill 按 skill 输出资产看板
|
|
633
|
+
--json 输出 JSON
|
|
753
634
|
`,
|
|
754
635
|
en: `
|
|
755
|
-
omk
|
|
636
|
+
omk observe inbox — inspect previously ingested observe inbox items
|
|
756
637
|
|
|
757
638
|
Usage:
|
|
758
|
-
omk
|
|
639
|
+
omk observe inbox [options]
|
|
759
640
|
|
|
760
641
|
Options:
|
|
761
|
-
--
|
|
762
|
-
--
|
|
763
|
-
--
|
|
764
|
-
--
|
|
642
|
+
--input-dir <path> Input directory (default: .omk/observations; fallback: ~/.oh-my-knowledge/observations)
|
|
643
|
+
--limit <n> Show top N (default: 20)
|
|
644
|
+
--skill <name> Only show one skill
|
|
645
|
+
--explore <n> Sample long-tail issues from the latest 50 medium/low items
|
|
646
|
+
--include-noise Explicitly include the noise bucket with --explore
|
|
647
|
+
--by-skill Print the skill-level asset board
|
|
648
|
+
--json Print JSON
|
|
765
649
|
`,
|
|
766
650
|
},
|
|
767
|
-
'cli.help.
|
|
651
|
+
'cli.help.observe_show': {
|
|
768
652
|
zh: `
|
|
769
|
-
omk
|
|
770
|
-
|
|
771
|
-
用法:
|
|
772
|
-
omk export <report-id> [options]
|
|
773
|
-
omk export diff <report-id> [report-id] [options]
|
|
774
|
-
omk export verdict <report-id> [options]
|
|
775
|
-
omk export saturation <report-id> [options]
|
|
776
|
-
|
|
777
|
-
选项:
|
|
778
|
-
--format <format> html / markdown / github-summary(默认:html)
|
|
779
|
-
--out <path> 输出文件;markdown / github-summary 未指定时输出到 stdout
|
|
780
|
-
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
781
|
-
|
|
782
|
-
示例:
|
|
783
|
-
omk export v1-vs-v2-20260505-1200 --format github-summary
|
|
784
|
-
omk export v1-vs-v2-20260505-1200 --format markdown --out report.md
|
|
785
|
-
omk export diff v1-vs-v2-20260505-1200 --regressions-only
|
|
786
|
-
omk export verdict v1-vs-v2-20260505-1200
|
|
787
|
-
`,
|
|
788
|
-
en: `
|
|
789
|
-
omk export — export evidence packs for PR / CI / audit
|
|
790
|
-
|
|
791
|
-
Usage:
|
|
792
|
-
omk export <report-id> [options]
|
|
793
|
-
omk export diff <report-id> [report-id] [options]
|
|
794
|
-
omk export verdict <report-id> [options]
|
|
795
|
-
omk export saturation <report-id> [options]
|
|
796
|
-
|
|
797
|
-
Options:
|
|
798
|
-
--format <format> html / markdown / github-summary (default: html)
|
|
799
|
-
--out <path> Output file; markdown / github-summary print to stdout by default
|
|
800
|
-
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
801
|
-
|
|
802
|
-
Examples:
|
|
803
|
-
omk export v1-vs-v2-20260505-1200 --format github-summary
|
|
804
|
-
omk export v1-vs-v2-20260505-1200 --format markdown --out report.md
|
|
805
|
-
omk export diff v1-vs-v2-20260505-1200 --regressions-only
|
|
806
|
-
omk export verdict v1-vs-v2-20260505-1200
|
|
807
|
-
`,
|
|
808
|
-
},
|
|
809
|
-
'cli.help.export_diff': {
|
|
810
|
-
zh: `
|
|
811
|
-
omk export diff — 导出样本级或报告级差异
|
|
653
|
+
omk observe show — 查看单条 observation 的原始上下文
|
|
812
654
|
|
|
813
655
|
用法:
|
|
814
|
-
omk
|
|
815
|
-
omk export diff <report-id-a> <report-id-b>
|
|
656
|
+
omk observe show <inbox_id> [options]
|
|
816
657
|
|
|
817
658
|
选项:
|
|
818
|
-
--
|
|
819
|
-
--variant <name> 样本级 diff 的实验组 variant
|
|
820
|
-
--regressions-only 只显示回退用例
|
|
821
|
-
--top <n> 最多显示 N 条
|
|
659
|
+
--input-dir <path> 读取目录(默认:.omk/observations;兜底到 ~/.oh-my-knowledge/observations)
|
|
822
660
|
`,
|
|
823
661
|
en: `
|
|
824
|
-
omk
|
|
662
|
+
omk observe show — inspect the raw context around one observation
|
|
825
663
|
|
|
826
664
|
Usage:
|
|
827
|
-
omk
|
|
828
|
-
omk export diff <report-id-a> <report-id-b>
|
|
665
|
+
omk observe show <inbox_id> [options]
|
|
829
666
|
|
|
830
667
|
Options:
|
|
831
|
-
--
|
|
832
|
-
--variant <name> Treatment variant for sample-level diff
|
|
833
|
-
--regressions-only Show regressions only
|
|
834
|
-
--top <n> Show at most N rows
|
|
668
|
+
--input-dir <path> Input directory (default: .omk/observations; fallback: ~/.oh-my-knowledge/observations)
|
|
835
669
|
`,
|
|
836
670
|
},
|
|
837
|
-
'cli.help.
|
|
671
|
+
'cli.help.evolve': {
|
|
838
672
|
zh: `
|
|
839
|
-
omk
|
|
673
|
+
omk evolve——多轮自动迭代改进 skill
|
|
840
674
|
|
|
841
675
|
用法:
|
|
842
|
-
omk
|
|
676
|
+
omk evolve <skill-path> [options]
|
|
843
677
|
|
|
844
678
|
选项:
|
|
845
|
-
--
|
|
846
|
-
--
|
|
847
|
-
--
|
|
848
|
-
--
|
|
679
|
+
--rounds <n> 迭代轮数(默认:5)
|
|
680
|
+
--target <score> 目标分数
|
|
681
|
+
--model <name> 任务执行模型,每轮跑 eval samples 的被测模型(默认:sonnet)
|
|
682
|
+
--improve-model <name> skill 改写模型,每轮根据反馈改写 skill 的模型(默认:sonnet)
|
|
683
|
+
--judge-models <executor:model> 单评委配置(默认:claude:haiku)
|
|
684
|
+
|
|
685
|
+
示例:
|
|
686
|
+
omk evolve skills/code-review/SKILL.md
|
|
687
|
+
omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
|
|
688
|
+
omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
|
|
849
689
|
`,
|
|
850
690
|
en: `
|
|
851
|
-
omk
|
|
691
|
+
omk evolve — auto-iterate a skill through multi-round evaluation loops
|
|
852
692
|
|
|
853
693
|
Usage:
|
|
854
|
-
omk
|
|
694
|
+
omk evolve <skill-path> [options]
|
|
855
695
|
|
|
856
696
|
Options:
|
|
857
|
-
--
|
|
858
|
-
--
|
|
859
|
-
--
|
|
860
|
-
--
|
|
697
|
+
--rounds <n> Iteration rounds (default: 5)
|
|
698
|
+
--target <score> Target score
|
|
699
|
+
--model <name> Task executor model — runs eval samples each round (default: sonnet)
|
|
700
|
+
--improve-model <name> Skill rewriter model — rewrites the skill each round (default: sonnet)
|
|
701
|
+
--judge-models <executor:model> Single judge config (default: claude:haiku)
|
|
702
|
+
|
|
703
|
+
Examples:
|
|
704
|
+
omk evolve skills/code-review/SKILL.md
|
|
705
|
+
omk evolve skills/code-review/SKILL.md --rounds 10 --target 4.5
|
|
706
|
+
omk evolve skills/code-review/SKILL.md --model sonnet --improve-model opus
|
|
861
707
|
`,
|
|
862
708
|
},
|
|
863
|
-
'cli.help.
|
|
709
|
+
'cli.help.sample': {
|
|
864
710
|
zh: `
|
|
865
|
-
omk
|
|
711
|
+
omk sample——生成或补齐 eval-samples 评测用例
|
|
866
712
|
|
|
867
713
|
用法:
|
|
868
|
-
omk
|
|
714
|
+
omk sample <skill-path> [options]
|
|
715
|
+
omk sample --batch [--skill-dir <dir>] [options]
|
|
869
716
|
|
|
870
717
|
选项:
|
|
871
|
-
--
|
|
872
|
-
--
|
|
718
|
+
--count <n> 生成用例数量(默认:5)
|
|
719
|
+
--model <name> 生成模型(默认:sonnet)
|
|
720
|
+
--batch 为 skill 目录下缺少 eval-samples 的 skill 批量生成
|
|
721
|
+
--skill-dir <path> skill 目录(batch 使用,默认:skills)
|
|
873
722
|
`,
|
|
874
723
|
en: `
|
|
875
|
-
omk
|
|
724
|
+
omk sample — generate or fill eval-samples test cases
|
|
876
725
|
|
|
877
726
|
Usage:
|
|
878
|
-
omk
|
|
727
|
+
omk sample <skill-path> [options]
|
|
728
|
+
omk sample --batch [--skill-dir <dir>] [options]
|
|
879
729
|
|
|
880
730
|
Options:
|
|
881
|
-
--
|
|
882
|
-
--
|
|
731
|
+
--count <n> Number of test cases to generate (default: 5)
|
|
732
|
+
--model <name> Generation model (default: sonnet)
|
|
733
|
+
--batch Generate for skills that are missing eval-samples
|
|
734
|
+
--skill-dir <path> Skill directory for batch mode (default: skills)
|
|
883
735
|
`,
|
|
884
736
|
},
|
|
885
737
|
'cli.help.studio': {
|
|
886
738
|
zh: `
|
|
887
|
-
omk studio
|
|
739
|
+
omk studio——打开本地知识工作台
|
|
888
740
|
|
|
889
741
|
用法:
|
|
890
742
|
omk studio [options]
|
|
891
743
|
|
|
892
744
|
选项:
|
|
893
745
|
--port <n> 本地服务端口(默认:7799)
|
|
746
|
+
--host <host> 监听地址(默认:127.0.0.1;局域网访问可用 0.0.0.0)
|
|
894
747
|
--reports-dir <path> 报告目录(默认:~/.oh-my-knowledge/reports)
|
|
895
748
|
--analyses-dir <path> 观测分析目录
|
|
749
|
+
--observations-dir <path> observe inbox 数据目录(默认:.omk/observations)
|
|
896
750
|
--no-open 只启动服务,不自动打开浏览器
|
|
897
751
|
--dev 开发模式:文件变化时自动重启
|
|
898
752
|
|
|
899
753
|
示例:
|
|
900
754
|
omk studio
|
|
901
755
|
omk studio --port 7798
|
|
756
|
+
omk studio --host 0.0.0.0 --observations-dir .omk/observations
|
|
902
757
|
omk studio --no-open
|
|
903
758
|
`,
|
|
904
759
|
en: `
|
|
@@ -909,38 +764,20 @@ Usage:
|
|
|
909
764
|
|
|
910
765
|
Options:
|
|
911
766
|
--port <n> Local server port (default: 7799)
|
|
767
|
+
--host <host> Listen address (default: 127.0.0.1; use 0.0.0.0 for LAN access)
|
|
912
768
|
--reports-dir <path> Reports directory (default: ~/.oh-my-knowledge/reports)
|
|
913
769
|
--analyses-dir <path> Observation analyses directory
|
|
770
|
+
--observations-dir <path> Observe inbox data directory (default: .omk/observations)
|
|
914
771
|
--no-open Start the server without opening a browser
|
|
915
772
|
--dev Dev mode: restart on file changes
|
|
916
773
|
|
|
917
774
|
Examples:
|
|
918
775
|
omk studio
|
|
919
776
|
omk studio --port 7798
|
|
777
|
+
omk studio --host 0.0.0.0 --observations-dir .omk/observations
|
|
920
778
|
omk studio --no-open
|
|
921
779
|
`,
|
|
922
780
|
},
|
|
923
|
-
// sample design coverage block strings
|
|
924
|
-
'cli.diagnose.coverage_header': {
|
|
925
|
-
zh: '用例设计覆盖度 (Sample design coverage):',
|
|
926
|
-
en: 'Sample design coverage:',
|
|
927
|
-
},
|
|
928
|
-
'cli.diagnose.coverage_unspecified': {
|
|
929
|
-
zh: '(未声明)',
|
|
930
|
-
en: '(unspecified)',
|
|
931
|
-
},
|
|
932
|
-
'cli.diagnose.coverage_chars': {
|
|
933
|
-
zh: '字符',
|
|
934
|
-
en: 'chars',
|
|
935
|
-
},
|
|
936
|
-
'cli.diagnose.coverage_hint_empty': {
|
|
937
|
-
zh: 'ℹ 该用例集未声明任何 capability / difficulty / construct / provenance 元数据。详见 docs/sample-design-spec.md',
|
|
938
|
-
en: 'ℹ No samples in this set declare capability / difficulty / construct / provenance metadata. See docs/sample-design-spec.md',
|
|
939
|
-
},
|
|
940
|
-
'cli.diagnose.coverage_declared': {
|
|
941
|
-
zh: '声明',
|
|
942
|
-
en: 'declared',
|
|
943
|
-
},
|
|
944
781
|
// ============ omk doctor 健康检查 ============
|
|
945
782
|
'cli.doctor.rule.skill_readable': {
|
|
946
783
|
zh: 'skill 文件可读',
|
|
@@ -958,6 +795,67 @@ Examples:
|
|
|
958
795
|
zh: '用例 ↔ skill 输入约定',
|
|
959
796
|
en: 'samples ↔ skill contract',
|
|
960
797
|
},
|
|
798
|
+
'cli.doctor.rule.skill_health_check': {
|
|
799
|
+
zh: '健康度体检',
|
|
800
|
+
en: 'Health check',
|
|
801
|
+
},
|
|
802
|
+
// ============ skill_health composer (CLI default; --static-only disables it) ============
|
|
803
|
+
'cli.doctor.health.skipped': {
|
|
804
|
+
zh: '健康度体检已跳过(runHealthCheck=false)',
|
|
805
|
+
en: 'health check skipped (runHealthCheck=false)',
|
|
806
|
+
},
|
|
807
|
+
'cli.doctor.health.no_dimensions': {
|
|
808
|
+
zh: '没有注册任何健康度维度,跳过',
|
|
809
|
+
en: 'no health dimensions registered, skipped',
|
|
810
|
+
},
|
|
811
|
+
'cli.doctor.health.fail.executor': {
|
|
812
|
+
zh: 'LLM 调用失败: {error}',
|
|
813
|
+
en: 'LLM call failed: {error}',
|
|
814
|
+
},
|
|
815
|
+
'cli.doctor.health.fail.parse': {
|
|
816
|
+
zh: 'LLM 输出解析失败: {error}',
|
|
817
|
+
en: 'failed to parse LLM output: {error}',
|
|
818
|
+
},
|
|
819
|
+
'cli.doctor.health.fail.empty_output': {
|
|
820
|
+
zh: 'LLM 返回了空输出',
|
|
821
|
+
en: 'LLM returned empty output',
|
|
822
|
+
},
|
|
823
|
+
'cli.doctor.health.hint.executor': {
|
|
824
|
+
zh: '检查 executor 配置(--executor / --model)与网络连通,或调大 --timeout',
|
|
825
|
+
en: 'Verify executor config (--executor / --model) and connectivity, or raise --timeout',
|
|
826
|
+
},
|
|
827
|
+
'cli.doctor.health.hint.parse': {
|
|
828
|
+
zh: 'LLM 没返回合法 JSON;原文存在 detail.rawOutput 截断片段,可重跑或换 model',
|
|
829
|
+
en: 'LLM did not return valid JSON; raw snippet stored in detail.rawOutput. Re-run or switch model',
|
|
830
|
+
},
|
|
831
|
+
'cli.doctor.health.dim.message': {
|
|
832
|
+
zh: '{level}: 错误 {err}/警告 {warn}/建议 {sug}',
|
|
833
|
+
en: '{level}: error {err}/warn {warn}/suggest {sug}',
|
|
834
|
+
},
|
|
835
|
+
'cli.doctor.health.dim.missing': {
|
|
836
|
+
zh: 'LLM 未输出此维度({dim}),已置不适用',
|
|
837
|
+
en: 'LLM omitted dimension ({dim}); treated as N/A',
|
|
838
|
+
},
|
|
839
|
+
'cli.doctor.health.summary.label': {
|
|
840
|
+
zh: '健康度总览',
|
|
841
|
+
en: 'Health summary',
|
|
842
|
+
},
|
|
843
|
+
'cli.doctor.health.summary.message': {
|
|
844
|
+
zh: '{overall} | 维度: 健康 {h}/亚健康 {sh}/不健康 {bad}/不适用 {na} | finding: 错误 {err}/警告 {warn}/建议 {sug}',
|
|
845
|
+
en: '{overall} | dims: healthy {h}/sub {sh}/unhealthy {bad}/n-a {na} | findings: err {err}/warn {warn}/sug {sug}',
|
|
846
|
+
},
|
|
847
|
+
'cli.doctor.health.summary.no_top': {
|
|
848
|
+
zh: '完整详情见 --json 输出或 --html 报告',
|
|
849
|
+
en: 'Full detail in --json output or --html report',
|
|
850
|
+
},
|
|
851
|
+
// 7 内置维度 labelKey (id-based)
|
|
852
|
+
'cli.doctor.health.dim.trigger-boundary': { zh: '触发与边界', en: 'Trigger & boundary' },
|
|
853
|
+
'cli.doctor.health.dim.doc-clarity': { zh: '文档清晰', en: 'Documentation clarity' },
|
|
854
|
+
'cli.doctor.health.dim.instr-precision': { zh: '指令精确性', en: 'Instruction precision' },
|
|
855
|
+
'cli.doctor.health.dim.dependency': { zh: '依赖检查', en: 'Dependency check' },
|
|
856
|
+
'cli.doctor.health.dim.tool-conventions': { zh: '工具规范', en: 'Tool conventions' },
|
|
857
|
+
'cli.doctor.health.dim.security': { zh: '安全与合规', en: 'Security & compliance' },
|
|
858
|
+
'cli.doctor.health.dim.examples': { zh: '示例完备', en: 'Example completeness' },
|
|
961
859
|
// pass
|
|
962
860
|
'cli.doctor.skill_readable.pass': {
|
|
963
861
|
zh: 'skill 内容长度 {length} 字符',
|
|
@@ -1072,10 +970,10 @@ Examples:
|
|
|
1072
970
|
// ============ omk doctor CLI level ============
|
|
1073
971
|
'cli.help.doctor_usage': {
|
|
1074
972
|
zh: `
|
|
1075
|
-
oh-my-knowledge
|
|
973
|
+
oh-my-knowledge——omk doctor 健康度体检 (LLM-judge)
|
|
1076
974
|
|
|
1077
975
|
用法:
|
|
1078
|
-
omk doctor [path] 在 path
|
|
976
|
+
omk doctor [path] 在 path 上跑深度健康度体检
|
|
1079
977
|
omk doctor 在当前目录(或 ./skills)批量跑
|
|
1080
978
|
|
|
1081
979
|
参数:
|
|
@@ -1083,62 +981,70 @@ oh-my-knowledge — omk doctor 健康检查
|
|
|
1083
981
|
|
|
1084
982
|
选项:
|
|
1085
983
|
--json 把 DoctorReport 打到 stdout(CI 消费用)
|
|
1086
|
-
--gate 静默模式:
|
|
1087
|
-
--executor <name> executor
|
|
1088
|
-
--model <name>
|
|
984
|
+
--gate 静默模式: fatal 问题 exit 1; warnings_only 仍 exit 0, 仅 stderr 出摘要
|
|
985
|
+
--executor <name> LLM executor (默认 claude, 可换 anthropic-api/codex 等)
|
|
986
|
+
--model <name> 模型 (默认 sonnet)
|
|
1089
987
|
--samples <path> 显式指定评测用例文件
|
|
1090
|
-
--timeout <seconds>
|
|
988
|
+
--timeout <seconds> 单次 LLM 会话超时 (默认 600)
|
|
989
|
+
--html <path> 产出可视化 HTML 报告到 <path> (可与 --json 同时用)
|
|
990
|
+
--static-only 离线模式: 只跑静态检查 (skill 可读性 / 元数据 / 依赖 / samples 契约), 不调 LLM
|
|
1091
991
|
--lang <zh|en> 切换输出语言
|
|
1092
992
|
|
|
1093
993
|
示例:
|
|
1094
|
-
omk doctor
|
|
1095
|
-
omk doctor examples/code-review/skills --json
|
|
1096
|
-
omk doctor --gate; echo $?
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
-
|
|
1101
|
-
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
994
|
+
omk doctor my-skill --html /tmp/report.html # 深度体检 + HTML 报告 (默认)
|
|
995
|
+
omk doctor examples/code-review/skills --json > r.json # JSON 给 CI / 外部工具消费
|
|
996
|
+
omk doctor --gate; echo $? # CI 模式: fatal 问题 exit 1, 警告不阻断
|
|
997
|
+
omk doctor --static-only # 无 LLM 环境 (CI / 断网) 跑纯静态检查
|
|
998
|
+
|
|
999
|
+
doctor = LLM 健康度体检 (单次 LLM 会话):
|
|
1000
|
+
- 7 个内置维度: 触发与边界 / 文档清晰 / 指令精确性 / 依赖检查 / 工具规范 / 安全与合规 / 示例完备
|
|
1001
|
+
- 用户可扩展: 在自己代码里 registerHealthDimension(spec) 加自定义维度,
|
|
1002
|
+
会自动加入同一次 LLM 调用的 prompt + 报告 (顺序 = 注册顺序)
|
|
1003
|
+
- 每维度独立给 健康/亚健康/不健康/不适用 + findings + 改进建议
|
|
1004
|
+
- HTML 报告: 维度按 fail→warn→pass→skipped 排, 错误 finding 排前面
|
|
1005
|
+
|
|
1006
|
+
注: omk eval 内部仍跑静态 skill-readability/metadata/dependency
|
|
1007
|
+
gate 保护评测质量, 不走 omk doctor 这条 LLM 路径 (角色分离: doctor=审计, eval=评测)。
|
|
1008
|
+
LLM 连通性可用 omk eval --skip-connectivity 跳过 (--resume 时自动)。
|
|
1107
1009
|
`.trim() + '\n',
|
|
1108
1010
|
en: `
|
|
1109
|
-
oh-my-knowledge — omk doctor health
|
|
1011
|
+
oh-my-knowledge — omk doctor health audit (LLM-judge)
|
|
1110
1012
|
|
|
1111
1013
|
Usage:
|
|
1112
|
-
omk doctor [path] Run
|
|
1113
|
-
omk doctor Batch
|
|
1014
|
+
omk doctor [path] Run deep LLM-based health audit on path
|
|
1015
|
+
omk doctor Batch audit current dir (or ./skills)
|
|
1114
1016
|
|
|
1115
1017
|
Arguments:
|
|
1116
|
-
path A .md file, directory, or omit (= cwd). Directory
|
|
1018
|
+
path A .md file, directory, or omit (= cwd). Directory batches all skills.
|
|
1117
1019
|
|
|
1118
1020
|
Options:
|
|
1119
1021
|
--json Print DoctorReport JSON to stdout (CI-friendly)
|
|
1120
|
-
--gate Silent mode: exit
|
|
1121
|
-
--executor <name> executor
|
|
1122
|
-
--model <name> model name (
|
|
1022
|
+
--gate Silent mode: exit 1 only on fatal failure; warnings_only exits 0
|
|
1023
|
+
--executor <name> LLM executor (default 'claude'; switchable to anthropic-api/codex etc)
|
|
1024
|
+
--model <name> model name (default 'sonnet')
|
|
1123
1025
|
--samples <path> Explicit eval samples file
|
|
1124
|
-
--timeout <seconds>
|
|
1026
|
+
--timeout <seconds> LLM session timeout (default 600)
|
|
1027
|
+
--html <path> Also write a visual HTML report to <path> (combines with --json)
|
|
1028
|
+
--static-only Offline mode: run only static checks (readability / metadata / deps / samples contract); no LLM call
|
|
1125
1029
|
--lang <zh|en> Output language
|
|
1126
1030
|
|
|
1127
1031
|
Examples:
|
|
1128
|
-
omk doctor
|
|
1129
|
-
omk doctor examples/code-review/skills --json
|
|
1130
|
-
omk doctor --gate; echo $?
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
-
|
|
1135
|
-
|
|
1136
|
-
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1032
|
+
omk doctor my-skill --html /tmp/report.html # deep audit + HTML report (default)
|
|
1033
|
+
omk doctor examples/code-review/skills --json > r.json # JSON for CI / external tools
|
|
1034
|
+
omk doctor --gate; echo $? # CI mode: fatal failures exit 1; warnings do not block
|
|
1035
|
+
omk doctor --static-only # offline (CI / no LLM) static checks only
|
|
1036
|
+
|
|
1037
|
+
doctor = LLM health audit (single LLM session):
|
|
1038
|
+
- 7 builtin dimensions: trigger & boundary / doc clarity / instruction precision /
|
|
1039
|
+
dependency / tool conventions / security & compliance / example completeness
|
|
1040
|
+
- User-extensible: call registerHealthDimension(spec) in your code to add custom
|
|
1041
|
+
dimensions; they join the same LLM call's prompt + report (order = registration order)
|
|
1042
|
+
- Each dim graded healthy / sub-healthy / unhealthy / N-A + findings + suggestions
|
|
1043
|
+
- HTML report: dims sorted fail→warn→pass→skipped; errors first within each dim
|
|
1044
|
+
|
|
1045
|
+
Note: omk eval still runs static skill-readability/metadata/dependency gates
|
|
1046
|
+
internally (separate from this doctor command). Roles: doctor=audit, eval=evaluate.
|
|
1047
|
+
LLM connectivity for omk eval can be skipped with --skip-connectivity (auto on --resume).
|
|
1142
1048
|
`.trim() + '\n',
|
|
1143
1049
|
},
|
|
1144
1050
|
'cli.doctor.no_skill_found': {
|
|
@@ -1150,7 +1056,7 @@ checks cost nothing to run. LLM connectivity can be skipped with --skip-connecti
|
|
|
1150
1056
|
en: '✓ Using eval samples file: {path}',
|
|
1151
1057
|
},
|
|
1152
1058
|
'cli.doctor.gate_blocked': {
|
|
1153
|
-
zh: 'skill
|
|
1059
|
+
zh: 'skill 健康检查未通过,评测已中止。doctor 是评测必经环节,无 skip 选项——请修复上述问题后重跑。',
|
|
1154
1060
|
en: 'skill health check failed; evaluation aborted. doctor is mandatory and not skippable — fix the issues above and re-run.',
|
|
1155
1061
|
},
|
|
1156
1062
|
'cli.run.skip_connectivity_warning': {
|