oh-my-knowledge 0.48.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +18 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +28 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +30 -2
- package/dist/cli/commands/sample.d.ts +6 -3
- package/dist/cli/commands/sample.js +72 -68
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +1 -1
- package/dist/observability/inbox-view-model.js +19 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +208 -90
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -1,38 +1,29 @@
|
|
|
1
|
-
import { resolve, join, basename, dirname, extname } from 'node:path';
|
|
1
|
+
import { resolve, join, basename, dirname, extname, relative, sep } from 'node:path';
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, readdirSync, statSync, writeFileSync } from 'node:fs';
|
|
3
|
-
import yaml from 'js-yaml';
|
|
4
3
|
import { Args, Flags } from '@oclif/core';
|
|
5
4
|
import { LANG_FLAG, bilingual } from '../oclif/i18n.js';
|
|
6
5
|
import { BaseCommand } from '../oclif/base-command.js';
|
|
7
6
|
import { integerStringParser } from '../oclif/parsers.js';
|
|
8
7
|
import { CliExit } from '../lib/cli-exit.js';
|
|
9
8
|
import { tCli } from '../lib/i18n.js';
|
|
9
|
+
import { formatSampleGenerationFailureHint } from '../lib/generation-failure-hint.js';
|
|
10
|
+
import { resolveRuntimeSelection } from '../lib/runtime-defaults.js';
|
|
10
11
|
import { projectReportsDir, globalReportsDir } from '../../eval-core/measurement-dirs.js';
|
|
11
|
-
import { loadSamples,
|
|
12
|
+
import { loadSamples, listSampleFilesInDir } from '../../inputs/load-samples.js';
|
|
13
|
+
import { getSamplesArray, parseSampleDocument, stringifySampleDocument, writeFixedSamplesToSources, } from '../../inputs/sample-document.js';
|
|
12
14
|
import { defaultFlatSkillSamplesFile, defaultSkillLocalSamplesFile, findFlatSkillSamplesPath, findSkillSamplesPath, } from '../../inputs/sample-locator.js';
|
|
13
15
|
import { hashSample } from '../../eval-core/evaluation-reporting.js';
|
|
14
16
|
import { hashArtifactSource } from '../../inputs/content-hash.js';
|
|
15
|
-
|
|
16
|
-
|
|
17
|
+
import { shellQuoteArg } from '../../shared/shell-quote.js';
|
|
18
|
+
function userFacingPath(filePath) {
|
|
19
|
+
const rel = relative(process.cwd(), filePath);
|
|
20
|
+
if (rel && rel !== '..' && !rel.startsWith(`..${sep}`))
|
|
21
|
+
return rel;
|
|
22
|
+
return filePath;
|
|
17
23
|
}
|
|
18
|
-
function
|
|
19
|
-
|
|
20
|
-
}
|
|
21
|
-
function parseSampleDocument(filePath) {
|
|
22
|
-
const raw = readFileSync(filePath, 'utf-8');
|
|
23
|
-
return isYamlPath(filePath) ? parseYaml(raw) : JSON.parse(raw);
|
|
24
|
-
}
|
|
25
|
-
function getSamplesArray(document, filePath) {
|
|
26
|
-
if (Array.isArray(document))
|
|
27
|
-
return document;
|
|
28
|
-
if (isRecord(document) && Array.isArray(document.samples))
|
|
29
|
-
return document.samples;
|
|
30
|
-
throw new Error(`invalid samples file shape: ${filePath} (expected an array or an object with a 'samples' field)`);
|
|
31
|
-
}
|
|
32
|
-
function stringifySampleDocument(filePath, document) {
|
|
33
|
-
if (isYamlPath(filePath))
|
|
34
|
-
return yaml.dump(document, { lineWidth: -1, noRefs: true });
|
|
35
|
-
return JSON.stringify(document, null, 2);
|
|
24
|
+
export function sampleNextEvalCommand(resolved) {
|
|
25
|
+
const treatmentPath = resolved.isDirectorySkill ? resolved.skillDir : resolved.skillPath;
|
|
26
|
+
return `omk eval --control baseline --treatment ${shellQuoteArg(userFacingPath(treatmentPath))}`;
|
|
36
27
|
}
|
|
37
28
|
/** --append 合并:已有用例原样保留,新用例逐条接在后面;sample_id 撞已有(或本批已用)时
|
|
38
29
|
* 自动加 `-2`/`-3` 后缀去重。模型每次从 s001 重编号,撞 id 不代表内容重复,所以是改名保留
|
|
@@ -122,7 +113,7 @@ export function collectSampleDesignFailureIds(report, treatmentName) {
|
|
|
122
113
|
return ids;
|
|
123
114
|
}
|
|
124
115
|
export function assertFixReportMatchesCurrentInputs(params) {
|
|
125
|
-
const { report, treatmentName, currentContentHash, samples, sampleIds } = params;
|
|
116
|
+
const { report, treatmentName, currentContentHash, samples, samplesBaseDir, sampleIds, } = params;
|
|
126
117
|
const lang = params.lang ?? 'zh';
|
|
127
118
|
const issues = [];
|
|
128
119
|
// schemaVersion < 2 的报告:artifactHashes 是旧「仅 SKILL.md 正文文本」哈,与当前「整棵可分发树」哈
|
|
@@ -168,7 +159,7 @@ export function assertFixReportMatchesCurrentInputs(params) {
|
|
|
168
159
|
missingReportHashes.push(sampleId);
|
|
169
160
|
continue;
|
|
170
161
|
}
|
|
171
|
-
const currentSampleHash = hashSample(currentSample);
|
|
162
|
+
const currentSampleHash = hashSample(currentSample, samplesBaseDir);
|
|
172
163
|
if (expectedSampleHash !== currentSampleHash) {
|
|
173
164
|
mismatchedSamples.push(sampleId);
|
|
174
165
|
}
|
|
@@ -199,36 +190,15 @@ export function assertFixReportMatchesCurrentInputs(params) {
|
|
|
199
190
|
: 'Re-run omk eval first, then run omk sample --fix again.';
|
|
200
191
|
throw new Error([heading, ...issues, hint].join('\n'));
|
|
201
192
|
}
|
|
202
|
-
export
|
|
203
|
-
if (changedIds.size === 0)
|
|
204
|
-
return [];
|
|
205
|
-
const fixedById = new Map(samples.map((sample) => [sample.sample_id, sample]));
|
|
206
|
-
const idsByFile = new Map();
|
|
207
|
-
for (const sampleId of changedIds) {
|
|
208
|
-
const filePath = loaded.sampleSourceById[sampleId];
|
|
209
|
-
if (!filePath)
|
|
210
|
-
throw new Error(`sample ${sampleId} source file not found`);
|
|
211
|
-
const ids = idsByFile.get(filePath) ?? new Set();
|
|
212
|
-
ids.add(sampleId);
|
|
213
|
-
idsByFile.set(filePath, ids);
|
|
214
|
-
}
|
|
215
|
-
const written = [];
|
|
216
|
-
for (const [filePath, ids] of idsByFile.entries()) {
|
|
217
|
-
const document = parseSampleDocument(filePath);
|
|
218
|
-
const fileSamples = getSamplesArray(document, filePath);
|
|
219
|
-
const nextSamples = fileSamples.map((sample) => (ids.has(sample.sample_id) ? (fixedById.get(sample.sample_id) ?? sample) : sample));
|
|
220
|
-
const nextDocument = Array.isArray(document)
|
|
221
|
-
? nextSamples
|
|
222
|
-
: { ...document, samples: nextSamples };
|
|
223
|
-
writeFileSync(filePath, stringifySampleDocument(filePath, nextDocument));
|
|
224
|
-
written.push(filePath);
|
|
225
|
-
}
|
|
226
|
-
return written;
|
|
227
|
-
}
|
|
193
|
+
export { writeFixedSamplesToSources };
|
|
228
194
|
async function runSampleFix(args, flags, lang) {
|
|
229
195
|
const { fixSamples } = await import('../../authoring/sample-fixer.js');
|
|
230
196
|
const { createFileStore, createOverlayReportStore } = await import('../../server/report-store.js');
|
|
231
197
|
const model = flags.model;
|
|
198
|
+
const executorName = flags.executor;
|
|
199
|
+
if (!model || !executorName) {
|
|
200
|
+
throw new Error('internal error: sample fix requires runtime selection before execution');
|
|
201
|
+
}
|
|
232
202
|
const skillPath = args.skillPath;
|
|
233
203
|
if (!skillPath) {
|
|
234
204
|
console.error(lang === 'zh' ? '请指定 skill 路径,如: omk sample skills/my-skill/SKILL.md --fix' : 'Specify skill path: omk sample skills/my-skill/SKILL.md --fix');
|
|
@@ -294,6 +264,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
294
264
|
treatmentName,
|
|
295
265
|
currentContentHash,
|
|
296
266
|
samples,
|
|
267
|
+
samplesBaseDir: loadedSamples.baseDir,
|
|
297
268
|
sampleIds: sampleDesignIds,
|
|
298
269
|
lang,
|
|
299
270
|
});
|
|
@@ -304,7 +275,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
304
275
|
}
|
|
305
276
|
process.stderr.write(lang === 'zh' ? `🔧 发现 ${sampleDesignCount} 条 sample_design 失败,开始修复...\n` : `🔧 Found ${sampleDesignCount} sample_design failure(s), fixing...\n`);
|
|
306
277
|
const { createExecutor } = await import('../../executors/index.js');
|
|
307
|
-
const exec = createExecutor(
|
|
278
|
+
const exec = createExecutor(executorName);
|
|
308
279
|
const executorFn = async (opts) => {
|
|
309
280
|
const result = await exec({
|
|
310
281
|
model: opts.model,
|
|
@@ -313,7 +284,12 @@ async function runSampleFix(args, flags, lang) {
|
|
|
313
284
|
timeoutMs: opts.timeoutMs,
|
|
314
285
|
lean: opts.lean,
|
|
315
286
|
});
|
|
316
|
-
return {
|
|
287
|
+
return {
|
|
288
|
+
ok: result.ok,
|
|
289
|
+
text: result.output ?? '',
|
|
290
|
+
costUSD: result.costUSD,
|
|
291
|
+
costReported: result.costReportedByExecutor !== false,
|
|
292
|
+
};
|
|
317
293
|
};
|
|
318
294
|
const result = await fixSamples({
|
|
319
295
|
skillContent,
|
|
@@ -336,7 +312,9 @@ async function runSampleFix(args, flags, lang) {
|
|
|
336
312
|
process.stderr.write(lang === 'zh' ? ` ⚠ ${f.sampleId} 未修改${f.error ? `: ${f.error}` : ''}\n` : ` ⚠ ${f.sampleId} unchanged${f.error ? `: ${f.error}` : ''}\n`);
|
|
337
313
|
}
|
|
338
314
|
}
|
|
339
|
-
const cost = result.
|
|
315
|
+
const cost = result.costReported
|
|
316
|
+
? ` $${result.costUSD.toFixed(4)}`
|
|
317
|
+
: ` ${lang === 'zh' ? '成本未完整上报' : 'cost not fully reported'}`;
|
|
340
318
|
const outputTarget = writtenFiles.length === 0
|
|
341
319
|
? samplesInput
|
|
342
320
|
: writtenFiles.length === 1
|
|
@@ -349,6 +327,11 @@ async function runSampleFix(args, flags, lang) {
|
|
|
349
327
|
export async function runSampleFromTraces(flags, lang) {
|
|
350
328
|
const { queryObservationInbox, DEFAULT_OBSERVATIONS_DIR } = await import('../../observability/inbox.js');
|
|
351
329
|
const { generateSamplesFromTraces } = await import('../../authoring/generator.js');
|
|
330
|
+
const model = flags.model;
|
|
331
|
+
const executorName = flags.executor;
|
|
332
|
+
if (!model || !executorName) {
|
|
333
|
+
throw new Error('internal error: sample generation requires runtime selection before execution');
|
|
334
|
+
}
|
|
352
335
|
const obsDir = resolve(flags['observations-dir'] ?? DEFAULT_OBSERVATIONS_DIR);
|
|
353
336
|
if (!existsSync(obsDir)) {
|
|
354
337
|
console.error(lang === 'zh'
|
|
@@ -380,7 +363,7 @@ export async function runSampleFromTraces(flags, lang) {
|
|
|
380
363
|
? `🔭 发现 ${items.length} 个${flags.skill ? ` ${flags.skill} 的` : ''}失败信号,正在生成回归用例草稿...\n`
|
|
381
364
|
: `🔭 Found ${items.length}${flags.skill ? ` ${flags.skill}` : ''} failure signal(s); generating regression-sample drafts...\n`);
|
|
382
365
|
try {
|
|
383
|
-
const { samples, costUSD } = await generateSamplesFromTraces({ items, count, model
|
|
366
|
+
const { samples, costUSD } = await generateSamplesFromTraces({ items, count, model, executorName });
|
|
384
367
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
385
368
|
if (samples.length === 0) {
|
|
386
369
|
// The model conservatively skipped every signal (noise / unreproducible). That's a
|
|
@@ -399,7 +382,9 @@ export async function runSampleFromTraces(flags, lang) {
|
|
|
399
382
|
catch (err) {
|
|
400
383
|
if (err instanceof CliExit)
|
|
401
384
|
throw err;
|
|
402
|
-
|
|
385
|
+
const message = err.message;
|
|
386
|
+
console.error((lang === 'zh' ? `生成失败: ${message}` : `Generation failed: ${message}`)
|
|
387
|
+
+ formatSampleGenerationFailureHint(message, flags.executor, lang));
|
|
403
388
|
throw new CliExit(1);
|
|
404
389
|
}
|
|
405
390
|
}
|
|
@@ -427,6 +412,10 @@ async function runSample(args, flags, lang) {
|
|
|
427
412
|
? Math.max(1, Number(flags.count) || 5)
|
|
428
413
|
: undefined;
|
|
429
414
|
const model = flags.model;
|
|
415
|
+
const executorName = flags.executor;
|
|
416
|
+
if (!executorName) {
|
|
417
|
+
throw new Error('internal error: sample generation requires runtime selection before execution');
|
|
418
|
+
}
|
|
430
419
|
const focus = flags.focus || undefined;
|
|
431
420
|
if (focus) {
|
|
432
421
|
process.stderr.write(tCli('cli.gen.focus_applied', lang, { focus }));
|
|
@@ -439,6 +428,7 @@ async function runSample(args, flags, lang) {
|
|
|
439
428
|
}
|
|
440
429
|
const entries = readdirSync(skillDir);
|
|
441
430
|
let generated = 0;
|
|
431
|
+
let failed = 0;
|
|
442
432
|
for (const entry of entries) {
|
|
443
433
|
let name;
|
|
444
434
|
let skillPath;
|
|
@@ -477,7 +467,7 @@ async function runSample(args, flags, lang) {
|
|
|
477
467
|
}
|
|
478
468
|
try {
|
|
479
469
|
const skillContent = readFileSync(skillPath, 'utf-8');
|
|
480
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName
|
|
470
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName });
|
|
481
471
|
mkdirSync(dirname(samplesPath), { recursive: true });
|
|
482
472
|
writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
|
|
483
473
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
@@ -487,11 +477,17 @@ async function runSample(args, flags, lang) {
|
|
|
487
477
|
generated++;
|
|
488
478
|
}
|
|
489
479
|
catch (err) {
|
|
480
|
+
failed++;
|
|
481
|
+
const message = err.message;
|
|
490
482
|
process.stderr.write(tCli('cli.gen.skill_failed', lang, {
|
|
491
|
-
name, message:
|
|
483
|
+
name, message: `${message}${formatSampleGenerationFailureHint(message, flags.executor, lang)}`,
|
|
492
484
|
}));
|
|
493
485
|
}
|
|
494
486
|
}
|
|
487
|
+
if (failed > 0) {
|
|
488
|
+
console.error(tCli('cli.gen.batch_failed_summary', lang, { generated, failed }));
|
|
489
|
+
throw new CliExit(1);
|
|
490
|
+
}
|
|
495
491
|
if (generated === 0) {
|
|
496
492
|
console.log(tCli('cli.gen.batch_none_needed', lang));
|
|
497
493
|
}
|
|
@@ -531,7 +527,7 @@ async function runSample(args, flags, lang) {
|
|
|
531
527
|
}
|
|
532
528
|
// 已有用例文件:默认报错保护;--append 时追加(下面合并),不报错。
|
|
533
529
|
if (existingFile && !flags.append) {
|
|
534
|
-
console.error(tCli('cli.gen.samples_already_exists', lang));
|
|
530
|
+
console.error(tCli('cli.gen.samples_already_exists', lang, { command: sampleNextEvalCommand(resolved) }));
|
|
535
531
|
throw new CliExit(1);
|
|
536
532
|
}
|
|
537
533
|
if (count !== undefined) {
|
|
@@ -541,7 +537,7 @@ async function runSample(args, flags, lang) {
|
|
|
541
537
|
process.stderr.write(tCli('cli.gen.single_generating_auto', lang));
|
|
542
538
|
}
|
|
543
539
|
try {
|
|
544
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName
|
|
540
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName });
|
|
545
541
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
546
542
|
if (existingFile && flags.append) {
|
|
547
543
|
// 追加:读已有 → 合并(撞 id 去重)→ 保留原 json/yaml 格式与 wrapper 写回。
|
|
@@ -560,12 +556,15 @@ async function runSample(args, flags, lang) {
|
|
|
560
556
|
n: samples.length, path: outputPath, cost,
|
|
561
557
|
}));
|
|
562
558
|
}
|
|
563
|
-
console.log(tCli('cli.gen.review_hint', lang));
|
|
559
|
+
console.log(tCli('cli.gen.review_hint', lang, { command: sampleNextEvalCommand(resolved) }));
|
|
564
560
|
}
|
|
565
561
|
catch (err) {
|
|
566
562
|
if (err instanceof CliExit)
|
|
567
563
|
throw err;
|
|
568
|
-
|
|
564
|
+
const message = err.message;
|
|
565
|
+
console.error(tCli('cli.gen.failed', lang, {
|
|
566
|
+
message: `${message}${formatSampleGenerationFailureHint(message, flags.executor, lang)}`,
|
|
567
|
+
}));
|
|
569
568
|
throw new CliExit(1);
|
|
570
569
|
}
|
|
571
570
|
}
|
|
@@ -632,15 +631,14 @@ export default class Sample extends BaseCommand {
|
|
|
632
631
|
}),
|
|
633
632
|
model: Flags.string({
|
|
634
633
|
description: bilingual({
|
|
635
|
-
zh: '生成 LLM model
|
|
636
|
-
en: 'Generation LLM model name
|
|
634
|
+
zh: '生成 LLM model 名。Codex 自动读取本机配置;也可用 OMK_MODEL 设置环境偏好。',
|
|
635
|
+
en: 'Generation LLM model name. Codex reads the local configured model; OMK_MODEL sets an environment preference.',
|
|
637
636
|
}),
|
|
638
|
-
default: 'sonnet',
|
|
639
637
|
}),
|
|
640
638
|
executor: Flags.string({
|
|
641
639
|
description: bilingual({
|
|
642
|
-
zh: '
|
|
643
|
-
en: 'Executor name
|
|
640
|
+
zh: '执行器名。Codex 任务内自动用 codex;也可用 OMK_EXECUTOR 设置环境偏好。',
|
|
641
|
+
en: 'Executor name. Defaults to codex inside Codex tasks; OMK_EXECUTOR sets an environment preference.',
|
|
644
642
|
}),
|
|
645
643
|
}),
|
|
646
644
|
'skill-dir': Flags.string({
|
|
@@ -713,7 +711,13 @@ export default class Sample extends BaseCommand {
|
|
|
713
711
|
const { args, flags } = await this.parse(Sample);
|
|
714
712
|
const lang = this.lang;
|
|
715
713
|
await this.runWithCliExit(async () => {
|
|
716
|
-
|
|
714
|
+
const runtime = resolveRuntimeSelection({ executor: flags.executor, model: flags.model }, { lang });
|
|
715
|
+
await runSample(args, {
|
|
716
|
+
...flags,
|
|
717
|
+
executor: runtime.executor,
|
|
718
|
+
model: runtime.model,
|
|
719
|
+
lang,
|
|
720
|
+
}, lang);
|
|
717
721
|
});
|
|
718
722
|
}
|
|
719
723
|
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export interface CodexModelSuggestion {
|
|
2
|
+
model: string;
|
|
3
|
+
fromConfig: boolean;
|
|
4
|
+
configPath?: string;
|
|
5
|
+
}
|
|
6
|
+
export declare function getCodexModelSuggestion(env?: NodeJS.ProcessEnv): CodexModelSuggestion;
|
|
7
|
+
export declare function codexModelHint(lang: 'zh' | 'en', env?: NodeJS.ProcessEnv): string;
|
|
8
|
+
export declare function codexModelFlagValue(env?: NodeJS.ProcessEnv): string;
|
|
9
|
+
export declare function codexExecutorFlags(env?: NodeJS.ProcessEnv): string;
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
2
|
+
import { homedir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
const CODEX_MODEL_PLACEHOLDER = '<codex-model>';
|
|
5
|
+
function parseTopLevelCodexModel(configText) {
|
|
6
|
+
for (const line of configText.split(/\r?\n/)) {
|
|
7
|
+
if (/^\s*\[/.test(line))
|
|
8
|
+
return null;
|
|
9
|
+
const match = line.match(/^\s*model\s*=\s*(?:"([^"]+)"|'([^']+)')\s*(?:#.*)?$/);
|
|
10
|
+
const model = match?.[1] ?? match?.[2];
|
|
11
|
+
if (model)
|
|
12
|
+
return model;
|
|
13
|
+
}
|
|
14
|
+
return null;
|
|
15
|
+
}
|
|
16
|
+
export function getCodexModelSuggestion(env = process.env) {
|
|
17
|
+
const codexHome = env.CODEX_HOME || join(homedir(), '.codex');
|
|
18
|
+
const configPath = join(codexHome, 'config.toml');
|
|
19
|
+
if (existsSync(configPath)) {
|
|
20
|
+
try {
|
|
21
|
+
const model = parseTopLevelCodexModel(readFileSync(configPath, 'utf-8'));
|
|
22
|
+
if (model)
|
|
23
|
+
return { model, fromConfig: true, configPath };
|
|
24
|
+
}
|
|
25
|
+
catch { /* best-effort hint only */ }
|
|
26
|
+
}
|
|
27
|
+
return { model: CODEX_MODEL_PLACEHOLDER, fromConfig: false, configPath };
|
|
28
|
+
}
|
|
29
|
+
export function codexModelHint(lang, env = process.env) {
|
|
30
|
+
const suggestion = getCodexModelSuggestion(env);
|
|
31
|
+
if (suggestion.fromConfig) {
|
|
32
|
+
return lang === 'zh'
|
|
33
|
+
? `已按本机 Codex 配置 model=${suggestion.model} 填入。`
|
|
34
|
+
: `Filled from local Codex config model=${suggestion.model}.`;
|
|
35
|
+
}
|
|
36
|
+
return lang === 'zh'
|
|
37
|
+
? `把 ${CODEX_MODEL_PLACEHOLDER} 换成本机 Codex 可用模型;可查看 ${suggestion.configPath} 的 model。`
|
|
38
|
+
: `Replace ${CODEX_MODEL_PLACEHOLDER} with a model your local Codex can run; check model in ${suggestion.configPath}.`;
|
|
39
|
+
}
|
|
40
|
+
export function codexModelFlagValue(env = process.env) {
|
|
41
|
+
return getCodexModelSuggestion(env).model;
|
|
42
|
+
}
|
|
43
|
+
export function codexExecutorFlags(env = process.env) {
|
|
44
|
+
return `--executor codex --model ${codexModelFlagValue(env)}`;
|
|
45
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { tCli } from './i18n.js';
|
|
2
|
+
import { codexExecutorFlags, codexModelFlagValue, codexModelHint } from './codex-model-hint.js';
|
|
3
|
+
import { looksLikeLlmSetupFailure, looksLikeModelUnavailableFailure } from './llm-failure-classifier.js';
|
|
4
|
+
const CLAUDE_SAMPLE_EXECUTORS = new Set(['claude', 'claude-sdk']);
|
|
5
|
+
const CODEX_SAMPLE_EXECUTORS = new Set(['codex', 'codex-sdk']);
|
|
6
|
+
const OPENAI_API_SAMPLE_EXECUTORS = new Set(['openai-api']);
|
|
7
|
+
const ANTHROPIC_API_SAMPLE_EXECUTORS = new Set(['anthropic-api']);
|
|
8
|
+
export function formatSampleGenerationFailureHint(message, executorName, lang, env = process.env) {
|
|
9
|
+
const executor = executorName?.trim();
|
|
10
|
+
if (!executor)
|
|
11
|
+
return '';
|
|
12
|
+
if (!looksLikeLlmSetupFailure(message))
|
|
13
|
+
return '';
|
|
14
|
+
if (CLAUDE_SAMPLE_EXECUTORS.has(executor)) {
|
|
15
|
+
return tCli('cli.gen.claude_auth_hint', lang, {
|
|
16
|
+
codexFlags: codexExecutorFlags(env),
|
|
17
|
+
codexModelHint: codexModelHint(lang, env),
|
|
18
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
19
|
+
});
|
|
20
|
+
}
|
|
21
|
+
if (CODEX_SAMPLE_EXECUTORS.has(executor)) {
|
|
22
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
23
|
+
return tCli('cli.gen.codex_model_hint', lang, {
|
|
24
|
+
codexFlags: codexExecutorFlags(env),
|
|
25
|
+
codexExec: `codex exec -m ${codexModelFlagValue(env)} "hi"`,
|
|
26
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
27
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
28
|
+
codexModelHint: codexModelHint(lang, env),
|
|
29
|
+
});
|
|
30
|
+
}
|
|
31
|
+
return tCli('cli.gen.codex_auth_hint', lang, {
|
|
32
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
33
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
34
|
+
});
|
|
35
|
+
}
|
|
36
|
+
if (OPENAI_API_SAMPLE_EXECUTORS.has(executor)) {
|
|
37
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
38
|
+
return tCli('cli.gen.openai_api_model_hint', lang, {
|
|
39
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
40
|
+
codexFlags: codexExecutorFlags(env),
|
|
41
|
+
codexModelHint: codexModelHint(lang, env),
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
return tCli('cli.gen.openai_api_auth_hint', lang, {
|
|
45
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
46
|
+
codexFlags: codexExecutorFlags(env),
|
|
47
|
+
codexModelHint: codexModelHint(lang, env),
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
if (ANTHROPIC_API_SAMPLE_EXECUTORS.has(executor)) {
|
|
51
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
52
|
+
return tCli('cli.gen.anthropic_api_model_hint', lang, {
|
|
53
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
return tCli('cli.gen.anthropic_api_auth_hint', lang, {
|
|
57
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
return '';
|
|
61
|
+
}
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.common.deprecated_skill_samples_path' | 'cli.common.samples_not_found' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
2
|
+
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.common.deprecated_skill_samples_path' | 'cli.common.samples_not_found' | 'cli.common.samples_not_found_hint' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
3
3
|
export declare const commonDict: Record<CommonMessageKey, CliMessage>;
|
|
@@ -43,6 +43,10 @@ export const commonDict = {
|
|
|
43
43
|
zh: '未找到评测用例:{path}。请通过 --samples 指定文件,或创建项目级 eval-samples.json;单 treatment 目录 skill 请使用 <skill>/.omk/samples.json。',
|
|
44
44
|
en: 'Eval samples not found: {path}. Pass --samples, create project-level eval-samples.json, or use <skill>/.omk/samples.json for a single-treatment directory skill.',
|
|
45
45
|
},
|
|
46
|
+
'cli.common.samples_not_found_hint': {
|
|
47
|
+
zh: '下一步:先运行 {command} 生成用例,人工 review 后再重跑 omk eval。',
|
|
48
|
+
en: 'Next: run {command} to generate samples, review them, then re-run omk eval.',
|
|
49
|
+
},
|
|
46
50
|
'cli.update.new_version_available': {
|
|
47
51
|
zh: '\n💡 新版本可用:{old} → {new},运行 npm i -g oh-my-knowledge@latest 升级\n\n',
|
|
48
52
|
en: '\n💡 New version available: {old} → {new}, run npm i -g oh-my-knowledge@latest to upgrade\n\n',
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type GenMessageKey = 'cli.gen.skill_skipped_existing' | 'cli.gen.skill_generating' | 'cli.gen.skill_generating_auto' | 'cli.gen.skill_done' | 'cli.gen.skill_failed' | 'cli.gen.batch_none_needed' | 'cli.gen.batch_summary' | 'cli.gen.specify_skill_path' | 'cli.gen.samples_already_exists' | 'cli.gen.single_generating' | 'cli.gen.single_generating_auto' | 'cli.gen.single_done' | 'cli.gen.append_done' | 'cli.gen.append_single_only' | 'cli.gen.review_hint' | 'cli.gen.failed' | 'cli.gen.focus_applied';
|
|
2
|
+
export type GenMessageKey = 'cli.gen.skill_skipped_existing' | 'cli.gen.skill_generating' | 'cli.gen.skill_generating_auto' | 'cli.gen.skill_done' | 'cli.gen.skill_failed' | 'cli.gen.batch_none_needed' | 'cli.gen.batch_failed_summary' | 'cli.gen.batch_summary' | 'cli.gen.specify_skill_path' | 'cli.gen.samples_already_exists' | 'cli.gen.single_generating' | 'cli.gen.single_generating_auto' | 'cli.gen.single_done' | 'cli.gen.append_done' | 'cli.gen.append_single_only' | 'cli.gen.review_hint' | 'cli.gen.claude_auth_hint' | 'cli.gen.codex_auth_hint' | 'cli.gen.codex_model_hint' | 'cli.gen.openai_api_auth_hint' | 'cli.gen.openai_api_model_hint' | 'cli.gen.anthropic_api_auth_hint' | 'cli.gen.anthropic_api_model_hint' | 'cli.gen.failed' | 'cli.gen.focus_applied';
|
|
3
3
|
export declare const genDict: Record<GenMessageKey, CliMessage>;
|
|
@@ -23,17 +23,21 @@ export const genDict = {
|
|
|
23
23
|
zh: '没有需要生成的 eval-samples (所有 skill 都已有配对文件)',
|
|
24
24
|
en: 'No eval-samples need generating (all skills already have paired files)',
|
|
25
25
|
},
|
|
26
|
+
'cli.gen.batch_failed_summary': {
|
|
27
|
+
zh: '\n生成未完成:已生成 {generated} 份,失败 {failed} 份。请按上面的错误提示修复后重试;全部生成成功后再运行 omk eval --batch --dry-run。',
|
|
28
|
+
en: '\nGeneration incomplete: generated {generated}, failed {failed}. Fix the errors above and retry; once everything succeeds, run omk eval --batch --dry-run.',
|
|
29
|
+
},
|
|
26
30
|
'cli.gen.batch_summary': {
|
|
27
|
-
zh: '\n共生成 {n} 份 eval-samples
|
|
28
|
-
en: '\nGenerated {n} eval-samples files. Review
|
|
31
|
+
zh: '\n共生成 {n} 份 eval-samples。下一步:\n 1. 人工审查生成的评测用例,删掉不可信样本,补边界、反例\n 2. 预览任务:omk eval --batch --dry-run\n 3. 跑评测:omk eval --batch',
|
|
32
|
+
en: '\nGenerated {n} eval-samples files. Next steps:\n 1. Review the generated samples; drop weak cases and add boundary / counterexamples\n 2. Preview the task plan: omk eval --batch --dry-run\n 3. Run the eval: omk eval --batch',
|
|
29
33
|
},
|
|
30
34
|
'cli.gen.specify_skill_path': {
|
|
31
35
|
zh: '请指定 skill 文件路径, 例如: omk sample skills/my-skill.md',
|
|
32
36
|
en: 'Please specify a skill file path, e.g.: omk sample skills/my-skill.md',
|
|
33
37
|
},
|
|
34
38
|
'cli.gen.samples_already_exists': {
|
|
35
|
-
zh: 'eval-samples
|
|
36
|
-
en: 'eval-samples
|
|
39
|
+
zh: 'eval-samples 已存在。要补场景请加 --append(常配 --focus);要继续评测,运行:{command}',
|
|
40
|
+
en: 'eval-samples already exist. To add scenarios, use --append (often with --focus); to continue, run: {command}',
|
|
37
41
|
},
|
|
38
42
|
'cli.gen.single_generating': {
|
|
39
43
|
zh: '🔄 正在生成 {count} 条评测用例...\n',
|
|
@@ -56,8 +60,36 @@ export const genDict = {
|
|
|
56
60
|
en: '--append currently supports single-skill mode only; it cannot be combined with --batch / --from-traces / --fix.\n',
|
|
57
61
|
},
|
|
58
62
|
'cli.gen.review_hint': {
|
|
59
|
-
zh: '\n
|
|
60
|
-
en: '\
|
|
63
|
+
zh: '\n下一步:\n 1. 人工审查生成的评测用例,删掉不可信样本,补边界、反例\n 2. 预览任务:{command} --dry-run\n 3. 跑评测:{command}',
|
|
64
|
+
en: '\nNext steps:\n 1. Review the generated samples; drop weak cases and add boundary / counterexamples\n 2. Preview the task plan: {command} --dry-run\n 3. Run the eval: {command}',
|
|
65
|
+
},
|
|
66
|
+
'cli.gen.claude_auth_hint': {
|
|
67
|
+
zh: '\n提示:当前 sample 生成使用 Claude 系列执行器。先确认 Claude Code 已安装并完成登录;如果你在 Codex 环境里,可以改用:{codexFlags}({codexModelHint});如果要走 OpenAI API,可以改用:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
68
|
+
en: '\nHint: sample generation is using a Claude-based executor. First confirm Claude Code is installed and authenticated; in a Codex environment, switch to: {codexFlags} ({codexModelHint}); to use the OpenAI API path, switch to: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
69
|
+
},
|
|
70
|
+
'cli.gen.codex_auth_hint': {
|
|
71
|
+
zh: '\n提示:当前 sample 生成使用 Codex 系列执行器。先确认 Codex CLI / SDK 已安装并完成登录;如果你有 Claude Code 可用,可以改用:{claudeFlags};如果要走 OpenAI API,可以改用:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
72
|
+
en: '\nHint: sample generation is using a Codex-based executor. First confirm the Codex CLI / SDK is installed and authenticated; if Claude Code is available, switch to: {claudeFlags}; to use the OpenAI API path, switch to: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
73
|
+
},
|
|
74
|
+
'cli.gen.codex_model_hint': {
|
|
75
|
+
zh: '\n提示:当前 sample 生成使用 Codex 系列执行器,但模型名看起来不可用。可以先按本机 Codex 配置重试:{codexFlags}({codexModelHint});也可以先运行 `{codexExec}` 验证模型是否可用。若只是想先跑通,可以改用:{claudeFlags};或走 OpenAI API:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
76
|
+
en: '\nHint: sample generation is using a Codex-based executor, but the model name appears unavailable. Retry with the local Codex config model: {codexFlags} ({codexModelHint}); you can also run `{codexExec}` to verify the model. To just get a first run through, switch to: {claudeFlags}; or use the OpenAI API path: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
77
|
+
},
|
|
78
|
+
'cli.gen.openai_api_auth_hint': {
|
|
79
|
+
zh: '\n提示:当前 sample 生成使用 OpenAI API 执行器。请检查 OPENAI_API_KEY / OPENAI_BASE_URL 是否可用,并确认模型名对当前端点可用;如果只是想先跑通,也可以改用:{claudeFlags},或:{codexFlags}({codexModelHint})。',
|
|
80
|
+
en: '\nHint: sample generation is using the OpenAI API executor. Check OPENAI_API_KEY / OPENAI_BASE_URL and confirm the model is available on that endpoint; to just get a first run through, you can also switch to: {claudeFlags}, or: {codexFlags} ({codexModelHint}).',
|
|
81
|
+
},
|
|
82
|
+
'cli.gen.openai_api_model_hint': {
|
|
83
|
+
zh: '\n提示:当前 sample 生成使用 OpenAI API 执行器,但模型名看起来对当前端点不可用。请检查 --model、OPENAI_BASE_URL 与账号权限是否匹配;如果只是想先跑通,也可以改用:{claudeFlags},或:{codexFlags}({codexModelHint})。',
|
|
84
|
+
en: '\nHint: sample generation is using the OpenAI API executor, but the model name appears unavailable on the current endpoint. Check --model, OPENAI_BASE_URL, and account access; to just get a first run through, you can also switch to: {claudeFlags}, or: {codexFlags} ({codexModelHint}).',
|
|
85
|
+
},
|
|
86
|
+
'cli.gen.anthropic_api_auth_hint': {
|
|
87
|
+
zh: '\n提示:当前 sample 生成使用 Anthropic API 执行器。请检查 ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL 是否可用,并确认模型名对当前端点可用;如果你有 Claude Code 可用,也可以改用:{claudeFlags}。',
|
|
88
|
+
en: '\nHint: sample generation is using the Anthropic API executor. Check ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL and confirm the model is available on that endpoint; if Claude Code is available, you can also switch to: {claudeFlags}.',
|
|
89
|
+
},
|
|
90
|
+
'cli.gen.anthropic_api_model_hint': {
|
|
91
|
+
zh: '\n提示:当前 sample 生成使用 Anthropic API 执行器,但模型名看起来对当前端点不可用。请检查 --model、ANTHROPIC_BASE_URL 与账号权限是否匹配;如果你有 Claude Code 可用,也可以改用:{claudeFlags}。',
|
|
92
|
+
en: '\nHint: sample generation is using the Anthropic API executor, but the model name appears unavailable on the current endpoint. Check --model, ANTHROPIC_BASE_URL, and account access; if Claude Code is available, you can also switch to: {claudeFlags}.',
|
|
61
93
|
},
|
|
62
94
|
'cli.gen.failed': {
|
|
63
95
|
zh: '生成失败: {message}',
|
|
@@ -133,9 +133,9 @@ omk evolve——多轮自动迭代改进 skill
|
|
|
133
133
|
选项:
|
|
134
134
|
--rounds <n> 迭代轮数(默认:5)
|
|
135
135
|
--target <score> 目标分数
|
|
136
|
-
--model <name>
|
|
137
|
-
--improve-model <name> skill
|
|
138
|
-
--judge-models <executor:model>
|
|
136
|
+
--model <name> 任务执行模型,默认跟随 runtime;Codex 读取本机配置
|
|
137
|
+
--improve-model <name> skill 改写模型,默认沿用任务执行模型
|
|
138
|
+
--judge-models <executor:model> 单评委配置(默认跟随执行器;Codex 沿用被测模型)
|
|
139
139
|
|
|
140
140
|
示例:
|
|
141
141
|
omk evolve skills/code-review/SKILL.md
|
|
@@ -151,9 +151,9 @@ Usage:
|
|
|
151
151
|
Options:
|
|
152
152
|
--rounds <n> Iteration rounds (default: 5)
|
|
153
153
|
--target <score> Target score
|
|
154
|
-
--model <name> Task executor model
|
|
155
|
-
--improve-model <name> Skill rewriter model
|
|
156
|
-
--judge-models <executor:model> Single judge config (
|
|
154
|
+
--model <name> Task executor model; follows runtime (Codex reads local config)
|
|
155
|
+
--improve-model <name> Skill rewriter model; defaults to the task model
|
|
156
|
+
--judge-models <executor:model> Single judge config (follows executor; Codex reuses task model)
|
|
157
157
|
|
|
158
158
|
Examples:
|
|
159
159
|
omk evolve skills/code-review/SKILL.md
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_customize' | 'cli.init.
|
|
2
|
+
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_report' | 'cli.init.next_step_customize' | 'cli.init.note_skill_injection';
|
|
3
3
|
export declare const initDict: Record<InitMessageKey, CliMessage>;
|
|
@@ -11,19 +11,23 @@ export const initDict = {
|
|
|
11
11
|
// 跑出第一份报告是冷启动最该先发生的事;「换成你自己的」放到跑通之后。这也消除了
|
|
12
12
|
// 主 README「不用改任何文件」与旧 init「先编辑」的矛盾。
|
|
13
13
|
'cli.init.next_step_run': {
|
|
14
|
-
zh: ' 1. 直接跑通(无需先改任何文件):
|
|
15
|
-
en: ' 1. Run it as-is (no edits needed):
|
|
14
|
+
zh: ' 1. 直接跑通(无需先改任何文件):{command}',
|
|
15
|
+
en: ' 1. Run it as-is (no edits needed): {command}',
|
|
16
|
+
},
|
|
17
|
+
'cli.init.next_step_report': {
|
|
18
|
+
zh: ' 2. 看报告里的 verdict 和“下一步”:PROGRESS 才能发布;UNDERPOWERED / NOISE 先扩样到约 20 条以上后重跑。',
|
|
19
|
+
en: ' 2. Read the report verdict and Next line: PROGRESS can ship; UNDERPOWERED / NOISE means grow to roughly 20+ samples and re-run.',
|
|
16
20
|
},
|
|
17
21
|
'cli.init.next_step_executor': {
|
|
18
|
-
zh: '
|
|
19
|
-
en: ' The
|
|
22
|
+
zh: ' executor / judge 会按运行环境选择;Codex 任务自动使用本机 Codex 配置。也可用 OMK_EXECUTOR / OMK_MODEL 固定环境偏好,详见 https://oh-my-knowledge.pages.dev/zh/reference/executors。',
|
|
23
|
+
en: ' The executor / judge follow the runtime environment; Codex tasks use the local Codex configuration automatically. OMK_EXECUTOR / OMK_MODEL pin environment preferences. See https://oh-my-knowledge.pages.dev/reference/executors.',
|
|
20
24
|
},
|
|
21
25
|
'cli.init.next_step_customize': {
|
|
22
|
-
zh: '
|
|
23
|
-
en: '
|
|
26
|
+
zh: ' 3. 跑通后,替换为你自己的 skill 和 eval-samples.json;还没有用例时先运行 omk sample <skill-path>。',
|
|
27
|
+
en: ' 3. Once it runs, replace the starter skills and eval-samples.json with your own; if you have no samples yet, run omk sample <skill-path>.',
|
|
24
28
|
},
|
|
25
|
-
'cli.init.
|
|
26
|
-
zh: '
|
|
27
|
-
en: '
|
|
29
|
+
'cli.init.note_skill_injection': {
|
|
30
|
+
zh: ' 注:omk eval 会把 SKILL.md 作为 system prompt 注入;模板 frontmatter 只是方便同一目录复用为 agent skill。',
|
|
31
|
+
en: ' Note: omk eval injects SKILL.md as the system prompt; the starter frontmatter only helps reuse the same directory as an agent skill.',
|
|
28
32
|
},
|
|
29
33
|
};
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_verdict_next_step' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
|
|
2
|
+
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_verdict_next_step' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_promotable' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.codex_fallback_hint' | 'cli.run.codex_auth_hint' | 'cli.run.codex_model_hint' | 'cli.run.openai_api_auth_hint' | 'cli.run.openai_api_model_hint' | 'cli.run.anthropic_api_auth_hint' | 'cli.run.anthropic_api_model_hint' | 'cli.run.skip_connectivity_warning';
|
|
3
3
|
export declare const runDict: Record<RunMessageKey, CliMessage>;
|