oh-my-knowledge 0.47.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +19 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +44 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +35 -4
- package/dist/cli/commands/sample.d.ts +9 -3
- package/dist/cli/commands/sample.js +91 -74
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +2 -1
- package/dist/observability/inbox-view-model.js +20 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +227 -91
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/shell-quote.d.ts +2 -0
- package/dist/shared/shell-quote.js +7 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -1,38 +1,29 @@
|
|
|
1
|
-
import { resolve, join, basename, dirname, extname } from 'node:path';
|
|
1
|
+
import { resolve, join, basename, dirname, extname, relative, sep } from 'node:path';
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, readdirSync, statSync, writeFileSync } from 'node:fs';
|
|
3
|
-
import yaml from 'js-yaml';
|
|
4
3
|
import { Args, Flags } from '@oclif/core';
|
|
5
4
|
import { LANG_FLAG, bilingual } from '../oclif/i18n.js';
|
|
6
5
|
import { BaseCommand } from '../oclif/base-command.js';
|
|
7
6
|
import { integerStringParser } from '../oclif/parsers.js';
|
|
8
7
|
import { CliExit } from '../lib/cli-exit.js';
|
|
9
8
|
import { tCli } from '../lib/i18n.js';
|
|
9
|
+
import { formatSampleGenerationFailureHint } from '../lib/generation-failure-hint.js';
|
|
10
|
+
import { resolveRuntimeSelection } from '../lib/runtime-defaults.js';
|
|
10
11
|
import { projectReportsDir, globalReportsDir } from '../../eval-core/measurement-dirs.js';
|
|
11
|
-
import { loadSamples,
|
|
12
|
+
import { loadSamples, listSampleFilesInDir } from '../../inputs/load-samples.js';
|
|
13
|
+
import { getSamplesArray, parseSampleDocument, stringifySampleDocument, writeFixedSamplesToSources, } from '../../inputs/sample-document.js';
|
|
12
14
|
import { defaultFlatSkillSamplesFile, defaultSkillLocalSamplesFile, findFlatSkillSamplesPath, findSkillSamplesPath, } from '../../inputs/sample-locator.js';
|
|
13
15
|
import { hashSample } from '../../eval-core/evaluation-reporting.js';
|
|
14
16
|
import { hashArtifactSource } from '../../inputs/content-hash.js';
|
|
15
|
-
|
|
16
|
-
|
|
17
|
+
import { shellQuoteArg } from '../../shared/shell-quote.js';
|
|
18
|
+
function userFacingPath(filePath) {
|
|
19
|
+
const rel = relative(process.cwd(), filePath);
|
|
20
|
+
if (rel && rel !== '..' && !rel.startsWith(`..${sep}`))
|
|
21
|
+
return rel;
|
|
22
|
+
return filePath;
|
|
17
23
|
}
|
|
18
|
-
function
|
|
19
|
-
|
|
20
|
-
}
|
|
21
|
-
function parseSampleDocument(filePath) {
|
|
22
|
-
const raw = readFileSync(filePath, 'utf-8');
|
|
23
|
-
return isYamlPath(filePath) ? parseYaml(raw) : JSON.parse(raw);
|
|
24
|
-
}
|
|
25
|
-
function getSamplesArray(document, filePath) {
|
|
26
|
-
if (Array.isArray(document))
|
|
27
|
-
return document;
|
|
28
|
-
if (isRecord(document) && Array.isArray(document.samples))
|
|
29
|
-
return document.samples;
|
|
30
|
-
throw new Error(`invalid samples file shape: ${filePath} (expected an array or an object with a 'samples' field)`);
|
|
31
|
-
}
|
|
32
|
-
function stringifySampleDocument(filePath, document) {
|
|
33
|
-
if (isYamlPath(filePath))
|
|
34
|
-
return yaml.dump(document, { lineWidth: -1, noRefs: true });
|
|
35
|
-
return JSON.stringify(document, null, 2);
|
|
24
|
+
export function sampleNextEvalCommand(resolved) {
|
|
25
|
+
const treatmentPath = resolved.isDirectorySkill ? resolved.skillDir : resolved.skillPath;
|
|
26
|
+
return `omk eval --control baseline --treatment ${shellQuoteArg(userFacingPath(treatmentPath))}`;
|
|
36
27
|
}
|
|
37
28
|
/** --append 合并:已有用例原样保留,新用例逐条接在后面;sample_id 撞已有(或本批已用)时
|
|
38
29
|
* 自动加 `-2`/`-3` 后缀去重。模型每次从 s001 重编号,撞 id 不代表内容重复,所以是改名保留
|
|
@@ -122,7 +113,7 @@ export function collectSampleDesignFailureIds(report, treatmentName) {
|
|
|
122
113
|
return ids;
|
|
123
114
|
}
|
|
124
115
|
export function assertFixReportMatchesCurrentInputs(params) {
|
|
125
|
-
const { report, treatmentName, currentContentHash, samples, sampleIds } = params;
|
|
116
|
+
const { report, treatmentName, currentContentHash, samples, samplesBaseDir, sampleIds, } = params;
|
|
126
117
|
const lang = params.lang ?? 'zh';
|
|
127
118
|
const issues = [];
|
|
128
119
|
// schemaVersion < 2 的报告:artifactHashes 是旧「仅 SKILL.md 正文文本」哈,与当前「整棵可分发树」哈
|
|
@@ -168,7 +159,7 @@ export function assertFixReportMatchesCurrentInputs(params) {
|
|
|
168
159
|
missingReportHashes.push(sampleId);
|
|
169
160
|
continue;
|
|
170
161
|
}
|
|
171
|
-
const currentSampleHash = hashSample(currentSample);
|
|
162
|
+
const currentSampleHash = hashSample(currentSample, samplesBaseDir);
|
|
172
163
|
if (expectedSampleHash !== currentSampleHash) {
|
|
173
164
|
mismatchedSamples.push(sampleId);
|
|
174
165
|
}
|
|
@@ -199,36 +190,15 @@ export function assertFixReportMatchesCurrentInputs(params) {
|
|
|
199
190
|
: 'Re-run omk eval first, then run omk sample --fix again.';
|
|
200
191
|
throw new Error([heading, ...issues, hint].join('\n'));
|
|
201
192
|
}
|
|
202
|
-
export
|
|
203
|
-
if (changedIds.size === 0)
|
|
204
|
-
return [];
|
|
205
|
-
const fixedById = new Map(samples.map((sample) => [sample.sample_id, sample]));
|
|
206
|
-
const idsByFile = new Map();
|
|
207
|
-
for (const sampleId of changedIds) {
|
|
208
|
-
const filePath = loaded.sampleSourceById[sampleId];
|
|
209
|
-
if (!filePath)
|
|
210
|
-
throw new Error(`sample ${sampleId} source file not found`);
|
|
211
|
-
const ids = idsByFile.get(filePath) ?? new Set();
|
|
212
|
-
ids.add(sampleId);
|
|
213
|
-
idsByFile.set(filePath, ids);
|
|
214
|
-
}
|
|
215
|
-
const written = [];
|
|
216
|
-
for (const [filePath, ids] of idsByFile.entries()) {
|
|
217
|
-
const document = parseSampleDocument(filePath);
|
|
218
|
-
const fileSamples = getSamplesArray(document, filePath);
|
|
219
|
-
const nextSamples = fileSamples.map((sample) => (ids.has(sample.sample_id) ? (fixedById.get(sample.sample_id) ?? sample) : sample));
|
|
220
|
-
const nextDocument = Array.isArray(document)
|
|
221
|
-
? nextSamples
|
|
222
|
-
: { ...document, samples: nextSamples };
|
|
223
|
-
writeFileSync(filePath, stringifySampleDocument(filePath, nextDocument));
|
|
224
|
-
written.push(filePath);
|
|
225
|
-
}
|
|
226
|
-
return written;
|
|
227
|
-
}
|
|
193
|
+
export { writeFixedSamplesToSources };
|
|
228
194
|
async function runSampleFix(args, flags, lang) {
|
|
229
195
|
const { fixSamples } = await import('../../authoring/sample-fixer.js');
|
|
230
196
|
const { createFileStore, createOverlayReportStore } = await import('../../server/report-store.js');
|
|
231
197
|
const model = flags.model;
|
|
198
|
+
const executorName = flags.executor;
|
|
199
|
+
if (!model || !executorName) {
|
|
200
|
+
throw new Error('internal error: sample fix requires runtime selection before execution');
|
|
201
|
+
}
|
|
232
202
|
const skillPath = args.skillPath;
|
|
233
203
|
if (!skillPath) {
|
|
234
204
|
console.error(lang === 'zh' ? '请指定 skill 路径,如: omk sample skills/my-skill/SKILL.md --fix' : 'Specify skill path: omk sample skills/my-skill/SKILL.md --fix');
|
|
@@ -294,6 +264,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
294
264
|
treatmentName,
|
|
295
265
|
currentContentHash,
|
|
296
266
|
samples,
|
|
267
|
+
samplesBaseDir: loadedSamples.baseDir,
|
|
297
268
|
sampleIds: sampleDesignIds,
|
|
298
269
|
lang,
|
|
299
270
|
});
|
|
@@ -304,7 +275,7 @@ async function runSampleFix(args, flags, lang) {
|
|
|
304
275
|
}
|
|
305
276
|
process.stderr.write(lang === 'zh' ? `🔧 发现 ${sampleDesignCount} 条 sample_design 失败,开始修复...\n` : `🔧 Found ${sampleDesignCount} sample_design failure(s), fixing...\n`);
|
|
306
277
|
const { createExecutor } = await import('../../executors/index.js');
|
|
307
|
-
const exec = createExecutor(
|
|
278
|
+
const exec = createExecutor(executorName);
|
|
308
279
|
const executorFn = async (opts) => {
|
|
309
280
|
const result = await exec({
|
|
310
281
|
model: opts.model,
|
|
@@ -313,7 +284,12 @@ async function runSampleFix(args, flags, lang) {
|
|
|
313
284
|
timeoutMs: opts.timeoutMs,
|
|
314
285
|
lean: opts.lean,
|
|
315
286
|
});
|
|
316
|
-
return {
|
|
287
|
+
return {
|
|
288
|
+
ok: result.ok,
|
|
289
|
+
text: result.output ?? '',
|
|
290
|
+
costUSD: result.costUSD,
|
|
291
|
+
costReported: result.costReportedByExecutor !== false,
|
|
292
|
+
};
|
|
317
293
|
};
|
|
318
294
|
const result = await fixSamples({
|
|
319
295
|
skillContent,
|
|
@@ -336,7 +312,9 @@ async function runSampleFix(args, flags, lang) {
|
|
|
336
312
|
process.stderr.write(lang === 'zh' ? ` ⚠ ${f.sampleId} 未修改${f.error ? `: ${f.error}` : ''}\n` : ` ⚠ ${f.sampleId} unchanged${f.error ? `: ${f.error}` : ''}\n`);
|
|
337
313
|
}
|
|
338
314
|
}
|
|
339
|
-
const cost = result.
|
|
315
|
+
const cost = result.costReported
|
|
316
|
+
? ` $${result.costUSD.toFixed(4)}`
|
|
317
|
+
: ` ${lang === 'zh' ? '成本未完整上报' : 'cost not fully reported'}`;
|
|
340
318
|
const outputTarget = writtenFiles.length === 0
|
|
341
319
|
? samplesInput
|
|
342
320
|
: writtenFiles.length === 1
|
|
@@ -346,9 +324,14 @@ async function runSampleFix(args, flags, lang) {
|
|
|
346
324
|
? `\n🔧 修复完成: ${result.fixedCount}/${sampleDesignCount} 条已修复 → ${outputTarget}${cost}\n`
|
|
347
325
|
: `\n🔧 Fix complete: ${result.fixedCount}/${sampleDesignCount} fixed → ${outputTarget}${cost}\n`);
|
|
348
326
|
}
|
|
349
|
-
async function runSampleFromTraces(flags, lang) {
|
|
327
|
+
export async function runSampleFromTraces(flags, lang) {
|
|
350
328
|
const { queryObservationInbox, DEFAULT_OBSERVATIONS_DIR } = await import('../../observability/inbox.js');
|
|
351
329
|
const { generateSamplesFromTraces } = await import('../../authoring/generator.js');
|
|
330
|
+
const model = flags.model;
|
|
331
|
+
const executorName = flags.executor;
|
|
332
|
+
if (!model || !executorName) {
|
|
333
|
+
throw new Error('internal error: sample generation requires runtime selection before execution');
|
|
334
|
+
}
|
|
352
335
|
const obsDir = resolve(flags['observations-dir'] ?? DEFAULT_OBSERVATIONS_DIR);
|
|
353
336
|
if (!existsSync(obsDir)) {
|
|
354
337
|
console.error(lang === 'zh'
|
|
@@ -358,11 +341,14 @@ async function runSampleFromTraces(flags, lang) {
|
|
|
358
341
|
}
|
|
359
342
|
// Drop noise-tier signals up front: they're exactly what the generator is told to
|
|
360
343
|
// skip, so filtering here avoids feeding junk to the LLM and keeps the no-op path clean.
|
|
361
|
-
|
|
344
|
+
let items = queryObservationInbox(obsDir).filter((it) => it.severity !== 'noise');
|
|
345
|
+
if (flags.skill) {
|
|
346
|
+
items = items.filter((it) => it.skillName === flags.skill);
|
|
347
|
+
}
|
|
362
348
|
if (items.length === 0) {
|
|
363
349
|
process.stderr.write(lang === 'zh'
|
|
364
|
-
? `✅ ${obsDir} 没有可回流的失败信号(噪声级已跳过)\n`
|
|
365
|
-
: `✅ No recyclable failure signals in ${obsDir} (noise-level skipped)\n`);
|
|
350
|
+
? `✅ ${obsDir}${flags.skill ? ` 中 ${flags.skill}` : ''} 没有可回流的失败信号(噪声级已跳过)\n`
|
|
351
|
+
: `✅ No recyclable failure signals${flags.skill ? ` for ${flags.skill}` : ''} in ${obsDir} (noise-level skipped)\n`);
|
|
366
352
|
return;
|
|
367
353
|
}
|
|
368
354
|
const outPath = join(obsDir, 'sample-drafts.json');
|
|
@@ -374,10 +360,10 @@ async function runSampleFromTraces(flags, lang) {
|
|
|
374
360
|
}
|
|
375
361
|
const count = flags.count !== undefined ? Math.max(1, Number(flags.count) || 5) : undefined;
|
|
376
362
|
process.stderr.write(lang === 'zh'
|
|
377
|
-
? `🔭 发现 ${items.length}
|
|
378
|
-
: `🔭 Found ${items.length} failure signal(s); generating regression-sample drafts...\n`);
|
|
363
|
+
? `🔭 发现 ${items.length} 个${flags.skill ? ` ${flags.skill} 的` : ''}失败信号,正在生成回归用例草稿...\n`
|
|
364
|
+
: `🔭 Found ${items.length}${flags.skill ? ` ${flags.skill}` : ''} failure signal(s); generating regression-sample drafts...\n`);
|
|
379
365
|
try {
|
|
380
|
-
const { samples, costUSD } = await generateSamplesFromTraces({ items, count, model
|
|
366
|
+
const { samples, costUSD } = await generateSamplesFromTraces({ items, count, model, executorName });
|
|
381
367
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
382
368
|
if (samples.length === 0) {
|
|
383
369
|
// The model conservatively skipped every signal (noise / unreproducible). That's a
|
|
@@ -396,11 +382,17 @@ async function runSampleFromTraces(flags, lang) {
|
|
|
396
382
|
catch (err) {
|
|
397
383
|
if (err instanceof CliExit)
|
|
398
384
|
throw err;
|
|
399
|
-
|
|
385
|
+
const message = err.message;
|
|
386
|
+
console.error((lang === 'zh' ? `生成失败: ${message}` : `Generation failed: ${message}`)
|
|
387
|
+
+ formatSampleGenerationFailureHint(message, flags.executor, lang));
|
|
400
388
|
throw new CliExit(1);
|
|
401
389
|
}
|
|
402
390
|
}
|
|
403
391
|
async function runSample(args, flags, lang) {
|
|
392
|
+
if (flags.skill && !flags['from-traces']) {
|
|
393
|
+
console.error(lang === 'zh' ? '--skill 仅支持 --from-traces 模式。' : '--skill is only supported with --from-traces.');
|
|
394
|
+
throw new CliExit(2);
|
|
395
|
+
}
|
|
404
396
|
// --append 目前只在单 skill 生成路径实现;batch / from-traces / fix 不处理它,
|
|
405
397
|
// 静默忽略会误导(用户以为在追加,实际没有)。提前互斥校验,明确报错。
|
|
406
398
|
if (flags.append && (flags.batch || flags['from-traces'] || flags.fix)) {
|
|
@@ -420,6 +412,10 @@ async function runSample(args, flags, lang) {
|
|
|
420
412
|
? Math.max(1, Number(flags.count) || 5)
|
|
421
413
|
: undefined;
|
|
422
414
|
const model = flags.model;
|
|
415
|
+
const executorName = flags.executor;
|
|
416
|
+
if (!executorName) {
|
|
417
|
+
throw new Error('internal error: sample generation requires runtime selection before execution');
|
|
418
|
+
}
|
|
423
419
|
const focus = flags.focus || undefined;
|
|
424
420
|
if (focus) {
|
|
425
421
|
process.stderr.write(tCli('cli.gen.focus_applied', lang, { focus }));
|
|
@@ -432,6 +428,7 @@ async function runSample(args, flags, lang) {
|
|
|
432
428
|
}
|
|
433
429
|
const entries = readdirSync(skillDir);
|
|
434
430
|
let generated = 0;
|
|
431
|
+
let failed = 0;
|
|
435
432
|
for (const entry of entries) {
|
|
436
433
|
let name;
|
|
437
434
|
let skillPath;
|
|
@@ -470,7 +467,7 @@ async function runSample(args, flags, lang) {
|
|
|
470
467
|
}
|
|
471
468
|
try {
|
|
472
469
|
const skillContent = readFileSync(skillPath, 'utf-8');
|
|
473
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName
|
|
470
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName });
|
|
474
471
|
mkdirSync(dirname(samplesPath), { recursive: true });
|
|
475
472
|
writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
|
|
476
473
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
@@ -480,11 +477,17 @@ async function runSample(args, flags, lang) {
|
|
|
480
477
|
generated++;
|
|
481
478
|
}
|
|
482
479
|
catch (err) {
|
|
480
|
+
failed++;
|
|
481
|
+
const message = err.message;
|
|
483
482
|
process.stderr.write(tCli('cli.gen.skill_failed', lang, {
|
|
484
|
-
name, message:
|
|
483
|
+
name, message: `${message}${formatSampleGenerationFailureHint(message, flags.executor, lang)}`,
|
|
485
484
|
}));
|
|
486
485
|
}
|
|
487
486
|
}
|
|
487
|
+
if (failed > 0) {
|
|
488
|
+
console.error(tCli('cli.gen.batch_failed_summary', lang, { generated, failed }));
|
|
489
|
+
throw new CliExit(1);
|
|
490
|
+
}
|
|
488
491
|
if (generated === 0) {
|
|
489
492
|
console.log(tCli('cli.gen.batch_none_needed', lang));
|
|
490
493
|
}
|
|
@@ -524,7 +527,7 @@ async function runSample(args, flags, lang) {
|
|
|
524
527
|
}
|
|
525
528
|
// 已有用例文件:默认报错保护;--append 时追加(下面合并),不报错。
|
|
526
529
|
if (existingFile && !flags.append) {
|
|
527
|
-
console.error(tCli('cli.gen.samples_already_exists', lang));
|
|
530
|
+
console.error(tCli('cli.gen.samples_already_exists', lang, { command: sampleNextEvalCommand(resolved) }));
|
|
528
531
|
throw new CliExit(1);
|
|
529
532
|
}
|
|
530
533
|
if (count !== undefined) {
|
|
@@ -534,7 +537,7 @@ async function runSample(args, flags, lang) {
|
|
|
534
537
|
process.stderr.write(tCli('cli.gen.single_generating_auto', lang));
|
|
535
538
|
}
|
|
536
539
|
try {
|
|
537
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName
|
|
540
|
+
const { samples, costUSD } = await generateSamples({ skillContent, count, model, focus, noMock: flags['no-mock'], executorName });
|
|
538
541
|
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
539
542
|
if (existingFile && flags.append) {
|
|
540
543
|
// 追加:读已有 → 合并(撞 id 去重)→ 保留原 json/yaml 格式与 wrapper 写回。
|
|
@@ -553,12 +556,15 @@ async function runSample(args, flags, lang) {
|
|
|
553
556
|
n: samples.length, path: outputPath, cost,
|
|
554
557
|
}));
|
|
555
558
|
}
|
|
556
|
-
console.log(tCli('cli.gen.review_hint', lang));
|
|
559
|
+
console.log(tCli('cli.gen.review_hint', lang, { command: sampleNextEvalCommand(resolved) }));
|
|
557
560
|
}
|
|
558
561
|
catch (err) {
|
|
559
562
|
if (err instanceof CliExit)
|
|
560
563
|
throw err;
|
|
561
|
-
|
|
564
|
+
const message = err.message;
|
|
565
|
+
console.error(tCli('cli.gen.failed', lang, {
|
|
566
|
+
message: `${message}${formatSampleGenerationFailureHint(message, flags.executor, lang)}`,
|
|
567
|
+
}));
|
|
562
568
|
throw new CliExit(1);
|
|
563
569
|
}
|
|
564
570
|
}
|
|
@@ -625,15 +631,14 @@ export default class Sample extends BaseCommand {
|
|
|
625
631
|
}),
|
|
626
632
|
model: Flags.string({
|
|
627
633
|
description: bilingual({
|
|
628
|
-
zh: '生成 LLM model
|
|
629
|
-
en: 'Generation LLM model name
|
|
634
|
+
zh: '生成 LLM model 名。Codex 自动读取本机配置;也可用 OMK_MODEL 设置环境偏好。',
|
|
635
|
+
en: 'Generation LLM model name. Codex reads the local configured model; OMK_MODEL sets an environment preference.',
|
|
630
636
|
}),
|
|
631
|
-
default: 'sonnet',
|
|
632
637
|
}),
|
|
633
638
|
executor: Flags.string({
|
|
634
639
|
description: bilingual({
|
|
635
|
-
zh: '
|
|
636
|
-
en: 'Executor name
|
|
640
|
+
zh: '执行器名。Codex 任务内自动用 codex;也可用 OMK_EXECUTOR 设置环境偏好。',
|
|
641
|
+
en: 'Executor name. Defaults to codex inside Codex tasks; OMK_EXECUTOR sets an environment preference.',
|
|
637
642
|
}),
|
|
638
643
|
}),
|
|
639
644
|
'skill-dir': Flags.string({
|
|
@@ -695,12 +700,24 @@ export default class Sample extends BaseCommand {
|
|
|
695
700
|
en: 'Observe inbox dir (from-traces mode), default project .omk/observe-inbox.',
|
|
696
701
|
}),
|
|
697
702
|
}),
|
|
703
|
+
skill: Flags.string({
|
|
704
|
+
description: bilingual({
|
|
705
|
+
zh: '仅从指定 skill 的 observe inbox 信号生成草稿(仅 from-traces 模式用)。',
|
|
706
|
+
en: 'Only draft from observe-inbox signals for the specified skill (from-traces mode only).',
|
|
707
|
+
}),
|
|
708
|
+
}),
|
|
698
709
|
};
|
|
699
710
|
async run() {
|
|
700
711
|
const { args, flags } = await this.parse(Sample);
|
|
701
712
|
const lang = this.lang;
|
|
702
713
|
await this.runWithCliExit(async () => {
|
|
703
|
-
|
|
714
|
+
const runtime = resolveRuntimeSelection({ executor: flags.executor, model: flags.model }, { lang });
|
|
715
|
+
await runSample(args, {
|
|
716
|
+
...flags,
|
|
717
|
+
executor: runtime.executor,
|
|
718
|
+
model: runtime.model,
|
|
719
|
+
lang,
|
|
720
|
+
}, lang);
|
|
704
721
|
});
|
|
705
722
|
}
|
|
706
723
|
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export interface CodexModelSuggestion {
|
|
2
|
+
model: string;
|
|
3
|
+
fromConfig: boolean;
|
|
4
|
+
configPath?: string;
|
|
5
|
+
}
|
|
6
|
+
export declare function getCodexModelSuggestion(env?: NodeJS.ProcessEnv): CodexModelSuggestion;
|
|
7
|
+
export declare function codexModelHint(lang: 'zh' | 'en', env?: NodeJS.ProcessEnv): string;
|
|
8
|
+
export declare function codexModelFlagValue(env?: NodeJS.ProcessEnv): string;
|
|
9
|
+
export declare function codexExecutorFlags(env?: NodeJS.ProcessEnv): string;
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
2
|
+
import { homedir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
const CODEX_MODEL_PLACEHOLDER = '<codex-model>';
|
|
5
|
+
function parseTopLevelCodexModel(configText) {
|
|
6
|
+
for (const line of configText.split(/\r?\n/)) {
|
|
7
|
+
if (/^\s*\[/.test(line))
|
|
8
|
+
return null;
|
|
9
|
+
const match = line.match(/^\s*model\s*=\s*(?:"([^"]+)"|'([^']+)')\s*(?:#.*)?$/);
|
|
10
|
+
const model = match?.[1] ?? match?.[2];
|
|
11
|
+
if (model)
|
|
12
|
+
return model;
|
|
13
|
+
}
|
|
14
|
+
return null;
|
|
15
|
+
}
|
|
16
|
+
export function getCodexModelSuggestion(env = process.env) {
|
|
17
|
+
const codexHome = env.CODEX_HOME || join(homedir(), '.codex');
|
|
18
|
+
const configPath = join(codexHome, 'config.toml');
|
|
19
|
+
if (existsSync(configPath)) {
|
|
20
|
+
try {
|
|
21
|
+
const model = parseTopLevelCodexModel(readFileSync(configPath, 'utf-8'));
|
|
22
|
+
if (model)
|
|
23
|
+
return { model, fromConfig: true, configPath };
|
|
24
|
+
}
|
|
25
|
+
catch { /* best-effort hint only */ }
|
|
26
|
+
}
|
|
27
|
+
return { model: CODEX_MODEL_PLACEHOLDER, fromConfig: false, configPath };
|
|
28
|
+
}
|
|
29
|
+
export function codexModelHint(lang, env = process.env) {
|
|
30
|
+
const suggestion = getCodexModelSuggestion(env);
|
|
31
|
+
if (suggestion.fromConfig) {
|
|
32
|
+
return lang === 'zh'
|
|
33
|
+
? `已按本机 Codex 配置 model=${suggestion.model} 填入。`
|
|
34
|
+
: `Filled from local Codex config model=${suggestion.model}.`;
|
|
35
|
+
}
|
|
36
|
+
return lang === 'zh'
|
|
37
|
+
? `把 ${CODEX_MODEL_PLACEHOLDER} 换成本机 Codex 可用模型;可查看 ${suggestion.configPath} 的 model。`
|
|
38
|
+
: `Replace ${CODEX_MODEL_PLACEHOLDER} with a model your local Codex can run; check model in ${suggestion.configPath}.`;
|
|
39
|
+
}
|
|
40
|
+
export function codexModelFlagValue(env = process.env) {
|
|
41
|
+
return getCodexModelSuggestion(env).model;
|
|
42
|
+
}
|
|
43
|
+
export function codexExecutorFlags(env = process.env) {
|
|
44
|
+
return `--executor codex --model ${codexModelFlagValue(env)}`;
|
|
45
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { tCli } from './i18n.js';
|
|
2
|
+
import { codexExecutorFlags, codexModelFlagValue, codexModelHint } from './codex-model-hint.js';
|
|
3
|
+
import { looksLikeLlmSetupFailure, looksLikeModelUnavailableFailure } from './llm-failure-classifier.js';
|
|
4
|
+
const CLAUDE_SAMPLE_EXECUTORS = new Set(['claude', 'claude-sdk']);
|
|
5
|
+
const CODEX_SAMPLE_EXECUTORS = new Set(['codex', 'codex-sdk']);
|
|
6
|
+
const OPENAI_API_SAMPLE_EXECUTORS = new Set(['openai-api']);
|
|
7
|
+
const ANTHROPIC_API_SAMPLE_EXECUTORS = new Set(['anthropic-api']);
|
|
8
|
+
export function formatSampleGenerationFailureHint(message, executorName, lang, env = process.env) {
|
|
9
|
+
const executor = executorName?.trim();
|
|
10
|
+
if (!executor)
|
|
11
|
+
return '';
|
|
12
|
+
if (!looksLikeLlmSetupFailure(message))
|
|
13
|
+
return '';
|
|
14
|
+
if (CLAUDE_SAMPLE_EXECUTORS.has(executor)) {
|
|
15
|
+
return tCli('cli.gen.claude_auth_hint', lang, {
|
|
16
|
+
codexFlags: codexExecutorFlags(env),
|
|
17
|
+
codexModelHint: codexModelHint(lang, env),
|
|
18
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
19
|
+
});
|
|
20
|
+
}
|
|
21
|
+
if (CODEX_SAMPLE_EXECUTORS.has(executor)) {
|
|
22
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
23
|
+
return tCli('cli.gen.codex_model_hint', lang, {
|
|
24
|
+
codexFlags: codexExecutorFlags(env),
|
|
25
|
+
codexExec: `codex exec -m ${codexModelFlagValue(env)} "hi"`,
|
|
26
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
27
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
28
|
+
codexModelHint: codexModelHint(lang, env),
|
|
29
|
+
});
|
|
30
|
+
}
|
|
31
|
+
return tCli('cli.gen.codex_auth_hint', lang, {
|
|
32
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
33
|
+
openaiFlags: '--executor openai-api --model <openai-model>',
|
|
34
|
+
});
|
|
35
|
+
}
|
|
36
|
+
if (OPENAI_API_SAMPLE_EXECUTORS.has(executor)) {
|
|
37
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
38
|
+
return tCli('cli.gen.openai_api_model_hint', lang, {
|
|
39
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
40
|
+
codexFlags: codexExecutorFlags(env),
|
|
41
|
+
codexModelHint: codexModelHint(lang, env),
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
return tCli('cli.gen.openai_api_auth_hint', lang, {
|
|
45
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
46
|
+
codexFlags: codexExecutorFlags(env),
|
|
47
|
+
codexModelHint: codexModelHint(lang, env),
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
if (ANTHROPIC_API_SAMPLE_EXECUTORS.has(executor)) {
|
|
51
|
+
if (looksLikeModelUnavailableFailure(message)) {
|
|
52
|
+
return tCli('cli.gen.anthropic_api_model_hint', lang, {
|
|
53
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
return tCli('cli.gen.anthropic_api_auth_hint', lang, {
|
|
57
|
+
claudeFlags: '--executor claude --model sonnet',
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
return '';
|
|
61
|
+
}
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.common.deprecated_skill_samples_path' | 'cli.common.samples_not_found' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
2
|
+
export type CommonMessageKey = 'cli.common.unknown_domain' | 'cli.common.error_prefix' | 'cli.common.skill_dir_not_found' | 'cli.common.skill_file_not_found' | 'cli.common.skill_dir_no_skill_md' | 'cli.common.report_not_found' | 'cli.common.no_judge_model' | 'cli.common.judge_models_single_only' | 'cli.common.warn_load_samples_failed' | 'cli.common.deprecated_skill_samples_path' | 'cli.common.samples_not_found' | 'cli.common.samples_not_found_hint' | 'cli.update.new_version_available' | 'cli.update.box_title' | 'cli.update.box_version_line' | 'cli.update.box_upgrade_line' | 'cli.update.box_silence_line' | 'cli.observe.view_hint' | 'cli.observe.observation_recorded' | 'cli.observe.production_gap' | 'cli.studio.started' | 'cli.studio.stop_hint' | 'cli.studio.open_failed' | 'cli.doctor.no_skill_found' | 'cli.doctor.progress_skill_start' | 'cli.doctor.progress_skill_done';
|
|
3
3
|
export declare const commonDict: Record<CommonMessageKey, CliMessage>;
|
|
@@ -43,6 +43,10 @@ export const commonDict = {
|
|
|
43
43
|
zh: '未找到评测用例:{path}。请通过 --samples 指定文件,或创建项目级 eval-samples.json;单 treatment 目录 skill 请使用 <skill>/.omk/samples.json。',
|
|
44
44
|
en: 'Eval samples not found: {path}. Pass --samples, create project-level eval-samples.json, or use <skill>/.omk/samples.json for a single-treatment directory skill.',
|
|
45
45
|
},
|
|
46
|
+
'cli.common.samples_not_found_hint': {
|
|
47
|
+
zh: '下一步:先运行 {command} 生成用例,人工 review 后再重跑 omk eval。',
|
|
48
|
+
en: 'Next: run {command} to generate samples, review them, then re-run omk eval.',
|
|
49
|
+
},
|
|
46
50
|
'cli.update.new_version_available': {
|
|
47
51
|
zh: '\n💡 新版本可用:{old} → {new},运行 npm i -g oh-my-knowledge@latest 升级\n\n',
|
|
48
52
|
en: '\n💡 New version available: {old} → {new}, run npm i -g oh-my-knowledge@latest to upgrade\n\n',
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type GenMessageKey = 'cli.gen.skill_skipped_existing' | 'cli.gen.skill_generating' | 'cli.gen.skill_generating_auto' | 'cli.gen.skill_done' | 'cli.gen.skill_failed' | 'cli.gen.batch_none_needed' | 'cli.gen.batch_summary' | 'cli.gen.specify_skill_path' | 'cli.gen.samples_already_exists' | 'cli.gen.single_generating' | 'cli.gen.single_generating_auto' | 'cli.gen.single_done' | 'cli.gen.append_done' | 'cli.gen.append_single_only' | 'cli.gen.review_hint' | 'cli.gen.failed' | 'cli.gen.focus_applied';
|
|
2
|
+
export type GenMessageKey = 'cli.gen.skill_skipped_existing' | 'cli.gen.skill_generating' | 'cli.gen.skill_generating_auto' | 'cli.gen.skill_done' | 'cli.gen.skill_failed' | 'cli.gen.batch_none_needed' | 'cli.gen.batch_failed_summary' | 'cli.gen.batch_summary' | 'cli.gen.specify_skill_path' | 'cli.gen.samples_already_exists' | 'cli.gen.single_generating' | 'cli.gen.single_generating_auto' | 'cli.gen.single_done' | 'cli.gen.append_done' | 'cli.gen.append_single_only' | 'cli.gen.review_hint' | 'cli.gen.claude_auth_hint' | 'cli.gen.codex_auth_hint' | 'cli.gen.codex_model_hint' | 'cli.gen.openai_api_auth_hint' | 'cli.gen.openai_api_model_hint' | 'cli.gen.anthropic_api_auth_hint' | 'cli.gen.anthropic_api_model_hint' | 'cli.gen.failed' | 'cli.gen.focus_applied';
|
|
3
3
|
export declare const genDict: Record<GenMessageKey, CliMessage>;
|
|
@@ -23,17 +23,21 @@ export const genDict = {
|
|
|
23
23
|
zh: '没有需要生成的 eval-samples (所有 skill 都已有配对文件)',
|
|
24
24
|
en: 'No eval-samples need generating (all skills already have paired files)',
|
|
25
25
|
},
|
|
26
|
+
'cli.gen.batch_failed_summary': {
|
|
27
|
+
zh: '\n生成未完成:已生成 {generated} 份,失败 {failed} 份。请按上面的错误提示修复后重试;全部生成成功后再运行 omk eval --batch --dry-run。',
|
|
28
|
+
en: '\nGeneration incomplete: generated {generated}, failed {failed}. Fix the errors above and retry; once everything succeeds, run omk eval --batch --dry-run.',
|
|
29
|
+
},
|
|
26
30
|
'cli.gen.batch_summary': {
|
|
27
|
-
zh: '\n共生成 {n} 份 eval-samples
|
|
28
|
-
en: '\nGenerated {n} eval-samples files. Review
|
|
31
|
+
zh: '\n共生成 {n} 份 eval-samples。下一步:\n 1. 人工审查生成的评测用例,删掉不可信样本,补边界、反例\n 2. 预览任务:omk eval --batch --dry-run\n 3. 跑评测:omk eval --batch',
|
|
32
|
+
en: '\nGenerated {n} eval-samples files. Next steps:\n 1. Review the generated samples; drop weak cases and add boundary / counterexamples\n 2. Preview the task plan: omk eval --batch --dry-run\n 3. Run the eval: omk eval --batch',
|
|
29
33
|
},
|
|
30
34
|
'cli.gen.specify_skill_path': {
|
|
31
35
|
zh: '请指定 skill 文件路径, 例如: omk sample skills/my-skill.md',
|
|
32
36
|
en: 'Please specify a skill file path, e.g.: omk sample skills/my-skill.md',
|
|
33
37
|
},
|
|
34
38
|
'cli.gen.samples_already_exists': {
|
|
35
|
-
zh: 'eval-samples
|
|
36
|
-
en: 'eval-samples
|
|
39
|
+
zh: 'eval-samples 已存在。要补场景请加 --append(常配 --focus);要继续评测,运行:{command}',
|
|
40
|
+
en: 'eval-samples already exist. To add scenarios, use --append (often with --focus); to continue, run: {command}',
|
|
37
41
|
},
|
|
38
42
|
'cli.gen.single_generating': {
|
|
39
43
|
zh: '🔄 正在生成 {count} 条评测用例...\n',
|
|
@@ -56,8 +60,36 @@ export const genDict = {
|
|
|
56
60
|
en: '--append currently supports single-skill mode only; it cannot be combined with --batch / --from-traces / --fix.\n',
|
|
57
61
|
},
|
|
58
62
|
'cli.gen.review_hint': {
|
|
59
|
-
zh: '\n
|
|
60
|
-
en: '\
|
|
63
|
+
zh: '\n下一步:\n 1. 人工审查生成的评测用例,删掉不可信样本,补边界、反例\n 2. 预览任务:{command} --dry-run\n 3. 跑评测:{command}',
|
|
64
|
+
en: '\nNext steps:\n 1. Review the generated samples; drop weak cases and add boundary / counterexamples\n 2. Preview the task plan: {command} --dry-run\n 3. Run the eval: {command}',
|
|
65
|
+
},
|
|
66
|
+
'cli.gen.claude_auth_hint': {
|
|
67
|
+
zh: '\n提示:当前 sample 生成使用 Claude 系列执行器。先确认 Claude Code 已安装并完成登录;如果你在 Codex 环境里,可以改用:{codexFlags}({codexModelHint});如果要走 OpenAI API,可以改用:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
68
|
+
en: '\nHint: sample generation is using a Claude-based executor. First confirm Claude Code is installed and authenticated; in a Codex environment, switch to: {codexFlags} ({codexModelHint}); to use the OpenAI API path, switch to: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
69
|
+
},
|
|
70
|
+
'cli.gen.codex_auth_hint': {
|
|
71
|
+
zh: '\n提示:当前 sample 生成使用 Codex 系列执行器。先确认 Codex CLI / SDK 已安装并完成登录;如果你有 Claude Code 可用,可以改用:{claudeFlags};如果要走 OpenAI API,可以改用:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
72
|
+
en: '\nHint: sample generation is using a Codex-based executor. First confirm the Codex CLI / SDK is installed and authenticated; if Claude Code is available, switch to: {claudeFlags}; to use the OpenAI API path, switch to: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
73
|
+
},
|
|
74
|
+
'cli.gen.codex_model_hint': {
|
|
75
|
+
zh: '\n提示:当前 sample 生成使用 Codex 系列执行器,但模型名看起来不可用。可以先按本机 Codex 配置重试:{codexFlags}({codexModelHint});也可以先运行 `{codexExec}` 验证模型是否可用。若只是想先跑通,可以改用:{claudeFlags};或走 OpenAI API:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
76
|
+
en: '\nHint: sample generation is using a Codex-based executor, but the model name appears unavailable. Retry with the local Codex config model: {codexFlags} ({codexModelHint}); you can also run `{codexExec}` to verify the model. To just get a first run through, switch to: {claudeFlags}; or use the OpenAI API path: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
77
|
+
},
|
|
78
|
+
'cli.gen.openai_api_auth_hint': {
|
|
79
|
+
zh: '\n提示:当前 sample 生成使用 OpenAI API 执行器。请检查 OPENAI_API_KEY / OPENAI_BASE_URL 是否可用,并确认模型名对当前端点可用;如果只是想先跑通,也可以改用:{claudeFlags},或:{codexFlags}({codexModelHint})。',
|
|
80
|
+
en: '\nHint: sample generation is using the OpenAI API executor. Check OPENAI_API_KEY / OPENAI_BASE_URL and confirm the model is available on that endpoint; to just get a first run through, you can also switch to: {claudeFlags}, or: {codexFlags} ({codexModelHint}).',
|
|
81
|
+
},
|
|
82
|
+
'cli.gen.openai_api_model_hint': {
|
|
83
|
+
zh: '\n提示:当前 sample 生成使用 OpenAI API 执行器,但模型名看起来对当前端点不可用。请检查 --model、OPENAI_BASE_URL 与账号权限是否匹配;如果只是想先跑通,也可以改用:{claudeFlags},或:{codexFlags}({codexModelHint})。',
|
|
84
|
+
en: '\nHint: sample generation is using the OpenAI API executor, but the model name appears unavailable on the current endpoint. Check --model, OPENAI_BASE_URL, and account access; to just get a first run through, you can also switch to: {claudeFlags}, or: {codexFlags} ({codexModelHint}).',
|
|
85
|
+
},
|
|
86
|
+
'cli.gen.anthropic_api_auth_hint': {
|
|
87
|
+
zh: '\n提示:当前 sample 生成使用 Anthropic API 执行器。请检查 ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL 是否可用,并确认模型名对当前端点可用;如果你有 Claude Code 可用,也可以改用:{claudeFlags}。',
|
|
88
|
+
en: '\nHint: sample generation is using the Anthropic API executor. Check ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL and confirm the model is available on that endpoint; if Claude Code is available, you can also switch to: {claudeFlags}.',
|
|
89
|
+
},
|
|
90
|
+
'cli.gen.anthropic_api_model_hint': {
|
|
91
|
+
zh: '\n提示:当前 sample 生成使用 Anthropic API 执行器,但模型名看起来对当前端点不可用。请检查 --model、ANTHROPIC_BASE_URL 与账号权限是否匹配;如果你有 Claude Code 可用,也可以改用:{claudeFlags}。',
|
|
92
|
+
en: '\nHint: sample generation is using the Anthropic API executor, but the model name appears unavailable on the current endpoint. Check --model, ANTHROPIC_BASE_URL, and account access; if Claude Code is available, you can also switch to: {claudeFlags}.',
|
|
61
93
|
},
|
|
62
94
|
'cli.gen.failed': {
|
|
63
95
|
zh: '生成失败: {message}',
|
|
@@ -133,9 +133,9 @@ omk evolve——多轮自动迭代改进 skill
|
|
|
133
133
|
选项:
|
|
134
134
|
--rounds <n> 迭代轮数(默认:5)
|
|
135
135
|
--target <score> 目标分数
|
|
136
|
-
--model <name>
|
|
137
|
-
--improve-model <name> skill
|
|
138
|
-
--judge-models <executor:model>
|
|
136
|
+
--model <name> 任务执行模型,默认跟随 runtime;Codex 读取本机配置
|
|
137
|
+
--improve-model <name> skill 改写模型,默认沿用任务执行模型
|
|
138
|
+
--judge-models <executor:model> 单评委配置(默认跟随执行器;Codex 沿用被测模型)
|
|
139
139
|
|
|
140
140
|
示例:
|
|
141
141
|
omk evolve skills/code-review/SKILL.md
|
|
@@ -151,9 +151,9 @@ Usage:
|
|
|
151
151
|
Options:
|
|
152
152
|
--rounds <n> Iteration rounds (default: 5)
|
|
153
153
|
--target <score> Target score
|
|
154
|
-
--model <name> Task executor model
|
|
155
|
-
--improve-model <name> Skill rewriter model
|
|
156
|
-
--judge-models <executor:model> Single judge config (
|
|
154
|
+
--model <name> Task executor model; follows runtime (Codex reads local config)
|
|
155
|
+
--improve-model <name> Skill rewriter model; defaults to the task model
|
|
156
|
+
--judge-models <executor:model> Single judge config (follows executor; Codex reuses task model)
|
|
157
157
|
|
|
158
158
|
Examples:
|
|
159
159
|
omk evolve skills/code-review/SKILL.md
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_customize' | 'cli.init.
|
|
2
|
+
export type InitMessageKey = 'cli.init.scaffolded' | 'cli.init.next_steps_title' | 'cli.init.next_step_run' | 'cli.init.next_step_executor' | 'cli.init.next_step_report' | 'cli.init.next_step_customize' | 'cli.init.note_skill_injection';
|
|
3
3
|
export declare const initDict: Record<InitMessageKey, CliMessage>;
|