oh-my-knowledge 0.30.0 → 0.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +198 -26
- package/README.zh.md +198 -26
- package/dist/src/authoring/evolver.d.ts +26 -2
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +410 -49
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/authoring/generator.d.ts +5 -2
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +22 -6
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/authoring/sample-fixer.d.ts +10 -4
- package/dist/src/authoring/sample-fixer.d.ts.map +1 -1
- package/dist/src/authoring/sample-fixer.js +153 -126
- package/dist/src/authoring/sample-fixer.js.map +1 -1
- package/dist/src/cli/commands/doctor.d.ts +25 -1
- package/dist/src/cli/commands/doctor.d.ts.map +1 -1
- package/dist/src/cli/commands/doctor.js +233 -124
- package/dist/src/cli/commands/doctor.js.map +1 -1
- package/dist/src/cli/commands/eval/gold/compare.d.ts +17 -0
- package/dist/src/cli/commands/eval/gold/compare.d.ts.map +1 -0
- package/dist/src/cli/commands/eval/gold/compare.js +91 -0
- package/dist/src/cli/commands/eval/gold/compare.js.map +1 -0
- package/dist/src/cli/commands/eval/gold/index.d.ts +9 -0
- package/dist/src/cli/commands/eval/gold/index.d.ts.map +1 -0
- package/dist/src/cli/commands/eval/gold/index.js +34 -0
- package/dist/src/cli/commands/eval/gold/index.js.map +1 -0
- package/dist/src/cli/commands/eval/gold/init.d.ts +11 -0
- package/dist/src/cli/commands/eval/gold/init.d.ts.map +1 -0
- package/dist/src/cli/commands/eval/gold/init.js +51 -0
- package/dist/src/cli/commands/eval/gold/init.js.map +1 -0
- package/dist/src/cli/commands/eval/gold/validate.d.ts +12 -0
- package/dist/src/cli/commands/eval/gold/validate.d.ts.map +1 -0
- package/dist/src/cli/commands/eval/gold/validate.js +46 -0
- package/dist/src/cli/commands/eval/gold/validate.js.map +1 -0
- package/dist/src/cli/commands/eval/index.d.ts +54 -0
- package/dist/src/cli/commands/eval/index.d.ts.map +1 -0
- package/dist/src/cli/commands/{eval-runner.js → eval/index.js} +216 -50
- package/dist/src/cli/commands/eval/index.js.map +1 -0
- package/dist/src/cli/commands/evolve.d.ts +36 -1
- package/dist/src/cli/commands/evolve.d.ts.map +1 -1
- package/dist/src/cli/commands/evolve.js +191 -47
- package/dist/src/cli/commands/evolve.js.map +1 -1
- package/dist/src/cli/commands/init.d.ts +15 -1
- package/dist/src/cli/commands/init.d.ts.map +1 -1
- package/dist/src/cli/commands/init.js +70 -28
- package/dist/src/cli/commands/init.js.map +1 -1
- package/dist/src/cli/commands/observe/inbox.d.ts +23 -0
- package/dist/src/cli/commands/observe/inbox.d.ts.map +1 -0
- package/dist/src/cli/commands/observe/inbox.js +258 -0
- package/dist/src/cli/commands/observe/inbox.js.map +1 -0
- package/dist/src/cli/commands/observe/index.d.ts +22 -0
- package/dist/src/cli/commands/observe/index.d.ts.map +1 -0
- package/dist/src/cli/commands/observe/index.js +118 -0
- package/dist/src/cli/commands/observe/index.js.map +1 -0
- package/dist/src/cli/commands/observe/ingest.d.ts +13 -0
- package/dist/src/cli/commands/observe/ingest.d.ts.map +1 -0
- package/dist/src/cli/commands/observe/ingest.js +69 -0
- package/dist/src/cli/commands/observe/ingest.js.map +1 -0
- package/dist/src/cli/commands/observe/show.d.ts +13 -0
- package/dist/src/cli/commands/observe/show.d.ts.map +1 -0
- package/dist/src/cli/commands/observe/show.js +49 -0
- package/dist/src/cli/commands/observe/show.js.map +1 -0
- package/dist/src/cli/commands/sample.d.ts +28 -5
- package/dist/src/cli/commands/sample.d.ts.map +1 -1
- package/dist/src/cli/commands/sample.js +242 -169
- package/dist/src/cli/commands/sample.js.map +1 -1
- package/dist/src/cli/commands/skill-extract.d.ts +20 -0
- package/dist/src/cli/commands/skill-extract.d.ts.map +1 -0
- package/dist/src/cli/commands/skill-extract.js +84 -0
- package/dist/src/cli/commands/skill-extract.js.map +1 -0
- package/dist/src/cli/commands/studio.d.ts +22 -1
- package/dist/src/cli/commands/studio.d.ts.map +1 -1
- package/dist/src/cli/commands/studio.js +111 -37
- package/dist/src/cli/commands/studio.js.map +1 -1
- package/dist/src/cli/index.js +15 -47
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/lib/cli-exit.d.ts +16 -0
- package/dist/src/cli/lib/cli-exit.d.ts.map +1 -0
- package/dist/src/cli/lib/cli-exit.js +20 -0
- package/dist/src/cli/lib/cli-exit.js.map +1 -0
- package/dist/src/cli/lib/cmd-flags.d.ts +236 -0
- package/dist/src/cli/lib/cmd-flags.d.ts.map +1 -0
- package/dist/src/cli/lib/cmd-flags.js +46 -0
- package/dist/src/cli/lib/cmd-flags.js.map +1 -0
- package/dist/src/cli/{i18n-dict.d.ts → lib/i18n-dict.d.ts} +1 -1
- package/dist/src/cli/lib/i18n-dict.d.ts.map +1 -0
- package/dist/src/cli/{i18n-dict.js → lib/i18n-dict.js} +39 -264
- package/dist/src/cli/lib/i18n-dict.js.map +1 -0
- package/dist/src/cli/{i18n.d.ts → lib/i18n.d.ts} +11 -6
- package/dist/src/cli/lib/i18n.d.ts.map +1 -0
- package/dist/src/cli/{i18n.js → lib/i18n.js} +12 -7
- package/dist/src/cli/lib/i18n.js.map +1 -0
- package/dist/src/cli/{parse-run-config.d.ts → lib/parse-run-config.d.ts} +5 -6
- package/dist/src/cli/lib/parse-run-config.d.ts.map +1 -0
- package/dist/src/cli/{parse-run-config.js → lib/parse-run-config.js} +11 -61
- package/dist/src/cli/lib/parse-run-config.js.map +1 -0
- package/dist/src/cli/lib/progress.d.ts.map +1 -0
- package/dist/src/cli/lib/progress.js.map +1 -0
- package/dist/src/cli/{run-tally.d.ts → lib/run-tally.d.ts} +1 -1
- package/dist/src/cli/lib/run-tally.d.ts.map +1 -0
- package/dist/src/cli/lib/run-tally.js.map +1 -0
- package/dist/src/cli/{commands/_shared.d.ts → lib/shared.d.ts} +2 -2
- package/dist/src/cli/lib/shared.d.ts.map +1 -0
- package/dist/src/cli/{commands/_shared.js → lib/shared.js} +3 -3
- package/dist/src/cli/lib/shared.js.map +1 -0
- package/dist/src/cli/lib/update-check.d.ts.map +1 -0
- package/dist/src/cli/lib/update-check.js.map +1 -0
- package/dist/src/cli/oclif/base-command.d.ts +8 -0
- package/dist/src/cli/oclif/base-command.d.ts.map +1 -0
- package/dist/src/cli/oclif/base-command.js +27 -0
- package/dist/src/cli/oclif/base-command.js.map +1 -0
- package/dist/src/cli/oclif/help.d.ts +16 -0
- package/dist/src/cli/oclif/help.d.ts.map +1 -0
- package/dist/src/cli/oclif/help.js +33 -0
- package/dist/src/cli/oclif/help.js.map +1 -0
- package/dist/src/cli/oclif/i18n.d.ts +16 -0
- package/dist/src/cli/oclif/i18n.d.ts.map +1 -0
- package/dist/src/cli/oclif/i18n.js +62 -0
- package/dist/src/cli/oclif/i18n.js.map +1 -0
- package/dist/src/cli/oclif/parsers.d.ts +10 -0
- package/dist/src/cli/oclif/parsers.d.ts.map +1 -0
- package/dist/src/cli/oclif/parsers.js +102 -0
- package/dist/src/cli/oclif/parsers.js.map +1 -0
- package/dist/src/cli/oclif/projection.d.ts +23 -0
- package/dist/src/cli/oclif/projection.d.ts.map +1 -0
- package/dist/src/cli/oclif/projection.js +52 -0
- package/dist/src/cli/oclif/projection.js.map +1 -0
- package/dist/src/cli/oclif/run.d.ts +2 -0
- package/dist/src/cli/oclif/run.d.ts.map +1 -0
- package/dist/src/cli/oclif/run.js +11 -0
- package/dist/src/cli/oclif/run.js.map +1 -0
- package/dist/src/diagnosis/observe-mapper.d.ts +68 -0
- package/dist/src/diagnosis/observe-mapper.d.ts.map +1 -0
- package/dist/src/diagnosis/observe-mapper.js +276 -0
- package/dist/src/diagnosis/observe-mapper.js.map +1 -0
- package/dist/src/diagnosis/observe-producer.d.ts +9 -0
- package/dist/src/diagnosis/observe-producer.d.ts.map +1 -0
- package/dist/src/diagnosis/observe-producer.js +197 -0
- package/dist/src/diagnosis/observe-producer.js.map +1 -0
- package/dist/src/diagnosis/studio-projection.d.ts +14 -0
- package/dist/src/diagnosis/studio-projection.d.ts.map +1 -0
- package/dist/src/diagnosis/studio-projection.js +84 -0
- package/dist/src/diagnosis/studio-projection.js.map +1 -0
- package/dist/src/diagnosis/types.d.ts +79 -0
- package/dist/src/diagnosis/types.d.ts.map +1 -0
- package/dist/src/diagnosis/types.js +20 -0
- package/dist/src/diagnosis/types.js.map +1 -0
- package/dist/src/doctor/fixer.d.ts +12 -0
- package/dist/src/doctor/fixer.d.ts.map +1 -0
- package/dist/src/doctor/fixer.js +326 -0
- package/dist/src/doctor/fixer.js.map +1 -0
- package/dist/src/doctor/health/composer.js +7 -4
- package/dist/src/doctor/health/composer.js.map +1 -1
- package/dist/src/doctor/health/parser.d.ts +0 -1
- package/dist/src/doctor/health/parser.d.ts.map +1 -1
- package/dist/src/doctor/health/parser.js +37 -0
- package/dist/src/doctor/health/parser.js.map +1 -1
- package/dist/src/doctor/health/prompt-builder.d.ts +2 -21
- package/dist/src/doctor/health/prompt-builder.d.ts.map +1 -1
- package/dist/src/doctor/health/prompt-builder.js +1 -169
- package/dist/src/doctor/health/prompt-builder.js.map +1 -1
- package/dist/src/doctor/html-renderer.d.ts +1 -1
- package/dist/src/doctor/html-renderer.d.ts.map +1 -1
- package/dist/src/doctor/html-renderer.js +2 -2
- package/dist/src/doctor/html-renderer.js.map +1 -1
- package/dist/src/doctor/index.d.ts.map +1 -1
- package/dist/src/doctor/index.js +1 -0
- package/dist/src/doctor/index.js.map +1 -1
- package/dist/src/doctor/renderer.d.ts +1 -1
- package/dist/src/doctor/renderer.d.ts.map +1 -1
- package/dist/src/doctor/renderer.js +2 -2
- package/dist/src/doctor/renderer.js.map +1 -1
- package/dist/src/doctor/rules.js +1 -1
- package/dist/src/doctor/rules.js.map +1 -1
- package/dist/src/eval-core/mock-hook.cjs +13 -1
- package/dist/src/eval-core/mocks-runtime.d.ts +7 -3
- package/dist/src/eval-core/mocks-runtime.d.ts.map +1 -1
- package/dist/src/eval-core/mocks-runtime.js +240 -4
- package/dist/src/eval-core/mocks-runtime.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +1 -1
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +6 -4
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/observability/experience.d.ts +184 -3
- package/dist/src/observability/experience.d.ts.map +1 -1
- package/dist/src/observability/experience.js +1208 -41
- package/dist/src/observability/experience.js.map +1 -1
- package/dist/src/observability/inbox-view-model.d.ts +3 -0
- package/dist/src/observability/inbox-view-model.d.ts.map +1 -1
- package/dist/src/observability/inbox-view-model.js +9 -0
- package/dist/src/observability/inbox-view-model.js.map +1 -1
- package/dist/src/observability/inbox.d.ts +2 -0
- package/dist/src/observability/inbox.d.ts.map +1 -1
- package/dist/src/observability/inbox.js +16 -2
- package/dist/src/observability/inbox.js.map +1 -1
- package/dist/src/observability/problem-patterns.d.ts +1 -0
- package/dist/src/observability/problem-patterns.d.ts.map +1 -1
- package/dist/src/observability/problem-patterns.js +8 -8
- package/dist/src/observability/problem-patterns.js.map +1 -1
- package/dist/src/observability/resolved-review.d.ts +25 -0
- package/dist/src/observability/resolved-review.d.ts.map +1 -0
- package/dist/src/observability/resolved-review.js +231 -0
- package/dist/src/observability/resolved-review.js.map +1 -0
- package/dist/src/observability/review-state.d.ts +9 -2
- package/dist/src/observability/review-state.d.ts.map +1 -1
- package/dist/src/observability/review-state.js +40 -2
- package/dist/src/observability/review-state.js.map +1 -1
- package/dist/src/observability/skill-chain-advisories.js +4 -4
- package/dist/src/observability/skill-chain-advisories.js.map +1 -1
- package/dist/src/observability/skill-chain.d.ts +1 -0
- package/dist/src/observability/skill-chain.d.ts.map +1 -1
- package/dist/src/observability/skill-chain.js +72 -35
- package/dist/src/observability/skill-chain.js.map +1 -1
- package/dist/src/observability/soft-standards.d.ts +130 -0
- package/dist/src/observability/soft-standards.d.ts.map +1 -0
- package/dist/src/observability/soft-standards.js +426 -0
- package/dist/src/observability/soft-standards.js.map +1 -0
- package/dist/src/observability/text-signals.d.ts +7 -0
- package/dist/src/observability/text-signals.d.ts.map +1 -1
- package/dist/src/observability/text-signals.js +88 -4
- package/dist/src/observability/text-signals.js.map +1 -1
- package/dist/src/observability/trace-segmenter.d.ts.map +1 -1
- package/dist/src/observability/trace-segmenter.js +4 -2
- package/dist/src/observability/trace-segmenter.js.map +1 -1
- package/dist/src/observability/trace-source.d.ts.map +1 -1
- package/dist/src/observability/trace-source.js +13 -3
- package/dist/src/observability/trace-source.js.map +1 -1
- package/dist/src/renderer/observation-inbox-renderer.d.ts.map +1 -1
- package/dist/src/renderer/observation-inbox-renderer.js +5914 -924
- package/dist/src/renderer/observation-inbox-renderer.js.map +1 -1
- package/dist/src/renderer/skill-detail-renderer.d.ts.map +1 -1
- package/dist/src/renderer/skill-detail-renderer.js +15 -5
- package/dist/src/renderer/skill-detail-renderer.js.map +1 -1
- package/dist/src/renderer/skill-list-renderer.d.ts.map +1 -1
- package/dist/src/renderer/skill-list-renderer.js +10 -3
- package/dist/src/renderer/skill-list-renderer.js.map +1 -1
- package/dist/src/renderer/test-view.js +2 -2
- package/dist/src/renderer/test-view.js.map +1 -1
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +95 -2
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/skill-index.d.ts +5 -1
- package/dist/src/server/skill-index.d.ts.map +1 -1
- package/dist/src/server/skill-index.js +72 -26
- package/dist/src/server/skill-index.js.map +1 -1
- package/dist/src/server/skill-insights.d.ts +10 -1
- package/dist/src/server/skill-insights.d.ts.map +1 -1
- package/dist/src/server/skill-insights.js +113 -1
- package/dist/src/server/skill-insights.js.map +1 -1
- package/dist/src/shared/hard-rules.d.ts +2 -0
- package/dist/src/shared/hard-rules.d.ts.map +1 -1
- package/dist/src/shared/hard-rules.js +42 -0
- package/dist/src/shared/hard-rules.js.map +1 -1
- package/dist/src/shared/llm-prompts/index.d.ts +14 -0
- package/dist/src/shared/llm-prompts/index.d.ts.map +1 -0
- package/dist/src/shared/llm-prompts/index.js +35 -0
- package/dist/src/shared/llm-prompts/index.js.map +1 -0
- package/dist/src/shared/llm-prompts/skill-health.d.ts +22 -0
- package/dist/src/shared/llm-prompts/skill-health.d.ts.map +1 -0
- package/dist/src/shared/llm-prompts/skill-health.js +170 -0
- package/dist/src/shared/llm-prompts/skill-health.js.map +1 -0
- package/dist/src/types/doctor.d.ts +2 -0
- package/dist/src/types/doctor.d.ts.map +1 -1
- package/dist/src/types/eval.d.ts +6 -1
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +6 -0
- package/dist/src/types/report.d.ts.map +1 -1
- package/docs/prompts/llm-enhanced-review.prompt.md +111 -0
- package/package.json +29 -4
- package/dist/src/cli/cli-exit.d.ts +0 -15
- package/dist/src/cli/cli-exit.d.ts.map +0 -1
- package/dist/src/cli/cli-exit.js +0 -19
- package/dist/src/cli/cli-exit.js.map +0 -1
- package/dist/src/cli/commands/_shared.d.ts.map +0 -1
- package/dist/src/cli/commands/_shared.js.map +0 -1
- package/dist/src/cli/commands/eval-gold.d.ts +0 -2
- package/dist/src/cli/commands/eval-gold.d.ts.map +0 -1
- package/dist/src/cli/commands/eval-gold.js +0 -137
- package/dist/src/cli/commands/eval-gold.js.map +0 -1
- package/dist/src/cli/commands/eval-runner.d.ts +0 -2
- package/dist/src/cli/commands/eval-runner.d.ts.map +0 -1
- package/dist/src/cli/commands/eval-runner.js.map +0 -1
- package/dist/src/cli/commands/eval.d.ts +0 -2
- package/dist/src/cli/commands/eval.d.ts.map +0 -1
- package/dist/src/cli/commands/eval.js +0 -11
- package/dist/src/cli/commands/eval.js.map +0 -1
- package/dist/src/cli/commands/observe.d.ts +0 -2
- package/dist/src/cli/commands/observe.d.ts.map +0 -1
- package/dist/src/cli/commands/observe.js +0 -247
- package/dist/src/cli/commands/observe.js.map +0 -1
- package/dist/src/cli/commands/registry.d.ts +0 -11
- package/dist/src/cli/commands/registry.d.ts.map +0 -1
- package/dist/src/cli/commands/registry.js +0 -32
- package/dist/src/cli/commands/registry.js.map +0 -1
- package/dist/src/cli/i18n-dict.d.ts.map +0 -1
- package/dist/src/cli/i18n-dict.js.map +0 -1
- package/dist/src/cli/i18n.d.ts.map +0 -1
- package/dist/src/cli/i18n.js.map +0 -1
- package/dist/src/cli/parse-run-config.d.ts.map +0 -1
- package/dist/src/cli/parse-run-config.js.map +0 -1
- package/dist/src/cli/parse-strict.d.ts +0 -7
- package/dist/src/cli/parse-strict.d.ts.map +0 -1
- package/dist/src/cli/parse-strict.js +0 -26
- package/dist/src/cli/parse-strict.js.map +0 -1
- package/dist/src/cli/progress.d.ts.map +0 -1
- package/dist/src/cli/progress.js.map +0 -1
- package/dist/src/cli/run-tally.d.ts.map +0 -1
- package/dist/src/cli/run-tally.js.map +0 -1
- package/dist/src/cli/update-check.d.ts.map +0 -1
- package/dist/src/cli/update-check.js.map +0 -1
- /package/dist/src/cli/{progress.d.ts → lib/progress.d.ts} +0 -0
- /package/dist/src/cli/{progress.js → lib/progress.js} +0 -0
- /package/dist/src/cli/{run-tally.js → lib/run-tally.js} +0 -0
- /package/dist/src/cli/{update-check.d.ts → lib/update-check.d.ts} +0 -0
- /package/dist/src/cli/{update-check.js → lib/update-check.js} +0 -0
|
@@ -2,9 +2,11 @@ import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'node:fs';
|
|
|
2
2
|
import { resolve, join, dirname, basename } from 'node:path';
|
|
3
3
|
import { runEvaluation } from '../eval-workflows/run-evaluation.js';
|
|
4
4
|
import { createExecutor, DEFAULT_MODEL, JUDGE_MODEL } from '../executors/index.js';
|
|
5
|
-
import { persistReport, DEFAULT_OUTPUT_DIR, generateRunId } from '../eval-core/evaluation-reporting.js';
|
|
5
|
+
import { persistReport, DEFAULT_OUTPUT_DIR, generateRunId, hashString } from '../eval-core/evaluation-reporting.js';
|
|
6
|
+
import { createFileStore } from '../server/report-store.js';
|
|
6
7
|
import { analyzeResults } from '../analysis/report-diagnostics.js';
|
|
7
8
|
import { loadSamples } from '../inputs/load-samples.js';
|
|
9
|
+
import { fixSamples } from './sample-fixer.js';
|
|
8
10
|
const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
|
|
9
11
|
|
|
10
12
|
改进原则:
|
|
@@ -12,13 +14,100 @@ const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任
|
|
|
12
14
|
2. 保留当前版本中已经表现良好的部分
|
|
13
15
|
3. 保持 skill 的整体结构和格式
|
|
14
16
|
4. 改进应该具体、可执行,不要空泛的描述
|
|
17
|
+
5. 改动必须最小化:只修改或新增与低分用例直接相关的段落,不要重写未出问题的部分
|
|
18
|
+
6. 禁止重新组织整篇结构、格式或措辞;保留原文中未出问题的段落原封不动
|
|
19
|
+
7. 每轮改动行数不超过原文的 20%
|
|
15
20
|
|
|
16
21
|
直接输出改进后的 skill 内容,不要包含 markdown 代码块标记或任何解释说明。`;
|
|
22
|
+
const IMPROVE_AGENT_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,使用 Edit 工具针对性地修改 skill 文件。
|
|
23
|
+
|
|
24
|
+
改进原则:
|
|
25
|
+
1. 只修改与低分用例直接相关的段落,不要动已经表现良好的部分
|
|
26
|
+
2. 使用 Edit 工具做最小化修改,不要重写整篇文件
|
|
27
|
+
3. 每次 Edit 只改一处具体问题,可以多次 Edit
|
|
28
|
+
4. 不要重新组织文档结构、不要调整格式、不要修改措辞风格
|
|
29
|
+
5. 新增内容应紧跟在相关现有段落之后,不要大段插入
|
|
30
|
+
6. 改进应该具体、可执行,不要空泛的描述
|
|
31
|
+
7. 修改完成后不要输出文件全文,只说明改了什么`;
|
|
17
32
|
/**
|
|
18
33
|
* Read the `name` field from a SKILL.md frontmatter.
|
|
19
34
|
* Returns null if the file has no frontmatter or no name field.
|
|
20
35
|
* Used to give evolve reports semantic filenames instead of "evolve-SKILL-xxx".
|
|
21
36
|
*/
|
|
37
|
+
function canonicalStringify(value) {
|
|
38
|
+
if (value === null || typeof value !== 'object')
|
|
39
|
+
return JSON.stringify(value);
|
|
40
|
+
if (Array.isArray(value))
|
|
41
|
+
return '[' + value.map(canonicalStringify).join(',') + ']';
|
|
42
|
+
const entries = Object.keys(value).sort();
|
|
43
|
+
return '{' + entries.map((k) => JSON.stringify(k) + ':' + canonicalStringify(value[k])).join(',') + '}';
|
|
44
|
+
}
|
|
45
|
+
function hashSampleForReuse(sample) {
|
|
46
|
+
return hashString(canonicalStringify({
|
|
47
|
+
prompt: sample.prompt,
|
|
48
|
+
rubric: sample.rubric ?? null,
|
|
49
|
+
dimensions: sample.dimensions ?? null,
|
|
50
|
+
assertions: sample.assertions ?? null,
|
|
51
|
+
schema: sample.schema ?? null,
|
|
52
|
+
}));
|
|
53
|
+
}
|
|
54
|
+
function sameJudgeModels(report, judges) {
|
|
55
|
+
const metaJudges = report.meta.judgeModels ?? [];
|
|
56
|
+
if (metaJudges.length !== judges.length)
|
|
57
|
+
return false;
|
|
58
|
+
return metaJudges.every((j, i) => j.executor === judges[i].executor && j.model === judges[i].model);
|
|
59
|
+
}
|
|
60
|
+
function sampleHashesMatch(report, samples) {
|
|
61
|
+
const hashes = report.meta.sampleHashes;
|
|
62
|
+
if (!hashes)
|
|
63
|
+
return false;
|
|
64
|
+
return samples.every((s) => hashes[s.sample_id] === hashSampleForReuse(s));
|
|
65
|
+
}
|
|
66
|
+
function singleVariantReport(report, variantKey) {
|
|
67
|
+
const summary = report.summary[variantKey];
|
|
68
|
+
return {
|
|
69
|
+
...report,
|
|
70
|
+
meta: {
|
|
71
|
+
...report.meta,
|
|
72
|
+
variants: [variantKey],
|
|
73
|
+
artifactHashes: report.meta.artifactHashes?.[variantKey]
|
|
74
|
+
? { [variantKey]: report.meta.artifactHashes[variantKey] }
|
|
75
|
+
: {},
|
|
76
|
+
...(report.meta.variantConfigs ? { variantConfigs: report.meta.variantConfigs.filter((cfg) => cfg.variant === variantKey) } : {}),
|
|
77
|
+
...(report.meta.skillIsolation ? { skillIsolation: { [variantKey]: report.meta.skillIsolation[variantKey] ?? null } } : {}),
|
|
78
|
+
},
|
|
79
|
+
summary: { [variantKey]: summary },
|
|
80
|
+
results: report.results.map((entry) => ({
|
|
81
|
+
sample_id: entry.sample_id,
|
|
82
|
+
variants: entry.variants[variantKey] ? { [variantKey]: entry.variants[variantKey] } : {},
|
|
83
|
+
})),
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
async function findReusableBaselineReport(opts) {
|
|
87
|
+
const store = createFileStore(DEFAULT_OUTPUT_DIR);
|
|
88
|
+
const { samples } = loadSamples(opts.samplesPath);
|
|
89
|
+
const artifactHash = hashString(opts.skillContent);
|
|
90
|
+
const reports = await store.findByArtifactHash(artifactHash);
|
|
91
|
+
for (const report of reports) {
|
|
92
|
+
if (report.meta.model !== opts.model || report.meta.executor !== opts.executorName)
|
|
93
|
+
continue;
|
|
94
|
+
if ((report.meta.effort ?? undefined) !== (opts.effort ?? undefined))
|
|
95
|
+
continue;
|
|
96
|
+
if (report.meta.noJudge)
|
|
97
|
+
continue;
|
|
98
|
+
if (report.meta.budgetExhausted)
|
|
99
|
+
continue;
|
|
100
|
+
if (!sameJudgeModels(report, opts.judgeModels))
|
|
101
|
+
continue;
|
|
102
|
+
if (!sampleHashesMatch(report, samples))
|
|
103
|
+
continue;
|
|
104
|
+
const variantKey = report.meta.variants.find((name) => report.meta.artifactHashes?.[name] === artifactHash);
|
|
105
|
+
if (!variantKey || !report.summary[variantKey])
|
|
106
|
+
continue;
|
|
107
|
+
return singleVariantReport(report, variantKey);
|
|
108
|
+
}
|
|
109
|
+
return null;
|
|
110
|
+
}
|
|
22
111
|
function readSkillName(skillPath) {
|
|
23
112
|
try {
|
|
24
113
|
const content = readFileSync(skillPath, 'utf-8');
|
|
@@ -33,25 +122,163 @@ function readSkillName(skillPath) {
|
|
|
33
122
|
}
|
|
34
123
|
}
|
|
35
124
|
export function extractWeakSamples(report, variantKey, count = 5) {
|
|
36
|
-
|
|
37
|
-
|
|
125
|
+
const weakSamples = [];
|
|
126
|
+
for (const r of report.results) {
|
|
38
127
|
const v = r.variants[variantKey];
|
|
39
128
|
if (!v || typeof v.compositeScore !== 'number')
|
|
40
|
-
|
|
41
|
-
|
|
129
|
+
continue;
|
|
130
|
+
const suggestion = v.diagnostic?.suggestion;
|
|
131
|
+
weakSamples.push({
|
|
42
132
|
sample_id: r.sample_id,
|
|
43
133
|
compositeScore: v.compositeScore,
|
|
44
134
|
llmReason: v.llmReason || null,
|
|
45
|
-
failedAssertions: v.assertions?.details?.filter((a) => !a.passed).map((a) => `${a.type}: ${a.value}`) || [],
|
|
135
|
+
failedAssertions: v.assertions?.details?.filter((a) => !a.passed).map((a) => `${a.type}: ${a.value ?? ''}`) || [],
|
|
46
136
|
dimensions: v.dimensions
|
|
47
137
|
? Object.fromEntries(Object.entries(v.dimensions).map(([k, info]) => [k, typeof info === 'object' ? info.score : info]))
|
|
48
138
|
: null,
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
139
|
+
...(v.diagnostic ? {
|
|
140
|
+
diagnostic: {
|
|
141
|
+
summary: v.diagnostic.summary,
|
|
142
|
+
expected: v.diagnostic.expected,
|
|
143
|
+
actual: v.diagnostic.actual,
|
|
144
|
+
rootCause: v.diagnostic.rootCause,
|
|
145
|
+
skillSuggestion: suggestion?.skill,
|
|
146
|
+
sampleSuggestion: suggestion?.sample,
|
|
147
|
+
none: suggestion?.none,
|
|
148
|
+
},
|
|
149
|
+
} : {}),
|
|
150
|
+
});
|
|
151
|
+
}
|
|
152
|
+
return weakSamples
|
|
52
153
|
.sort((a, b) => a.compositeScore - b.compositeScore)
|
|
53
154
|
.slice(0, count);
|
|
54
155
|
}
|
|
156
|
+
export function allNonTripwireAssertionsPass(report, variantKey) {
|
|
157
|
+
for (const entry of report.results) {
|
|
158
|
+
const variant = entry.variants[variantKey];
|
|
159
|
+
if (!variant || variant.ok === false)
|
|
160
|
+
return false;
|
|
161
|
+
const snapshotTripwire = report.sampleSnapshots?.[entry.sample_id]?.tripwire === true;
|
|
162
|
+
const diagnosticTripwire = (variant.diagnostic?.rootCause ?? []).includes('tripwire_intentional');
|
|
163
|
+
if (snapshotTripwire || diagnosticTripwire)
|
|
164
|
+
continue;
|
|
165
|
+
const details = variant.assertions?.details ?? [];
|
|
166
|
+
if (!details.every((d) => d.passed))
|
|
167
|
+
return false;
|
|
168
|
+
}
|
|
169
|
+
return true;
|
|
170
|
+
}
|
|
171
|
+
const FIX_AGENT_SYSTEM_PROMPT = `你是一个评测用例修复专家。你的任务是使用 Edit 工具直接修改 samples JSON 文件中有问题的用例。
|
|
172
|
+
|
|
173
|
+
修复原则:
|
|
174
|
+
1. 先分析失败原因——是 sample 设计问题(mock 缺失 / 断言写错 / 工具名不匹配)还是 LLM 行为问题(LLM 真的做错了,低分合理)
|
|
175
|
+
2. 如果是 LLM 行为问题(低分合理),不要改 sample
|
|
176
|
+
3. 如果是 sample 设计问题,用 Edit 工具修复对应 sample 的 mocks / assertions / environment
|
|
177
|
+
4. 保持 sample_id 不变,保持测试意图不变
|
|
178
|
+
5. 每次 Edit 只改一个 sample 的一处问题,可以多次 Edit
|
|
179
|
+
6. 不要重写整个 JSON 文件,只改有问题的字段
|
|
180
|
+
7. 断言修复只保留核心行为节点,不要过度绑定实现细节
|
|
181
|
+
8. 不要删除整条用例
|
|
182
|
+
9. 只能修改 sample,绝对不能建议修改 SKILL.md`;
|
|
183
|
+
function createSampleFixExecutor(executorName) {
|
|
184
|
+
const executor = createExecutor(executorName);
|
|
185
|
+
return async (opts) => {
|
|
186
|
+
const result = await executor({
|
|
187
|
+
model: opts.model,
|
|
188
|
+
system: opts.system,
|
|
189
|
+
prompt: opts.prompt,
|
|
190
|
+
timeoutMs: opts.timeoutMs,
|
|
191
|
+
lean: opts.lean,
|
|
192
|
+
...(opts.cwd && { cwd: opts.cwd }),
|
|
193
|
+
});
|
|
194
|
+
return { ok: result.ok, text: result.output ?? '', costUSD: result.costUSD };
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
async function autoFixSamplesAgent(opts) {
|
|
198
|
+
const { samples } = loadSamples(opts.samplesPath);
|
|
199
|
+
const sampleMap = new Map(samples.map((s) => [s.sample_id, s]));
|
|
200
|
+
const fixContexts = [];
|
|
201
|
+
for (const entry of opts.report.results) {
|
|
202
|
+
const variant = entry.variants?.[opts.treatmentKey];
|
|
203
|
+
if (!variant)
|
|
204
|
+
continue;
|
|
205
|
+
const variantObj = variant;
|
|
206
|
+
if (variantObj.ok === false)
|
|
207
|
+
continue;
|
|
208
|
+
const assertionDetails = variantObj.assertions?.details ?? [];
|
|
209
|
+
if (assertionDetails.length === 0 || assertionDetails.every((d) => d.passed))
|
|
210
|
+
continue;
|
|
211
|
+
const sid = entry.sample_id;
|
|
212
|
+
const original = sampleMap.get(sid);
|
|
213
|
+
if (!original)
|
|
214
|
+
continue;
|
|
215
|
+
if (typeof original.omkFix === 'object') {
|
|
216
|
+
const attempts = original.omkFix?.attempts;
|
|
217
|
+
if (typeof attempts === 'number' && attempts >= opts.maxAttemptsPerSample)
|
|
218
|
+
continue;
|
|
219
|
+
}
|
|
220
|
+
const diag = variantObj.diagnostic;
|
|
221
|
+
const toolCalls = (variantObj.toolCalls ?? []);
|
|
222
|
+
const failedList = assertionDetails.filter((a) => !a.passed).map((a) => `${a.type}: ${a.value}`).join('\n');
|
|
223
|
+
const toolSummary = toolCalls.length > 0
|
|
224
|
+
? toolCalls.map((tc, i) => `[${i}] ${tc.tool} success=${tc.success}`).join('\n')
|
|
225
|
+
: '(无工具调用)';
|
|
226
|
+
fixContexts.push({
|
|
227
|
+
sampleId: sid,
|
|
228
|
+
diag: `归因: ${(diag?.rootCause ?? []).join(', ') || '(无)'}\n摘要: ${diag?.summary ?? ''}\n期望: ${diag?.expected ?? ''}\n实际: ${diag?.actual ?? ''}\n建议: ${(diag?.suggestion?.sample) ?? '(无)'}`,
|
|
229
|
+
failedAssertions: `失败断言:\n${failedList}\n\n实际工具调用:\n${toolSummary}`,
|
|
230
|
+
});
|
|
231
|
+
}
|
|
232
|
+
if (fixContexts.length === 0)
|
|
233
|
+
return { fixedCount: 0, costUSD: 0 };
|
|
234
|
+
const skillPreview = opts.skillContent.length > 4000
|
|
235
|
+
? opts.skillContent.slice(0, 4000) + '\n\n... (truncated)'
|
|
236
|
+
: opts.skillContent;
|
|
237
|
+
const sampleSections = fixContexts.map((ctx) => `### ${ctx.sampleId}\n\n${ctx.diag}\n\n${ctx.failedAssertions}`).join('\n\n---\n\n');
|
|
238
|
+
const prompt = `以下有 ${fixContexts.length} 条失败的评测用例需要分析修复。
|
|
239
|
+
|
|
240
|
+
## Skill 原文(参考,不可修改)
|
|
241
|
+
|
|
242
|
+
${skillPreview}
|
|
243
|
+
|
|
244
|
+
## 待修复的用例
|
|
245
|
+
|
|
246
|
+
${sampleSections}
|
|
247
|
+
|
|
248
|
+
请使用 Edit 工具修改文件 ${opts.samplesPath},只改有问题的 sample 的 assertions / mocks / environment 字段。如果判断是 LLM 行为问题(低分合理),不要改该 sample。`;
|
|
249
|
+
const executor = createExecutor(opts.executorName);
|
|
250
|
+
const beforeContent = readFileSync(opts.samplesPath, 'utf-8');
|
|
251
|
+
const result = await executor({
|
|
252
|
+
model: opts.model,
|
|
253
|
+
system: FIX_AGENT_SYSTEM_PROMPT,
|
|
254
|
+
prompt,
|
|
255
|
+
cwd: dirname(opts.samplesPath),
|
|
256
|
+
timeoutMs: 300_000,
|
|
257
|
+
});
|
|
258
|
+
const afterContent = readFileSync(opts.samplesPath, 'utf-8');
|
|
259
|
+
const changed = afterContent !== beforeContent;
|
|
260
|
+
const fixedCount = changed ? fixContexts.length : 0;
|
|
261
|
+
return { fixedCount, costUSD: result.costUSD };
|
|
262
|
+
}
|
|
263
|
+
async function autoFixSamplesAfterSkillRound(opts) {
|
|
264
|
+
const { samples } = loadSamples(opts.samplesPath);
|
|
265
|
+
if (opts.improveMode === 'agent') {
|
|
266
|
+
return autoFixSamplesAgent(opts);
|
|
267
|
+
}
|
|
268
|
+
const result = await fixSamples({
|
|
269
|
+
skillContent: opts.skillContent,
|
|
270
|
+
samples: samples,
|
|
271
|
+
report: opts.report,
|
|
272
|
+
treatmentKey: opts.treatmentKey,
|
|
273
|
+
executor: createSampleFixExecutor(opts.executorName),
|
|
274
|
+
model: opts.model,
|
|
275
|
+
maxAttemptsPerSample: opts.maxAttemptsPerSample,
|
|
276
|
+
});
|
|
277
|
+
if (result.fixedCount > 0) {
|
|
278
|
+
writeFileSync(opts.samplesPath, JSON.stringify(result.samples, null, 2));
|
|
279
|
+
}
|
|
280
|
+
return { fixedCount: result.fixedCount, costUSD: result.costUSD };
|
|
281
|
+
}
|
|
55
282
|
export function buildImprovementPrompt(skillContent, score, weakSamples) {
|
|
56
283
|
const weakDetails = weakSamples.map((s) => {
|
|
57
284
|
const parts = [`### ${s.sample_id}(${s.compositeScore}/5.0)`];
|
|
@@ -59,6 +286,22 @@ export function buildImprovementPrompt(skillContent, score, weakSamples) {
|
|
|
59
286
|
parts.push(`评委反馈: ${s.llmReason}`);
|
|
60
287
|
if (s.failedAssertions.length > 0)
|
|
61
288
|
parts.push(`失败断言: ${s.failedAssertions.join(', ')}`);
|
|
289
|
+
if (s.diagnostic) {
|
|
290
|
+
if (s.diagnostic.rootCause?.length)
|
|
291
|
+
parts.push(`诊断归因: ${s.diagnostic.rootCause.join(', ')}`);
|
|
292
|
+
if (s.diagnostic.summary)
|
|
293
|
+
parts.push(`诊断摘要: ${s.diagnostic.summary}`);
|
|
294
|
+
if (s.diagnostic.expected)
|
|
295
|
+
parts.push(`期望行为: ${s.diagnostic.expected}`);
|
|
296
|
+
if (s.diagnostic.actual)
|
|
297
|
+
parts.push(`实际行为: ${s.diagnostic.actual}`);
|
|
298
|
+
if (s.diagnostic.skillSuggestion)
|
|
299
|
+
parts.push(`建议修改 skill: ${s.diagnostic.skillSuggestion}`);
|
|
300
|
+
if (s.diagnostic.sampleSuggestion)
|
|
301
|
+
parts.push(`建议修改 sample: ${s.diagnostic.sampleSuggestion}`);
|
|
302
|
+
if (s.diagnostic.none)
|
|
303
|
+
parts.push(`无需改动说明: ${s.diagnostic.none}`);
|
|
304
|
+
}
|
|
62
305
|
if (s.dimensions) {
|
|
63
306
|
const dimStr = Object.entries(s.dimensions).map(([k, v]) => `${k}: ${v}`).join(', ');
|
|
64
307
|
parts.push(`维度分数: ${dimStr}`);
|
|
@@ -71,9 +314,13 @@ ${skillContent}
|
|
|
71
314
|
|
|
72
315
|
## 低分用例分析
|
|
73
316
|
|
|
74
|
-
${weakDetails || '(无低分用例)'}
|
|
75
|
-
|
|
76
|
-
|
|
317
|
+
${weakDetails || '(无低分用例)'}`;
|
|
318
|
+
}
|
|
319
|
+
function buildImprovementSuffix(mode, candidatePath) {
|
|
320
|
+
if (mode === 'agent') {
|
|
321
|
+
return `\n\n请使用 Edit 工具修改文件 ${candidatePath},只改与上述问题直接相关的部分。不要重写整篇文件。`;
|
|
322
|
+
}
|
|
323
|
+
return '\n\n请基于以上分析改进 Skill,使其在这些场景中表现更好。直接输出改进后的完整 Skill 内容。';
|
|
77
324
|
}
|
|
78
325
|
function parseImprovedSkill(output) {
|
|
79
326
|
let content = output.trim();
|
|
@@ -90,7 +337,7 @@ function parseImprovedSkill(output) {
|
|
|
90
337
|
}
|
|
91
338
|
return content;
|
|
92
339
|
}
|
|
93
|
-
export function mergeEvolveReports(roundReports, skillName, totalCostUSD, samples) {
|
|
340
|
+
export function mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, skillPath) {
|
|
94
341
|
const firstReport = roundReports[0].report;
|
|
95
342
|
// Build variant labels: "round-0", "round-1", "round-2", ...
|
|
96
343
|
const variantLabels = roundReports.map(({ round }) => `round-${round}`);
|
|
@@ -139,9 +386,14 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
139
386
|
artifactHashes,
|
|
140
387
|
totalCostUSD: Number(totalCostUSD.toFixed(6)),
|
|
141
388
|
timestamp: new Date().toISOString(),
|
|
389
|
+
evolve: {
|
|
390
|
+
skillName,
|
|
391
|
+
...(skillPath ? { skillPath } : {}),
|
|
392
|
+
},
|
|
142
393
|
},
|
|
143
394
|
summary,
|
|
144
395
|
results,
|
|
396
|
+
...(firstReport.sampleSnapshots ? { sampleSnapshots: firstReport.sampleSnapshots } : {}),
|
|
145
397
|
};
|
|
146
398
|
// pass samples to populate analysis.sampleQuality (capability/difficulty/
|
|
147
399
|
// construct/provenance coverage). Without samples, sampleQuality is omitted but
|
|
@@ -149,7 +401,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
149
401
|
report.analysis = analyzeResults(report, { samples });
|
|
150
402
|
return report;
|
|
151
403
|
}
|
|
152
|
-
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, onProgress = null, onRoundProgress = null, }) {
|
|
404
|
+
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, onProgress = null, onRoundProgress = null, }) {
|
|
153
405
|
if (judgeModels && judgeModels.length > 1) {
|
|
154
406
|
throw new Error('evolveSkill does not support multi-judge ensemble (received '
|
|
155
407
|
+ `${judgeModels.length} judges). Pass a single-judge array, e.g. `
|
|
@@ -181,23 +433,66 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
181
433
|
let consecutiveRejects = 0;
|
|
182
434
|
const trajectory = [];
|
|
183
435
|
const roundReports = [];
|
|
436
|
+
const sampleFixes = [];
|
|
437
|
+
let reusedBaselineReportId;
|
|
438
|
+
let baselineReused = false;
|
|
439
|
+
let stopReason = 'rounds';
|
|
184
440
|
// 给定一个 report 看任一 variant 的 exec/judge cost 是否未报告
|
|
185
441
|
const reportHasUnreportedCost = (rep) => Object.values(rep.summary).some((v) => v.execCostReported === false || v.judgeCostReported === false);
|
|
186
442
|
// Round 0: baseline evaluation
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
443
|
+
let baselineReport = reuseLatestEval
|
|
444
|
+
? await findReusableBaselineReport({
|
|
445
|
+
skillContent: currentBest,
|
|
446
|
+
samplesPath: absSamplesPath,
|
|
447
|
+
model,
|
|
448
|
+
executorName,
|
|
449
|
+
judgeModels: effectiveJudgeModels,
|
|
450
|
+
effort,
|
|
451
|
+
noDiagnostic,
|
|
452
|
+
})
|
|
453
|
+
: null;
|
|
454
|
+
if (baselineReport) {
|
|
455
|
+
reusedBaselineReportId = baselineReport.id;
|
|
456
|
+
baselineReused = true;
|
|
457
|
+
}
|
|
458
|
+
else {
|
|
459
|
+
baselineReport = await evaluate(r0Path, {
|
|
460
|
+
samplesPath: absSamplesPath, skillDir, model, judgeModels: effectiveJudgeModels, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress,
|
|
461
|
+
});
|
|
462
|
+
}
|
|
190
463
|
const baselineVariantKey = Object.keys(baselineReport.summary)[0];
|
|
191
464
|
bestScore = baselineReport.summary[baselineVariantKey]?.avgCompositeScore ?? 0;
|
|
192
|
-
const baselineCost = baselineReport.meta.totalCostUSD;
|
|
465
|
+
const baselineCost = baselineReused ? 0 : baselineReport.meta.totalCostUSD;
|
|
193
466
|
totalCostUSD += baselineCost;
|
|
194
|
-
const baselineCostReported = !reportHasUnreportedCost(baselineReport);
|
|
467
|
+
const baselineCostReported = baselineReused || !reportHasUnreportedCost(baselineReport);
|
|
195
468
|
if (!baselineCostReported)
|
|
196
469
|
totalCostReported = false;
|
|
197
470
|
trajectory.push({ round: 0, score: bestScore, delta: 0, accepted: true, costUSD: baselineCost });
|
|
198
471
|
roundReports.push({ round: 0, accepted: true, report: baselineReport });
|
|
199
472
|
if (onRoundProgress)
|
|
200
|
-
onRoundProgress({ round: 0, totalRounds: rounds, phase: 'baseline', score: bestScore, costUSD: baselineCost, costReported: baselineCostReported });
|
|
473
|
+
onRoundProgress({ round: 0, totalRounds: rounds, phase: 'baseline', score: bestScore, costUSD: baselineCost, costReported: baselineCostReported, reused: baselineReused });
|
|
474
|
+
if (stopOnAssertionsPass && allNonTripwireAssertionsPass(baselineReport, baselineVariantKey)) {
|
|
475
|
+
stopReason = 'assertions-pass';
|
|
476
|
+
writeFileSync(absSkillPath, currentBest);
|
|
477
|
+
const { samples } = loadSamples(absSamplesPath);
|
|
478
|
+
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
|
|
479
|
+
persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
|
|
480
|
+
return {
|
|
481
|
+
startScore: bestScore,
|
|
482
|
+
finalScore: bestScore,
|
|
483
|
+
bestRound,
|
|
484
|
+
totalRounds: 0,
|
|
485
|
+
totalCostUSD: Number(totalCostUSD.toFixed(6)),
|
|
486
|
+
stopReason,
|
|
487
|
+
...(sampleFixes.length > 0 && { sampleFixes }),
|
|
488
|
+
...(reusedBaselineReportId && { reusedBaselineReportId }),
|
|
489
|
+
...(totalCostReported ? {} : { costReported: false }),
|
|
490
|
+
trajectory,
|
|
491
|
+
bestSkillPath: allVersions[bestRound],
|
|
492
|
+
allVersions,
|
|
493
|
+
reportId: mergedReport.id,
|
|
494
|
+
};
|
|
495
|
+
}
|
|
201
496
|
// Evolution loop
|
|
202
497
|
for (let round = 1; round <= rounds; round++) {
|
|
203
498
|
// Extract weak samples from last accepted evaluation. round=1 reuses baselineReport
|
|
@@ -220,41 +515,91 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
220
515
|
const lastVariantKey = Object.keys(lastReport.summary)[0];
|
|
221
516
|
const weakSamples = extractWeakSamples(lastReport, lastVariantKey);
|
|
222
517
|
// Generate improvement
|
|
223
|
-
const
|
|
518
|
+
const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
|
|
519
|
+
const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples);
|
|
224
520
|
const executor = createExecutor(executorName);
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
521
|
+
let candidateContent;
|
|
522
|
+
let improveCostUSD;
|
|
523
|
+
let improveCostReported = true;
|
|
524
|
+
if (improveMode === 'agent') {
|
|
525
|
+
writeFileSync(candidatePath, currentBest);
|
|
526
|
+
const improvePrompt = basePrompt + buildImprovementSuffix('agent', candidatePath);
|
|
527
|
+
const improveResult = await executor({ model: improveModel, system: IMPROVE_AGENT_SYSTEM_PROMPT, prompt: improvePrompt, cwd: skillDir, timeoutMs });
|
|
528
|
+
if (!improveResult.ok) {
|
|
529
|
+
if (onRoundProgress)
|
|
530
|
+
onRoundProgress({ round, totalRounds: rounds, phase: 'error', error: improveResult.error });
|
|
531
|
+
consecutiveRejects++;
|
|
532
|
+
trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveResult.costUSD });
|
|
533
|
+
totalCostUSD += improveResult.costUSD;
|
|
534
|
+
if (improveResult.costReportedByExecutor === false)
|
|
535
|
+
totalCostReported = false;
|
|
536
|
+
if (consecutiveRejects >= 2) {
|
|
537
|
+
stopReason = 'consecutive-rejects';
|
|
538
|
+
break;
|
|
539
|
+
}
|
|
540
|
+
continue;
|
|
541
|
+
}
|
|
542
|
+
improveCostUSD = improveResult.costUSD;
|
|
232
543
|
if (improveResult.costReportedByExecutor === false)
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
break;
|
|
236
|
-
continue;
|
|
544
|
+
improveCostReported = false;
|
|
545
|
+
candidateContent = readFileSync(candidatePath, 'utf-8');
|
|
237
546
|
}
|
|
238
|
-
|
|
547
|
+
else {
|
|
548
|
+
const improvePrompt = basePrompt + buildImprovementSuffix('rewrite');
|
|
549
|
+
const improveResult = await executor({ model: improveModel, system: IMPROVE_SYSTEM_PROMPT, prompt: improvePrompt, timeoutMs, lean: true });
|
|
550
|
+
if (!improveResult.ok) {
|
|
551
|
+
if (onRoundProgress)
|
|
552
|
+
onRoundProgress({ round, totalRounds: rounds, phase: 'error', error: improveResult.error });
|
|
553
|
+
consecutiveRejects++;
|
|
554
|
+
trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveResult.costUSD });
|
|
555
|
+
totalCostUSD += improveResult.costUSD;
|
|
556
|
+
if (improveResult.costReportedByExecutor === false)
|
|
557
|
+
totalCostReported = false;
|
|
558
|
+
if (consecutiveRejects >= 2) {
|
|
559
|
+
stopReason = 'consecutive-rejects';
|
|
560
|
+
break;
|
|
561
|
+
}
|
|
562
|
+
continue;
|
|
563
|
+
}
|
|
564
|
+
improveCostUSD = improveResult.costUSD;
|
|
565
|
+
if (improveResult.costReportedByExecutor === false)
|
|
566
|
+
improveCostReported = false;
|
|
567
|
+
candidateContent = parseImprovedSkill(improveResult.output);
|
|
568
|
+
writeFileSync(candidatePath, candidateContent);
|
|
569
|
+
}
|
|
570
|
+
if (!improveCostReported)
|
|
239
571
|
totalCostReported = false;
|
|
240
|
-
const candidateContent = parseImprovedSkill(improveResult.output);
|
|
241
|
-
const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
|
|
242
|
-
writeFileSync(candidatePath, candidateContent);
|
|
243
572
|
allVersions.push(candidatePath);
|
|
244
|
-
|
|
573
|
+
let preEvalSampleFixCost = 0;
|
|
574
|
+
if (autoFixSamples) {
|
|
575
|
+
const sampleFix = await autoFixSamplesAfterSkillRound({
|
|
576
|
+
samplesPath: absSamplesPath,
|
|
577
|
+
skillContent: candidateContent,
|
|
578
|
+
report: lastReport,
|
|
579
|
+
treatmentKey: lastVariantKey,
|
|
580
|
+
executorName,
|
|
581
|
+
model: improveModel,
|
|
582
|
+
maxAttemptsPerSample: sampleFixMaxAttempts,
|
|
583
|
+
improveMode,
|
|
584
|
+
});
|
|
585
|
+
if (sampleFix.fixedCount > 0) {
|
|
586
|
+
sampleFixes.push({ round, fixedCount: sampleFix.fixedCount, costUSD: sampleFix.costUSD });
|
|
587
|
+
}
|
|
588
|
+
preEvalSampleFixCost = sampleFix.costUSD;
|
|
589
|
+
totalCostUSD += sampleFix.costUSD;
|
|
590
|
+
}
|
|
591
|
+
// Evaluate candidate with any sample fixes already applied.
|
|
245
592
|
const candidateReport = await evaluate(candidatePath, {
|
|
246
593
|
samplesPath: absSamplesPath, skillDir, model, judgeModels: effectiveJudgeModels, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress,
|
|
247
594
|
});
|
|
248
595
|
const candidateVariantKey = Object.keys(candidateReport.summary)[0];
|
|
249
596
|
const candidateScore = candidateReport.summary[candidateVariantKey]?.avgCompositeScore ?? 0;
|
|
250
|
-
const roundCost =
|
|
251
|
-
const roundCostReported =
|
|
597
|
+
const roundCost = improveCostUSD + preEvalSampleFixCost + candidateReport.meta.totalCostUSD;
|
|
598
|
+
const roundCostReported = improveCostReported && !reportHasUnreportedCost(candidateReport);
|
|
252
599
|
if (!roundCostReported)
|
|
253
600
|
totalCostReported = false;
|
|
254
|
-
totalCostUSD +=
|
|
255
|
-
const delta = candidateScore - bestScore;
|
|
601
|
+
totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
|
|
256
602
|
const accepted = candidateScore > bestScore;
|
|
257
|
-
roundReports.push({ round, accepted, report: candidateReport });
|
|
258
603
|
if (accepted) {
|
|
259
604
|
currentBest = candidateContent;
|
|
260
605
|
bestScore = candidateScore;
|
|
@@ -264,23 +609,36 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
264
609
|
else {
|
|
265
610
|
consecutiveRejects++;
|
|
266
611
|
}
|
|
267
|
-
|
|
612
|
+
if (accepted)
|
|
613
|
+
roundReports.push({ round, accepted, report: candidateReport });
|
|
614
|
+
const roundDelta = candidateScore - trajectory[trajectory.length - 1].score;
|
|
615
|
+
trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost });
|
|
268
616
|
if (onRoundProgress)
|
|
269
|
-
onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta, accepted, costUSD: roundCost, costReported: roundCostReported });
|
|
617
|
+
onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported });
|
|
270
618
|
// Early stop
|
|
271
|
-
if (
|
|
619
|
+
if (stopOnAssertionsPass && accepted && allNonTripwireAssertionsPass(candidateReport, candidateVariantKey)) {
|
|
620
|
+
stopReason = 'assertions-pass';
|
|
272
621
|
break;
|
|
273
|
-
|
|
622
|
+
}
|
|
623
|
+
if (target && bestScore >= target) {
|
|
624
|
+
stopReason = 'target';
|
|
625
|
+
break;
|
|
626
|
+
}
|
|
627
|
+
if (consecutiveRejects >= 2) {
|
|
628
|
+
stopReason = 'consecutive-rejects';
|
|
274
629
|
break;
|
|
630
|
+
}
|
|
631
|
+
}
|
|
632
|
+
// Write best version back to original file only if an improvement was accepted
|
|
633
|
+
if (bestRound > 0) {
|
|
634
|
+
writeFileSync(absSkillPath, currentBest);
|
|
275
635
|
}
|
|
276
|
-
// Write best version back to original file
|
|
277
|
-
writeFileSync(absSkillPath, currentBest);
|
|
278
636
|
// Merge all round reports into one and persist
|
|
279
637
|
let reportId;
|
|
280
638
|
if (roundReports.length > 0) {
|
|
281
639
|
// load samples once to enable analysis.sampleQuality on the merged report.
|
|
282
640
|
const { samples } = loadSamples(absSamplesPath);
|
|
283
|
-
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples);
|
|
641
|
+
const mergedReport = mergeEvolveReports(roundReports, skillName, totalCostUSD, samples, absSkillPath);
|
|
284
642
|
persistReport(mergedReport, DEFAULT_OUTPUT_DIR);
|
|
285
643
|
reportId = mergedReport.id;
|
|
286
644
|
}
|
|
@@ -290,6 +648,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
290
648
|
bestRound,
|
|
291
649
|
totalRounds: trajectory.length - 1, // excluding baseline
|
|
292
650
|
totalCostUSD: Number(totalCostUSD.toFixed(6)),
|
|
651
|
+
stopReason,
|
|
652
|
+
...(sampleFixes.length > 0 && { sampleFixes }),
|
|
653
|
+
...(reusedBaselineReportId && { reusedBaselineReportId }),
|
|
293
654
|
...(totalCostReported ? {} : { costReported: false }),
|
|
294
655
|
trajectory,
|
|
295
656
|
bestSkillPath: allVersions[bestRound],
|