oh-my-knowledge 0.32.0 → 0.34.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -14
- package/README.zh.md +32 -17
- package/dist/analysis/coverage-analyzer.d.ts +0 -1
- package/dist/analysis/coverage-analyzer.js +0 -1
- package/dist/analysis/failure-clusterer.d.ts +0 -1
- package/dist/analysis/failure-clusterer.js +0 -1
- package/dist/analysis/gap-analyzer.d.ts +0 -1
- package/dist/analysis/gap-analyzer.js +0 -1
- package/dist/analysis/hedging-classifier.d.ts +0 -1
- package/dist/analysis/hedging-classifier.js +1 -2
- package/dist/analysis/report-diagnostics.d.ts +0 -1
- package/dist/analysis/report-diagnostics.js +0 -1
- package/dist/analysis/sample-diagnostics.d.ts +1 -2
- package/dist/analysis/sample-diagnostics.js +18 -19
- package/dist/analysis/saturation.d.ts +8 -1
- package/dist/analysis/saturation.js +12 -5
- package/dist/assets/agent-skills/omk/SKILL.md +197 -0
- package/dist/assets/agent-skills/omk/references/commands.md +547 -0
- package/dist/authoring/evolver.d.ts +169 -4
- package/dist/authoring/evolver.js +287 -15
- package/dist/authoring/generator.d.ts +29 -2
- package/dist/authoring/generator.js +113 -1
- package/dist/authoring/sample-fixer.d.ts +0 -1
- package/dist/authoring/sample-fixer.js +0 -1
- package/dist/cli/commands/doctor.d.ts +2 -1
- package/dist/cli/commands/doctor.js +24 -6
- package/dist/cli/commands/eval/gold/compare.d.ts +0 -1
- package/dist/cli/commands/eval/gold/compare.js +0 -1
- package/dist/cli/commands/eval/gold/index.d.ts +0 -1
- package/dist/cli/commands/eval/gold/index.js +0 -1
- package/dist/cli/commands/eval/gold/init.d.ts +0 -1
- package/dist/cli/commands/eval/gold/init.js +0 -1
- package/dist/cli/commands/eval/gold/validate.d.ts +0 -1
- package/dist/cli/commands/eval/gold/validate.js +0 -1
- package/dist/cli/commands/eval/index.d.ts +2 -1
- package/dist/cli/commands/eval/index.js +23 -11
- package/dist/cli/commands/evolve.d.ts +7 -1
- package/dist/cli/commands/evolve.js +93 -6
- package/dist/cli/commands/init.d.ts +0 -1
- package/dist/cli/commands/init.js +0 -1
- package/dist/cli/commands/install.d.ts +22 -0
- package/dist/cli/commands/install.js +411 -0
- package/dist/cli/commands/observe/inbox.d.ts +0 -1
- package/dist/cli/commands/observe/inbox.js +0 -1
- package/dist/cli/commands/observe/index.d.ts +0 -1
- package/dist/cli/commands/observe/index.js +6 -3
- package/dist/cli/commands/observe/ingest.d.ts +0 -1
- package/dist/cli/commands/observe/ingest.js +0 -1
- package/dist/cli/commands/observe/show.d.ts +0 -1
- package/dist/cli/commands/observe/show.js +0 -1
- package/dist/cli/commands/sample.d.ts +2 -1
- package/dist/cli/commands/sample.js +100 -9
- package/dist/cli/commands/studio.d.ts +0 -1
- package/dist/cli/commands/studio.js +0 -1
- package/dist/cli/index.d.ts +0 -1
- package/dist/cli/index.js +1 -2
- package/dist/cli/lib/cli-exit.d.ts +0 -1
- package/dist/cli/lib/cli-exit.js +0 -1
- package/dist/cli/lib/cmd-flags.d.ts +11 -1
- package/dist/cli/lib/cmd-flags.js +0 -1
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -2
- package/dist/cli/lib/i18n-dict/common.js +18 -3
- package/dist/cli/lib/i18n-dict/evolve.d.ts +1 -2
- package/dist/cli/lib/i18n-dict/evolve.js +24 -1
- package/dist/cli/lib/i18n-dict/gen.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/gen.js +0 -1
- package/dist/cli/lib/i18n-dict/help.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/help.js +0 -1
- package/dist/cli/lib/i18n-dict/init.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/init.js +0 -1
- package/dist/cli/lib/i18n-dict/install.d.ts +3 -0
- package/dist/cli/lib/i18n-dict/install.js +90 -0
- package/dist/cli/lib/i18n-dict/run.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/run.js +0 -1
- package/dist/cli/lib/i18n-dict/types.d.ts +0 -1
- package/dist/cli/lib/i18n-dict/types.js +0 -1
- package/dist/cli/lib/i18n-dict.d.ts +2 -2
- package/dist/cli/lib/i18n-dict.js +2 -1
- package/dist/cli/lib/i18n.d.ts +0 -1
- package/dist/cli/lib/i18n.js +0 -1
- package/dist/cli/lib/parse-run-config/judge-models.d.ts +0 -1
- package/dist/cli/lib/parse-run-config/judge-models.js +0 -1
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts +0 -1
- package/dist/cli/lib/parse-run-config/samples-discovery.js +1 -4
- package/dist/cli/lib/parse-run-config/variant-resolution.d.ts +8 -4
- package/dist/cli/lib/parse-run-config/variant-resolution.js +46 -14
- package/dist/cli/lib/parse-run-config.d.ts +0 -1
- package/dist/cli/lib/parse-run-config.js +1 -2
- package/dist/cli/lib/progress.d.ts +0 -1
- package/dist/cli/lib/progress.js +0 -1
- package/dist/cli/lib/resolve-skill-input.d.ts +0 -1
- package/dist/cli/lib/resolve-skill-input.js +7 -6
- package/dist/cli/lib/run-tally.d.ts +0 -1
- package/dist/cli/lib/run-tally.js +1 -2
- package/dist/cli/lib/shared.d.ts +0 -1
- package/dist/cli/lib/shared.js +1 -2
- package/dist/cli/lib/update-check.d.ts +65 -1
- package/dist/cli/lib/update-check.js +220 -32
- package/dist/cli/lib/update-fetch-worker.d.ts +1 -0
- package/dist/cli/lib/update-fetch-worker.js +34 -0
- package/dist/cli/oclif/base-command.d.ts +0 -1
- package/dist/cli/oclif/base-command.js +0 -1
- package/dist/cli/oclif/help.d.ts +0 -1
- package/dist/cli/oclif/help.js +0 -1
- package/dist/cli/oclif/i18n.d.ts +0 -1
- package/dist/cli/oclif/i18n.js +0 -1
- package/dist/cli/oclif/parsers.d.ts +0 -1
- package/dist/cli/oclif/parsers.js +0 -1
- package/dist/cli/oclif/projection.d.ts +0 -1
- package/dist/cli/oclif/projection.js +0 -1
- package/dist/cli/oclif/run.d.ts +0 -1
- package/dist/cli/oclif/run.js +0 -1
- package/dist/diagnosis/observe-mapper.d.ts +2 -3
- package/dist/diagnosis/observe-mapper.js +6 -7
- package/dist/diagnosis/observe-producer.d.ts +0 -1
- package/dist/diagnosis/observe-producer.js +3 -4
- package/dist/diagnosis/studio-projection.d.ts +0 -1
- package/dist/diagnosis/studio-projection.js +0 -1
- package/dist/diagnosis/types.d.ts +0 -1
- package/dist/diagnosis/types.js +0 -1
- package/dist/doctor/fixer.d.ts +0 -1
- package/dist/doctor/fixer.js +0 -1
- package/dist/doctor/health/builtin-dimensions.d.ts +0 -1
- package/dist/doctor/health/builtin-dimensions.js +0 -1
- package/dist/doctor/health/composer.d.ts +0 -1
- package/dist/doctor/health/composer.js +1 -2
- package/dist/doctor/health/dimension-registry.d.ts +0 -1
- package/dist/doctor/health/dimension-registry.js +0 -1
- package/dist/doctor/health/dimension-spec.d.ts +0 -1
- package/dist/doctor/health/dimension-spec.js +0 -1
- package/dist/doctor/health/load-custom-dimensions.d.ts +1 -0
- package/dist/doctor/health/load-custom-dimensions.js +30 -0
- package/dist/doctor/health/parser.d.ts +0 -1
- package/dist/doctor/health/parser.js +0 -1
- package/dist/doctor/health/prompt-builder.d.ts +0 -1
- package/dist/doctor/health/prompt-builder.js +0 -1
- package/dist/doctor/health/register.d.ts +0 -1
- package/dist/doctor/health/register.js +0 -1
- package/dist/doctor/index.d.ts +0 -1
- package/dist/doctor/index.js +1 -2
- package/dist/doctor/messages.d.ts +0 -1
- package/dist/doctor/messages.js +1 -2
- package/dist/doctor/preflight.d.ts +0 -1
- package/dist/doctor/preflight.js +0 -1
- package/dist/doctor/renderer.d.ts +0 -1
- package/dist/doctor/renderer.js +0 -1
- package/dist/doctor/rules.d.ts +0 -1
- package/dist/doctor/rules.js +0 -1
- package/dist/eval-core/bootstrap.d.ts +8 -1
- package/dist/eval-core/bootstrap.js +11 -4
- package/dist/eval-core/cache.d.ts +0 -1
- package/dist/eval-core/cache.js +0 -1
- package/dist/eval-core/comparability.d.ts +0 -1
- package/dist/eval-core/comparability.js +3 -4
- package/dist/eval-core/dependency-checker.d.ts +0 -1
- package/dist/eval-core/dependency-checker.js +2 -2
- package/dist/eval-core/evaluation-execution.d.ts +0 -1
- package/dist/eval-core/evaluation-execution.js +0 -1
- package/dist/eval-core/evaluation-job.d.ts +0 -1
- package/dist/eval-core/evaluation-job.js +0 -1
- package/dist/eval-core/evaluation-reporting.d.ts +0 -1
- package/dist/eval-core/evaluation-reporting.js +9 -8
- package/dist/eval-core/execution-strategy.d.ts +0 -1
- package/dist/eval-core/execution-strategy.js +6 -3
- package/dist/eval-core/fact-checker.d.ts +0 -1
- package/dist/eval-core/fact-checker.js +0 -1
- package/dist/eval-core/layer-gates.d.ts +0 -1
- package/dist/eval-core/layer-gates.js +0 -1
- package/dist/eval-core/mocks-runtime.d.ts +0 -1
- package/dist/eval-core/mocks-runtime.js +0 -1
- package/dist/eval-core/schema.d.ts +0 -1
- package/dist/eval-core/schema.js +0 -1
- package/dist/eval-core/statistics.d.ts +0 -1
- package/dist/eval-core/statistics.js +0 -1
- package/dist/eval-core/task-planner.d.ts +0 -1
- package/dist/eval-core/task-planner.js +0 -1
- package/dist/eval-core/verdict.d.ts +21 -4
- package/dist/eval-core/verdict.js +54 -5
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +18 -3
- package/dist/eval-workflows/batch-evaluation-workflow.js +31 -15
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +0 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +0 -1
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +0 -1
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +0 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +0 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +0 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +0 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +0 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +0 -1
- package/dist/eval-workflows/evaluation-pipeline.js +0 -1
- package/dist/eval-workflows/evaluation-preparation.d.ts +1 -5
- package/dist/eval-workflows/evaluation-preparation.js +20 -16
- package/dist/eval-workflows/messages.d.ts +0 -1
- package/dist/eval-workflows/messages.js +0 -1
- package/dist/eval-workflows/run-evaluation.d.ts +4 -9
- package/dist/eval-workflows/run-evaluation.js +16 -26
- package/dist/executors/anthropic-api.d.ts +0 -1
- package/dist/executors/anthropic-api.js +1 -2
- package/dist/executors/claude-cli.d.ts +0 -1
- package/dist/executors/claude-cli.js +0 -1
- package/dist/executors/claude-sdk-trace.d.ts +0 -1
- package/dist/executors/claude-sdk-trace.js +0 -1
- package/dist/executors/claude-sdk.d.ts +0 -1
- package/dist/executors/claude-sdk.js +0 -1
- package/dist/executors/codex-cli-trace.d.ts +0 -1
- package/dist/executors/codex-cli-trace.js +0 -1
- package/dist/executors/codex-cli.d.ts +0 -1
- package/dist/executors/codex-cli.js +0 -1
- package/dist/executors/codex-sdk.d.ts +0 -1
- package/dist/executors/codex-sdk.js +0 -1
- package/dist/executors/gemini.d.ts +0 -1
- package/dist/executors/gemini.js +0 -1
- package/dist/executors/index.d.ts +0 -1
- package/dist/executors/index.js +0 -1
- package/dist/executors/openai-api.d.ts +0 -1
- package/dist/executors/openai-api.js +0 -1
- package/dist/executors/runtime-fingerprint.d.ts +0 -1
- package/dist/executors/runtime-fingerprint.js +2 -3
- package/dist/executors/script.d.ts +0 -1
- package/dist/executors/script.js +0 -1
- package/dist/executors/shared.d.ts +0 -2
- package/dist/executors/shared.js +0 -1
- package/dist/grading/assertions.d.ts +0 -1
- package/dist/grading/assertions.js +0 -1
- package/dist/grading/debias-validate.d.ts +0 -1
- package/dist/grading/debias-validate.js +0 -1
- package/dist/grading/diagnostic.d.ts +0 -1
- package/dist/grading/diagnostic.js +0 -1
- package/dist/grading/gold-cli.d.ts +0 -1
- package/dist/grading/gold-cli.js +0 -1
- package/dist/grading/gold-dataset.d.ts +0 -1
- package/dist/grading/gold-dataset.js +0 -1
- package/dist/grading/human-gold.d.ts +0 -1
- package/dist/grading/human-gold.js +0 -1
- package/dist/grading/index.d.ts +0 -1
- package/dist/grading/index.js +0 -1
- package/dist/grading/judge.d.ts +0 -1
- package/dist/grading/judge.js +0 -1
- package/dist/grading/layered-scores.d.ts +0 -1
- package/dist/grading/layered-scores.js +0 -1
- package/dist/inputs/eval-config.d.ts +0 -1
- package/dist/inputs/eval-config.js +3 -2
- package/dist/inputs/load-samples.d.ts +0 -1
- package/dist/inputs/load-samples.js +0 -1
- package/dist/inputs/mcp-resolver.d.ts +0 -1
- package/dist/inputs/mcp-resolver.js +0 -1
- package/dist/inputs/skill-loader.d.ts +99 -7
- package/dist/inputs/skill-loader.js +325 -45
- package/dist/inputs/source-resolver.d.ts +28 -0
- package/dist/inputs/source-resolver.js +125 -0
- package/dist/inputs/url-fetcher.d.ts +0 -1
- package/dist/inputs/url-fetcher.js +0 -1
- package/dist/managed/index.d.ts +5 -0
- package/dist/managed/index.js +5 -0
- package/dist/managed/store.d.ts +76 -0
- package/dist/managed/store.js +260 -0
- package/dist/observability/experience-frontmatter.d.ts +0 -1
- package/dist/observability/experience-frontmatter.js +0 -1
- package/dist/observability/experience.d.ts +1 -2
- package/dist/observability/experience.js +1 -2
- package/dist/observability/feedback-matchers.d.ts +0 -1
- package/dist/observability/feedback-matchers.js +0 -1
- package/dist/observability/feedback-projection.d.ts +0 -1
- package/dist/observability/feedback-projection.js +0 -1
- package/dist/observability/inbox-view-model.d.ts +0 -1
- package/dist/observability/inbox-view-model.js +0 -1
- package/dist/observability/inbox.d.ts +0 -1
- package/dist/observability/inbox.js +2 -3
- package/dist/observability/problem-patterns.d.ts +0 -1
- package/dist/observability/problem-patterns.js +0 -1
- package/dist/observability/resolved-review.d.ts +0 -1
- package/dist/observability/resolved-review.js +4 -5
- package/dist/observability/review-state.d.ts +0 -1
- package/dist/observability/review-state.js +3 -4
- package/dist/observability/skill-chain-advisories.d.ts +0 -1
- package/dist/observability/skill-chain-advisories.js +0 -1
- package/dist/observability/skill-chain.d.ts +0 -1
- package/dist/observability/skill-chain.js +5 -6
- package/dist/observability/skill-health-analyzer.d.ts +13 -1
- package/dist/observability/skill-health-analyzer.js +17 -3
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +0 -1
- package/dist/observability/soft-standards/index.js +0 -1
- package/dist/observability/soft-standards/llm-extractor.d.ts +0 -1
- package/dist/observability/soft-standards/llm-extractor.js +7 -8
- package/dist/observability/soft-standards/runtime-evaluator.d.ts +0 -1
- package/dist/observability/soft-standards/runtime-evaluator.js +2 -3
- package/dist/observability/soft-standards/skill-standards-store.d.ts +0 -1
- package/dist/observability/soft-standards/skill-standards-store.js +5 -6
- package/dist/observability/soft-standards/types.d.ts +10 -11
- package/dist/observability/soft-standards/types.js +0 -1
- package/dist/observability/text-signals.d.ts +0 -1
- package/dist/observability/text-signals.js +0 -1
- package/dist/observability/trace-adapter.d.ts +0 -1
- package/dist/observability/trace-adapter.js +0 -1
- package/dist/observability/trace-attribution.d.ts +0 -1
- package/dist/observability/trace-attribution.js +0 -1
- package/dist/observability/trace-segmenter.d.ts +0 -1
- package/dist/observability/trace-segmenter.js +0 -1
- package/dist/observability/trace-source.d.ts +0 -1
- package/dist/observability/trace-source.js +0 -1
- package/dist/renderer/html-renderer.d.ts +0 -1
- package/dist/renderer/html-renderer.js +4 -5
- package/dist/renderer/layout.d.ts +0 -1
- package/dist/renderer/layout.js +2 -1
- package/dist/renderer/observation-inbox/helpers.d.ts +0 -1
- package/dist/renderer/observation-inbox/helpers.js +0 -1
- package/dist/renderer/observation-inbox/styles.d.ts +0 -1
- package/dist/renderer/observation-inbox/styles.js +0 -1
- package/dist/renderer/observation-inbox-renderer.d.ts +0 -1
- package/dist/renderer/observation-inbox-renderer.js +13 -14
- package/dist/renderer/skill-detail-renderer.d.ts +0 -1
- package/dist/renderer/skill-detail-renderer.js +101 -23
- package/dist/renderer/skill-health-renderer.d.ts +0 -1
- package/dist/renderer/skill-health-renderer.js +33 -5
- package/dist/renderer/skill-list-renderer.d.ts +0 -1
- package/dist/renderer/skill-list-renderer.js +18 -7
- package/dist/renderer/summary.d.ts +0 -1
- package/dist/renderer/summary.js +21 -10
- package/dist/renderer/table.d.ts +0 -1
- package/dist/renderer/table.js +0 -1
- package/dist/renderer/test-view.d.ts +0 -1
- package/dist/renderer/test-view.js +0 -1
- package/dist/renderer/trends.d.ts +0 -1
- package/dist/renderer/trends.js +0 -1
- package/dist/server/job-store.d.ts +0 -1
- package/dist/server/job-store.js +0 -1
- package/dist/server/report-server.d.ts +0 -1
- package/dist/server/report-server.js +10 -4
- package/dist/server/report-store.d.ts +1 -2
- package/dist/server/report-store.js +6 -7
- package/dist/server/skill-index.d.ts +0 -1
- package/dist/server/skill-index.js +7 -5
- package/dist/server/skill-insights.d.ts +0 -1
- package/dist/server/skill-insights.js +41 -14
- package/dist/shared/hard-rules.d.ts +0 -1
- package/dist/shared/hard-rules.js +0 -1
- package/dist/shared/llm-prompts/index.d.ts +0 -1
- package/dist/shared/llm-prompts/index.js +0 -1
- package/dist/shared/llm-prompts/skill-health.d.ts +0 -1
- package/dist/shared/llm-prompts/skill-health.js +0 -1
- package/dist/shared/time.d.ts +0 -1
- package/dist/shared/time.js +0 -1
- package/dist/shared/tool-search.d.ts +0 -1
- package/dist/shared/tool-search.js +0 -1
- package/dist/types/dependencies.d.ts +0 -1
- package/dist/types/dependencies.js +0 -1
- package/dist/types/diagnosis.d.ts +0 -1
- package/dist/types/diagnosis.js +0 -1
- package/dist/types/doctor.d.ts +4 -5
- package/dist/types/doctor.js +2 -3
- package/dist/types/eval.d.ts +2 -1
- package/dist/types/eval.js +0 -1
- package/dist/types/executor.d.ts +1 -2
- package/dist/types/executor.js +0 -1
- package/dist/types/index.d.ts +1 -1
- package/dist/types/index.js +1 -1
- package/dist/types/judge.d.ts +0 -1
- package/dist/types/judge.js +0 -1
- package/dist/types/managed.d.ts +85 -0
- package/dist/types/managed.js +1 -0
- package/dist/types/observability.d.ts +5 -6
- package/dist/types/observability.js +0 -1
- package/dist/types/report.d.ts +2 -3
- package/dist/types/report.js +0 -1
- package/dist/types/shared.d.ts +0 -1
- package/dist/types/shared.js +0 -1
- package/dist/types/skill-index.d.ts +3 -1
- package/dist/types/skill-index.js +0 -1
- package/dist/types/storage.d.ts +0 -1
- package/dist/types/storage.js +0 -1
- package/dist/util/safe-slice.d.ts +0 -1
- package/dist/util/safe-slice.js +0 -1
- package/package.json +10 -5
- package/dist/analysis/coverage-analyzer.d.ts.map +0 -1
- package/dist/analysis/coverage-analyzer.js.map +0 -1
- package/dist/analysis/failure-clusterer.d.ts.map +0 -1
- package/dist/analysis/failure-clusterer.js.map +0 -1
- package/dist/analysis/gap-analyzer.d.ts.map +0 -1
- package/dist/analysis/gap-analyzer.js.map +0 -1
- package/dist/analysis/hedging-classifier.d.ts.map +0 -1
- package/dist/analysis/hedging-classifier.js.map +0 -1
- package/dist/analysis/report-diagnostics.d.ts.map +0 -1
- package/dist/analysis/report-diagnostics.js.map +0 -1
- package/dist/analysis/sample-diagnostics.d.ts.map +0 -1
- package/dist/analysis/sample-diagnostics.js.map +0 -1
- package/dist/analysis/saturation.d.ts.map +0 -1
- package/dist/analysis/saturation.js.map +0 -1
- package/dist/authoring/evolver.d.ts.map +0 -1
- package/dist/authoring/evolver.js.map +0 -1
- package/dist/authoring/generator.d.ts.map +0 -1
- package/dist/authoring/generator.js.map +0 -1
- package/dist/authoring/sample-fixer.d.ts.map +0 -1
- package/dist/authoring/sample-fixer.js.map +0 -1
- package/dist/cli/commands/doctor.d.ts.map +0 -1
- package/dist/cli/commands/doctor.js.map +0 -1
- package/dist/cli/commands/eval/gold/compare.d.ts.map +0 -1
- package/dist/cli/commands/eval/gold/compare.js.map +0 -1
- package/dist/cli/commands/eval/gold/index.d.ts.map +0 -1
- package/dist/cli/commands/eval/gold/index.js.map +0 -1
- package/dist/cli/commands/eval/gold/init.d.ts.map +0 -1
- package/dist/cli/commands/eval/gold/init.js.map +0 -1
- package/dist/cli/commands/eval/gold/validate.d.ts.map +0 -1
- package/dist/cli/commands/eval/gold/validate.js.map +0 -1
- package/dist/cli/commands/eval/index.d.ts.map +0 -1
- package/dist/cli/commands/eval/index.js.map +0 -1
- package/dist/cli/commands/evolve.d.ts.map +0 -1
- package/dist/cli/commands/evolve.js.map +0 -1
- package/dist/cli/commands/init.d.ts.map +0 -1
- package/dist/cli/commands/init.js.map +0 -1
- package/dist/cli/commands/observe/inbox.d.ts.map +0 -1
- package/dist/cli/commands/observe/inbox.js.map +0 -1
- package/dist/cli/commands/observe/index.d.ts.map +0 -1
- package/dist/cli/commands/observe/index.js.map +0 -1
- package/dist/cli/commands/observe/ingest.d.ts.map +0 -1
- package/dist/cli/commands/observe/ingest.js.map +0 -1
- package/dist/cli/commands/observe/show.d.ts.map +0 -1
- package/dist/cli/commands/observe/show.js.map +0 -1
- package/dist/cli/commands/sample.d.ts.map +0 -1
- package/dist/cli/commands/sample.js.map +0 -1
- package/dist/cli/commands/studio.d.ts.map +0 -1
- package/dist/cli/commands/studio.js.map +0 -1
- package/dist/cli/index.d.ts.map +0 -1
- package/dist/cli/index.js.map +0 -1
- package/dist/cli/lib/cli-exit.d.ts.map +0 -1
- package/dist/cli/lib/cli-exit.js.map +0 -1
- package/dist/cli/lib/cmd-flags.d.ts.map +0 -1
- package/dist/cli/lib/cmd-flags.js.map +0 -1
- package/dist/cli/lib/i18n-dict/common.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/common.js.map +0 -1
- package/dist/cli/lib/i18n-dict/evolve.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/evolve.js.map +0 -1
- package/dist/cli/lib/i18n-dict/gen.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/gen.js.map +0 -1
- package/dist/cli/lib/i18n-dict/help.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/help.js.map +0 -1
- package/dist/cli/lib/i18n-dict/init.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/init.js.map +0 -1
- package/dist/cli/lib/i18n-dict/run.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/run.js.map +0 -1
- package/dist/cli/lib/i18n-dict/types.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict/types.js.map +0 -1
- package/dist/cli/lib/i18n-dict.d.ts.map +0 -1
- package/dist/cli/lib/i18n-dict.js.map +0 -1
- package/dist/cli/lib/i18n.d.ts.map +0 -1
- package/dist/cli/lib/i18n.js.map +0 -1
- package/dist/cli/lib/parse-run-config/judge-models.d.ts.map +0 -1
- package/dist/cli/lib/parse-run-config/judge-models.js.map +0 -1
- package/dist/cli/lib/parse-run-config/samples-discovery.d.ts.map +0 -1
- package/dist/cli/lib/parse-run-config/samples-discovery.js.map +0 -1
- package/dist/cli/lib/parse-run-config/variant-resolution.d.ts.map +0 -1
- package/dist/cli/lib/parse-run-config/variant-resolution.js.map +0 -1
- package/dist/cli/lib/parse-run-config.d.ts.map +0 -1
- package/dist/cli/lib/parse-run-config.js.map +0 -1
- package/dist/cli/lib/progress.d.ts.map +0 -1
- package/dist/cli/lib/progress.js.map +0 -1
- package/dist/cli/lib/resolve-skill-input.d.ts.map +0 -1
- package/dist/cli/lib/resolve-skill-input.js.map +0 -1
- package/dist/cli/lib/run-tally.d.ts.map +0 -1
- package/dist/cli/lib/run-tally.js.map +0 -1
- package/dist/cli/lib/shared.d.ts.map +0 -1
- package/dist/cli/lib/shared.js.map +0 -1
- package/dist/cli/lib/update-check.d.ts.map +0 -1
- package/dist/cli/lib/update-check.js.map +0 -1
- package/dist/cli/oclif/base-command.d.ts.map +0 -1
- package/dist/cli/oclif/base-command.js.map +0 -1
- package/dist/cli/oclif/help.d.ts.map +0 -1
- package/dist/cli/oclif/help.js.map +0 -1
- package/dist/cli/oclif/i18n.d.ts.map +0 -1
- package/dist/cli/oclif/i18n.js.map +0 -1
- package/dist/cli/oclif/parsers.d.ts.map +0 -1
- package/dist/cli/oclif/parsers.js.map +0 -1
- package/dist/cli/oclif/projection.d.ts.map +0 -1
- package/dist/cli/oclif/projection.js.map +0 -1
- package/dist/cli/oclif/run.d.ts.map +0 -1
- package/dist/cli/oclif/run.js.map +0 -1
- package/dist/diagnosis/observe-mapper.d.ts.map +0 -1
- package/dist/diagnosis/observe-mapper.js.map +0 -1
- package/dist/diagnosis/observe-producer.d.ts.map +0 -1
- package/dist/diagnosis/observe-producer.js.map +0 -1
- package/dist/diagnosis/studio-projection.d.ts.map +0 -1
- package/dist/diagnosis/studio-projection.js.map +0 -1
- package/dist/diagnosis/types.d.ts.map +0 -1
- package/dist/diagnosis/types.js.map +0 -1
- package/dist/doctor/fixer.d.ts.map +0 -1
- package/dist/doctor/fixer.js.map +0 -1
- package/dist/doctor/health/builtin-dimensions.d.ts.map +0 -1
- package/dist/doctor/health/builtin-dimensions.js.map +0 -1
- package/dist/doctor/health/composer.d.ts.map +0 -1
- package/dist/doctor/health/composer.js.map +0 -1
- package/dist/doctor/health/dimension-registry.d.ts.map +0 -1
- package/dist/doctor/health/dimension-registry.js.map +0 -1
- package/dist/doctor/health/dimension-spec.d.ts.map +0 -1
- package/dist/doctor/health/dimension-spec.js.map +0 -1
- package/dist/doctor/health/parser.d.ts.map +0 -1
- package/dist/doctor/health/parser.js.map +0 -1
- package/dist/doctor/health/prompt-builder.d.ts.map +0 -1
- package/dist/doctor/health/prompt-builder.js.map +0 -1
- package/dist/doctor/health/register.d.ts.map +0 -1
- package/dist/doctor/health/register.js.map +0 -1
- package/dist/doctor/index.d.ts.map +0 -1
- package/dist/doctor/index.js.map +0 -1
- package/dist/doctor/messages.d.ts.map +0 -1
- package/dist/doctor/messages.js.map +0 -1
- package/dist/doctor/preflight.d.ts.map +0 -1
- package/dist/doctor/preflight.js.map +0 -1
- package/dist/doctor/renderer.d.ts.map +0 -1
- package/dist/doctor/renderer.js.map +0 -1
- package/dist/doctor/rules.d.ts.map +0 -1
- package/dist/doctor/rules.js.map +0 -1
- package/dist/eval-core/bootstrap.d.ts.map +0 -1
- package/dist/eval-core/bootstrap.js.map +0 -1
- package/dist/eval-core/cache.d.ts.map +0 -1
- package/dist/eval-core/cache.js.map +0 -1
- package/dist/eval-core/comparability.d.ts.map +0 -1
- package/dist/eval-core/comparability.js.map +0 -1
- package/dist/eval-core/dependency-checker.d.ts.map +0 -1
- package/dist/eval-core/dependency-checker.js.map +0 -1
- package/dist/eval-core/evaluation-execution.d.ts.map +0 -1
- package/dist/eval-core/evaluation-execution.js.map +0 -1
- package/dist/eval-core/evaluation-job.d.ts.map +0 -1
- package/dist/eval-core/evaluation-job.js.map +0 -1
- package/dist/eval-core/evaluation-reporting.d.ts.map +0 -1
- package/dist/eval-core/evaluation-reporting.js.map +0 -1
- package/dist/eval-core/execution-strategy.d.ts.map +0 -1
- package/dist/eval-core/execution-strategy.js.map +0 -1
- package/dist/eval-core/fact-checker.d.ts.map +0 -1
- package/dist/eval-core/fact-checker.js.map +0 -1
- package/dist/eval-core/layer-gates.d.ts.map +0 -1
- package/dist/eval-core/layer-gates.js.map +0 -1
- package/dist/eval-core/mocks-runtime.d.ts.map +0 -1
- package/dist/eval-core/mocks-runtime.js.map +0 -1
- package/dist/eval-core/schema.d.ts.map +0 -1
- package/dist/eval-core/schema.js.map +0 -1
- package/dist/eval-core/statistics.d.ts.map +0 -1
- package/dist/eval-core/statistics.js.map +0 -1
- package/dist/eval-core/task-planner.d.ts.map +0 -1
- package/dist/eval-core/task-planner.js.map +0 -1
- package/dist/eval-core/verdict.d.ts.map +0 -1
- package/dist/eval-core/verdict.js.map +0 -1
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts.map +0 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-pipeline.js.map +0 -1
- package/dist/eval-workflows/evaluation-preparation.d.ts.map +0 -1
- package/dist/eval-workflows/evaluation-preparation.js.map +0 -1
- package/dist/eval-workflows/messages.d.ts.map +0 -1
- package/dist/eval-workflows/messages.js.map +0 -1
- package/dist/eval-workflows/run-evaluation.d.ts.map +0 -1
- package/dist/eval-workflows/run-evaluation.js.map +0 -1
- package/dist/executors/anthropic-api.d.ts.map +0 -1
- package/dist/executors/anthropic-api.js.map +0 -1
- package/dist/executors/claude-cli.d.ts.map +0 -1
- package/dist/executors/claude-cli.js.map +0 -1
- package/dist/executors/claude-sdk-trace.d.ts.map +0 -1
- package/dist/executors/claude-sdk-trace.js.map +0 -1
- package/dist/executors/claude-sdk.d.ts.map +0 -1
- package/dist/executors/claude-sdk.js.map +0 -1
- package/dist/executors/codex-cli-trace.d.ts.map +0 -1
- package/dist/executors/codex-cli-trace.js.map +0 -1
- package/dist/executors/codex-cli.d.ts.map +0 -1
- package/dist/executors/codex-cli.js.map +0 -1
- package/dist/executors/codex-sdk.d.ts.map +0 -1
- package/dist/executors/codex-sdk.js.map +0 -1
- package/dist/executors/gemini.d.ts.map +0 -1
- package/dist/executors/gemini.js.map +0 -1
- package/dist/executors/index.d.ts.map +0 -1
- package/dist/executors/index.js.map +0 -1
- package/dist/executors/openai-api.d.ts.map +0 -1
- package/dist/executors/openai-api.js.map +0 -1
- package/dist/executors/runtime-fingerprint.d.ts.map +0 -1
- package/dist/executors/runtime-fingerprint.js.map +0 -1
- package/dist/executors/script.d.ts.map +0 -1
- package/dist/executors/script.js.map +0 -1
- package/dist/executors/shared.d.ts.map +0 -1
- package/dist/executors/shared.js.map +0 -1
- package/dist/grading/assertions.d.ts.map +0 -1
- package/dist/grading/assertions.js.map +0 -1
- package/dist/grading/debias-validate.d.ts.map +0 -1
- package/dist/grading/debias-validate.js.map +0 -1
- package/dist/grading/diagnostic.d.ts.map +0 -1
- package/dist/grading/diagnostic.js.map +0 -1
- package/dist/grading/gold-cli.d.ts.map +0 -1
- package/dist/grading/gold-cli.js.map +0 -1
- package/dist/grading/gold-dataset.d.ts.map +0 -1
- package/dist/grading/gold-dataset.js.map +0 -1
- package/dist/grading/human-gold.d.ts.map +0 -1
- package/dist/grading/human-gold.js.map +0 -1
- package/dist/grading/index.d.ts.map +0 -1
- package/dist/grading/index.js.map +0 -1
- package/dist/grading/judge.d.ts.map +0 -1
- package/dist/grading/judge.js.map +0 -1
- package/dist/grading/layered-scores.d.ts.map +0 -1
- package/dist/grading/layered-scores.js.map +0 -1
- package/dist/inputs/eval-config.d.ts.map +0 -1
- package/dist/inputs/eval-config.js.map +0 -1
- package/dist/inputs/load-samples.d.ts.map +0 -1
- package/dist/inputs/load-samples.js.map +0 -1
- package/dist/inputs/mcp-resolver.d.ts.map +0 -1
- package/dist/inputs/mcp-resolver.js.map +0 -1
- package/dist/inputs/skill-loader.d.ts.map +0 -1
- package/dist/inputs/skill-loader.js.map +0 -1
- package/dist/inputs/url-fetcher.d.ts.map +0 -1
- package/dist/inputs/url-fetcher.js.map +0 -1
- package/dist/observability/experience-frontmatter.d.ts.map +0 -1
- package/dist/observability/experience-frontmatter.js.map +0 -1
- package/dist/observability/experience.d.ts.map +0 -1
- package/dist/observability/experience.js.map +0 -1
- package/dist/observability/feedback-matchers.d.ts.map +0 -1
- package/dist/observability/feedback-matchers.js.map +0 -1
- package/dist/observability/feedback-projection.d.ts.map +0 -1
- package/dist/observability/feedback-projection.js.map +0 -1
- package/dist/observability/inbox-view-model.d.ts.map +0 -1
- package/dist/observability/inbox-view-model.js.map +0 -1
- package/dist/observability/inbox.d.ts.map +0 -1
- package/dist/observability/inbox.js.map +0 -1
- package/dist/observability/problem-patterns.d.ts.map +0 -1
- package/dist/observability/problem-patterns.js.map +0 -1
- package/dist/observability/resolved-review.d.ts.map +0 -1
- package/dist/observability/resolved-review.js.map +0 -1
- package/dist/observability/review-state.d.ts.map +0 -1
- package/dist/observability/review-state.js.map +0 -1
- package/dist/observability/skill-chain-advisories.d.ts.map +0 -1
- package/dist/observability/skill-chain-advisories.js.map +0 -1
- package/dist/observability/skill-chain.d.ts.map +0 -1
- package/dist/observability/skill-chain.js.map +0 -1
- package/dist/observability/skill-health-analyzer.d.ts.map +0 -1
- package/dist/observability/skill-health-analyzer.js.map +0 -1
- package/dist/observability/soft-standards/constants.d.ts.map +0 -1
- package/dist/observability/soft-standards/constants.js.map +0 -1
- package/dist/observability/soft-standards/index.d.ts.map +0 -1
- package/dist/observability/soft-standards/index.js.map +0 -1
- package/dist/observability/soft-standards/llm-extractor.d.ts.map +0 -1
- package/dist/observability/soft-standards/llm-extractor.js.map +0 -1
- package/dist/observability/soft-standards/runtime-evaluator.d.ts.map +0 -1
- package/dist/observability/soft-standards/runtime-evaluator.js.map +0 -1
- package/dist/observability/soft-standards/skill-standards-store.d.ts.map +0 -1
- package/dist/observability/soft-standards/skill-standards-store.js.map +0 -1
- package/dist/observability/soft-standards/types.d.ts.map +0 -1
- package/dist/observability/soft-standards/types.js.map +0 -1
- package/dist/observability/text-signals.d.ts.map +0 -1
- package/dist/observability/text-signals.js.map +0 -1
- package/dist/observability/trace-adapter.d.ts.map +0 -1
- package/dist/observability/trace-adapter.js.map +0 -1
- package/dist/observability/trace-attribution.d.ts.map +0 -1
- package/dist/observability/trace-attribution.js.map +0 -1
- package/dist/observability/trace-segmenter.d.ts.map +0 -1
- package/dist/observability/trace-segmenter.js.map +0 -1
- package/dist/observability/trace-source.d.ts.map +0 -1
- package/dist/observability/trace-source.js.map +0 -1
- package/dist/renderer/html-renderer.d.ts.map +0 -1
- package/dist/renderer/html-renderer.js.map +0 -1
- package/dist/renderer/layout.d.ts.map +0 -1
- package/dist/renderer/layout.js.map +0 -1
- package/dist/renderer/observation-inbox/helpers.d.ts.map +0 -1
- package/dist/renderer/observation-inbox/helpers.js.map +0 -1
- package/dist/renderer/observation-inbox/styles.d.ts.map +0 -1
- package/dist/renderer/observation-inbox/styles.js.map +0 -1
- package/dist/renderer/observation-inbox-renderer.d.ts.map +0 -1
- package/dist/renderer/observation-inbox-renderer.js.map +0 -1
- package/dist/renderer/skill-detail-renderer.d.ts.map +0 -1
- package/dist/renderer/skill-detail-renderer.js.map +0 -1
- package/dist/renderer/skill-health-renderer.d.ts.map +0 -1
- package/dist/renderer/skill-health-renderer.js.map +0 -1
- package/dist/renderer/skill-list-renderer.d.ts.map +0 -1
- package/dist/renderer/skill-list-renderer.js.map +0 -1
- package/dist/renderer/summary.d.ts.map +0 -1
- package/dist/renderer/summary.js.map +0 -1
- package/dist/renderer/table.d.ts.map +0 -1
- package/dist/renderer/table.js.map +0 -1
- package/dist/renderer/test-view.d.ts.map +0 -1
- package/dist/renderer/test-view.js.map +0 -1
- package/dist/renderer/trends.d.ts.map +0 -1
- package/dist/renderer/trends.js.map +0 -1
- package/dist/server/job-store.d.ts.map +0 -1
- package/dist/server/job-store.js.map +0 -1
- package/dist/server/report-server.d.ts.map +0 -1
- package/dist/server/report-server.js.map +0 -1
- package/dist/server/report-store.d.ts.map +0 -1
- package/dist/server/report-store.js.map +0 -1
- package/dist/server/skill-index.d.ts.map +0 -1
- package/dist/server/skill-index.js.map +0 -1
- package/dist/server/skill-insights.d.ts.map +0 -1
- package/dist/server/skill-insights.js.map +0 -1
- package/dist/shared/hard-rules.d.ts.map +0 -1
- package/dist/shared/hard-rules.js.map +0 -1
- package/dist/shared/llm-prompts/index.d.ts.map +0 -1
- package/dist/shared/llm-prompts/index.js.map +0 -1
- package/dist/shared/llm-prompts/skill-health.d.ts.map +0 -1
- package/dist/shared/llm-prompts/skill-health.js.map +0 -1
- package/dist/shared/time.d.ts.map +0 -1
- package/dist/shared/time.js.map +0 -1
- package/dist/shared/tool-search.d.ts.map +0 -1
- package/dist/shared/tool-search.js.map +0 -1
- package/dist/types/dependencies.d.ts.map +0 -1
- package/dist/types/dependencies.js.map +0 -1
- package/dist/types/diagnosis.d.ts.map +0 -1
- package/dist/types/diagnosis.js.map +0 -1
- package/dist/types/doctor.d.ts.map +0 -1
- package/dist/types/doctor.js.map +0 -1
- package/dist/types/eval.d.ts.map +0 -1
- package/dist/types/eval.js.map +0 -1
- package/dist/types/executor.d.ts.map +0 -1
- package/dist/types/executor.js.map +0 -1
- package/dist/types/index.d.ts.map +0 -1
- package/dist/types/index.js.map +0 -1
- package/dist/types/judge.d.ts.map +0 -1
- package/dist/types/judge.js.map +0 -1
- package/dist/types/observability.d.ts.map +0 -1
- package/dist/types/observability.js.map +0 -1
- package/dist/types/report.d.ts.map +0 -1
- package/dist/types/report.js.map +0 -1
- package/dist/types/shared.d.ts.map +0 -1
- package/dist/types/shared.js.map +0 -1
- package/dist/types/skill-index.d.ts.map +0 -1
- package/dist/types/skill-index.js.map +0 -1
- package/dist/types/storage.d.ts.map +0 -1
- package/dist/types/storage.js.map +0 -1
- package/dist/util/safe-slice.d.ts.map +0 -1
- package/dist/util/safe-slice.js.map +0 -1
|
@@ -15,9 +15,96 @@ interface WeakSample {
|
|
|
15
15
|
none?: string;
|
|
16
16
|
};
|
|
17
17
|
}
|
|
18
|
-
export declare function extractWeakSamples(report: Report, variantKey: string, count?: number): WeakSample[];
|
|
18
|
+
export declare function extractWeakSamples(report: Report, variantKey: string, count?: number, sampleIdFilter?: Set<string>): WeakSample[];
|
|
19
|
+
/** A train / holdout partition of a sample set. */
|
|
20
|
+
interface HoldoutSplit {
|
|
21
|
+
trainIds: Set<string>;
|
|
22
|
+
holdoutIds: Set<string>;
|
|
23
|
+
}
|
|
24
|
+
/** A train / val / test partition. `val` drives the accept decision; `test` is
|
|
25
|
+
* locked — never seen during the loop, read once at the end for an unbiased
|
|
26
|
+
* generalization score. */
|
|
27
|
+
interface TrainValTestSplit {
|
|
28
|
+
trainIds: Set<string>;
|
|
29
|
+
valIds: Set<string>;
|
|
30
|
+
testIds: Set<string>;
|
|
31
|
+
}
|
|
32
|
+
/** Below this many decision (val) samples the bootstrap diff CI almost never
|
|
33
|
+
* excludes 0 for realistic effect sizes, so the significance gate would reject
|
|
34
|
+
* every candidate. Under that floor evolve degrades to the point-estimate accept
|
|
35
|
+
* and flags `gate.underpowered`. */
|
|
36
|
+
export declare const MIN_GATE_SAMPLES = 8;
|
|
37
|
+
/**
|
|
38
|
+
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
39
|
+
* held out). Holdout members are picked at an even stride so the partition is
|
|
40
|
+
* representative of the ordering, and the split is stable across rounds and runs
|
|
41
|
+
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
42
|
+
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
43
|
+
*/
|
|
44
|
+
export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
|
|
45
|
+
/**
|
|
46
|
+
* Deterministically split sample ids into train / val / test. `val` is carved
|
|
47
|
+
* first at an even stride; `test` is carved at an even stride over what remains,
|
|
48
|
+
* so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
|
|
49
|
+
* null when either ratio ≤ 0 or any of the three sides would drop below
|
|
50
|
+
* MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
|
|
51
|
+
*/
|
|
52
|
+
export declare function splitTrainValTest(sampleIds: string[], valRatio: number, testRatio: number): TrainValTestSplit | null;
|
|
53
|
+
export interface AcceptDecision {
|
|
54
|
+
accepted: boolean;
|
|
55
|
+
/** Diff CI (candidate − best) when the gate ran; absent when it degraded. */
|
|
56
|
+
diffCI?: {
|
|
57
|
+
low: number;
|
|
58
|
+
high: number;
|
|
59
|
+
estimate: number;
|
|
60
|
+
significant: boolean;
|
|
61
|
+
};
|
|
62
|
+
/** True when the gate was requested but the decision set was below MIN_GATE_SAMPLES,
|
|
63
|
+
* so the decision degraded to the point-estimate comparison. */
|
|
64
|
+
underpowered: boolean;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* The accept decision for one round. With the significance gate on and enough
|
|
68
|
+
* decision samples, a candidate is accepted only when it is **significantly** above
|
|
69
|
+
* the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
|
|
70
|
+
* AND its decision score actually beats the recorded best (`pointCand > pointBest`).
|
|
71
|
+
* The second clause preserves evolve's monotonic invariant: `bestScore` never
|
|
72
|
+
* decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
|
|
73
|
+
* a candidate that is significantly above that re-eval — yet still below the recorded
|
|
74
|
+
* best — win and overwrite the best downward. Off, or under-powered, the gate degrades
|
|
75
|
+
* to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
|
|
76
|
+
* the core behavior is unit-testable.
|
|
77
|
+
*/
|
|
78
|
+
export declare function decideAccept(bestScores: number[], candScores: number[], pointBest: number, pointCand: number, opts: {
|
|
79
|
+
significanceGate: boolean;
|
|
80
|
+
alpha: number;
|
|
81
|
+
seed: number;
|
|
82
|
+
}): AcceptDecision;
|
|
83
|
+
/**
|
|
84
|
+
* A view of `report` whose results are restricted to `sampleIds`. Used to keep the
|
|
85
|
+
* holdout split out of the sample-fixer: under an active holdout, only training-split
|
|
86
|
+
* samples may enter the --auto-fix-samples prompt or be rewritten — otherwise the
|
|
87
|
+
* skill's samples get tuned to the very samples that decide acceptance, reintroducing
|
|
88
|
+
* the leak holdout exists to prevent.
|
|
89
|
+
*/
|
|
90
|
+
export declare function restrictReportToSamples(report: Report, sampleIds: Set<string>): Report;
|
|
19
91
|
export declare function allNonTripwireAssertionsPass(report: Report, variantKey: string): boolean;
|
|
20
|
-
|
|
92
|
+
interface EditDelta {
|
|
93
|
+
/** Symmetric line difference (added + removed unique lines) over original line count. */
|
|
94
|
+
ratio: number;
|
|
95
|
+
/** Absolute count of added + removed unique lines. */
|
|
96
|
+
changedLines: number;
|
|
97
|
+
/** Compact `+`/`-` summary of the changed lines, truncated. */
|
|
98
|
+
summary: string;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* How a candidate differs from the current best, by trimmed non-empty line sets.
|
|
102
|
+
* `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
|
|
103
|
+
* improver doesn't re-propose changes that already failed. Order-insensitive and
|
|
104
|
+
* O(n) over small skill files.
|
|
105
|
+
*/
|
|
106
|
+
export declare function computeEditDelta(before: string, after: string, maxSummaryLines?: number): EditDelta;
|
|
107
|
+
export declare function buildImprovementPrompt(skillContent: string, score: number, weakSamples: WeakSample[], rejectedEdits?: string[]): string;
|
|
21
108
|
/** @deprecated Use ProgressCallback from evaluation-core.ts */
|
|
22
109
|
export type EvolveProgressInfo = Parameters<ProgressCallback>[0];
|
|
23
110
|
export interface EvolveRoundProgressInfo {
|
|
@@ -33,6 +120,10 @@ export interface EvolveRoundProgressInfo {
|
|
|
33
120
|
costReported?: boolean;
|
|
34
121
|
error?: string;
|
|
35
122
|
reused?: boolean;
|
|
123
|
+
/** When the significance gate ran: whether the candidate's gain was significant.
|
|
124
|
+
* False on a rejected round means "score rose but within noise" — lets the CLI
|
|
125
|
+
* explain an otherwise-confusing `(+0.0x) ✗ REJECT`. Undefined = gate didn't run. */
|
|
126
|
+
significant?: boolean;
|
|
36
127
|
}
|
|
37
128
|
interface EvolveOptions {
|
|
38
129
|
skillPath: string;
|
|
@@ -61,15 +152,62 @@ interface EvolveOptions {
|
|
|
61
152
|
noDiagnostic?: boolean;
|
|
62
153
|
/** 跳过 doctor 健康检查门禁。默认 false。 */
|
|
63
154
|
skipDoctor?: boolean;
|
|
155
|
+
/** Fraction of samples held out for the accept decision (0..1). Default 0 = off.
|
|
156
|
+
* When > 0, a candidate is accepted on its **holdout** composite rather than the
|
|
157
|
+
* training composite, and weak-sample extraction only sees the training split —
|
|
158
|
+
* so the skill is never tuned to the samples that judge it. Too small a split
|
|
159
|
+
* (either side < MIN_HOLDOUT_SUBSET) falls back to full-set scoring + a warning. */
|
|
160
|
+
holdoutRatio?: number;
|
|
161
|
+
/** Statistically gate acceptance: a candidate is accepted only when its
|
|
162
|
+
* per-sample composite is **significantly** above the current best on the
|
|
163
|
+
* decision (val) set — `bootstrapDiffCI(...).significant && estimate > 0` —
|
|
164
|
+
* not merely numerically higher. Default true (rejecting improvements
|
|
165
|
+
* indistinguishable from judge noise is the point). Below MIN_GATE_SAMPLES
|
|
166
|
+
* decision samples the gate is underpowered and degrades to the point-estimate
|
|
167
|
+
* comparison + a warning. Set false to force the legacy point-estimate accept. */
|
|
168
|
+
significanceGate?: boolean;
|
|
169
|
+
/** Significance level for the accept gate's diff CI. Default 0.05 (95% CI). */
|
|
170
|
+
significanceAlpha?: number;
|
|
171
|
+
/** Fraction of samples locked away as a **test** set (0..1). Default 0 = off.
|
|
172
|
+
* Only honored alongside `holdoutRatio` > 0 (test needs a separate val set to
|
|
173
|
+
* decide on). The test split is never seen during the loop — not by weak-sample
|
|
174
|
+
* extraction, not by the accept gate — and is read exactly once at the end for an
|
|
175
|
+
* unbiased `generalizationScore`. Too small a 3-way split degrades to 2-way. */
|
|
176
|
+
testRatio?: number;
|
|
177
|
+
/** Max fraction of skill lines a single round may change before the candidate is
|
|
178
|
+
* rejected **without paying for evaluation**. Default 0.2 (matches the "≤20%"
|
|
179
|
+
* the improvement prompt already asks for — this enforces it). A small floor
|
|
180
|
+
* always permits a handful of lines so tiny skills aren't frozen. Set 0 to disable. */
|
|
181
|
+
editBudget?: number;
|
|
182
|
+
/** Feed rejected candidate edits back into the next round's improvement prompt
|
|
183
|
+
* ("these were tried and did not help — don't repeat them"). Default true. */
|
|
184
|
+
rejectMemory?: boolean;
|
|
64
185
|
onProgress?: ProgressCallback | null;
|
|
65
186
|
onRoundProgress?: ((progress: EvolveRoundProgressInfo) => void) | null;
|
|
66
187
|
}
|
|
67
188
|
interface TrajectoryEntry {
|
|
68
189
|
round: number;
|
|
190
|
+
/** Accept-decision score: val composite when a holdout split is active, else full-set. */
|
|
69
191
|
score: number;
|
|
70
192
|
delta: number;
|
|
71
193
|
accepted: boolean;
|
|
72
194
|
costUSD: number;
|
|
195
|
+
/** Present when a holdout split is active: the training-split composite (improvement signal). */
|
|
196
|
+
trainScore?: number;
|
|
197
|
+
/** Present when a holdout split is active: the val-split composite (== score). */
|
|
198
|
+
holdoutScore?: number;
|
|
199
|
+
/** Significance-gate diff CI (candidate − current best) on the decision set, when
|
|
200
|
+
* the gate was powered enough to run. `significant` 决定接受。 */
|
|
201
|
+
diffCI?: {
|
|
202
|
+
low: number;
|
|
203
|
+
high: number;
|
|
204
|
+
estimate: number;
|
|
205
|
+
significant: boolean;
|
|
206
|
+
};
|
|
207
|
+
/** Fraction of skill lines this candidate changed vs the current best. */
|
|
208
|
+
editRatio?: number;
|
|
209
|
+
/** True when the candidate was rejected by the edit budget before evaluation. */
|
|
210
|
+
rejectedPreEval?: boolean;
|
|
73
211
|
}
|
|
74
212
|
export interface EvolveResult {
|
|
75
213
|
startScore: number;
|
|
@@ -86,6 +224,34 @@ export interface EvolveResult {
|
|
|
86
224
|
reusedBaselineReportId?: string;
|
|
87
225
|
/** False = 任一轮的 exec / judge 不报 cost → totalCostUSD 是 lower-bound 而非真值。 */
|
|
88
226
|
costReported?: boolean;
|
|
227
|
+
/** Holdout split summary when `--holdout-ratio` > 0. `disabled` is true when the
|
|
228
|
+
* split was too small and evolve fell back to full-set scoring (CLI formats the
|
|
229
|
+
* user-facing message bilingually). */
|
|
230
|
+
holdout?: {
|
|
231
|
+
ratio: number;
|
|
232
|
+
trainCount: number;
|
|
233
|
+
holdoutCount: number;
|
|
234
|
+
disabled?: boolean;
|
|
235
|
+
};
|
|
236
|
+
/** Locked-test split summary. `disabled` is true when `--test-ratio` was requested
|
|
237
|
+
* but the 3-way split was too small, so evolve fell back to a 2-way holdout and
|
|
238
|
+
* produced no generalization score. */
|
|
239
|
+
test?: {
|
|
240
|
+
ratio: number;
|
|
241
|
+
count: number;
|
|
242
|
+
disabled?: boolean;
|
|
243
|
+
};
|
|
244
|
+
/** Unbiased composite of the best skill on the locked test set — the headline honest
|
|
245
|
+
* number. Present only when a 3-way split was active. The test set never influenced
|
|
246
|
+
* selection or weak-sample extraction, so this is an out-of-sample estimate. */
|
|
247
|
+
generalizationScore?: number;
|
|
248
|
+
/** Accept-gate summary. `underpowered` = the decision set was below MIN_GATE_SAMPLES
|
|
249
|
+
* at least once, so the gate degraded to the point-estimate comparison + warned. */
|
|
250
|
+
gate?: {
|
|
251
|
+
enabled: boolean;
|
|
252
|
+
alpha: number;
|
|
253
|
+
underpowered?: boolean;
|
|
254
|
+
};
|
|
89
255
|
trajectory: TrajectoryEntry[];
|
|
90
256
|
bestSkillPath: string;
|
|
91
257
|
allVersions: string[];
|
|
@@ -97,6 +263,5 @@ export interface RoundReport {
|
|
|
97
263
|
report: Report;
|
|
98
264
|
}
|
|
99
265
|
export declare function mergeEvolveReports(roundReports: RoundReport[], skillName: string, totalCostUSD: number, samples?: Sample[], skillPath?: string): Report;
|
|
100
|
-
export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
|
|
266
|
+
export declare function evolveSkill({ skillPath, samplesPath, rounds, target, stopOnAssertionsPass, autoFixSamples, sampleFixMaxAttempts, reuseLatestEval, model, judgeModels, improveModel, improveMode, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, holdoutRatio, significanceGate, significanceAlpha, testRatio, editBudget, rejectMemory, onProgress, onRoundProgress, }: EvolveOptions): Promise<EvolveResult>;
|
|
101
267
|
export {};
|
|
102
|
-
//# sourceMappingURL=evolver.d.ts.map
|
|
@@ -6,6 +6,8 @@ import { persistReport, DEFAULT_OUTPUT_DIR, generateRunId, hashString } from '..
|
|
|
6
6
|
import { createFileStore } from '../server/report-store.js';
|
|
7
7
|
import { analyzeResults } from '../analysis/report-diagnostics.js';
|
|
8
8
|
import { loadSamples } from '../inputs/load-samples.js';
|
|
9
|
+
import { buildVariantSummary } from '../eval-core/schema.js';
|
|
10
|
+
import { bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES } from '../eval-core/bootstrap.js';
|
|
9
11
|
import { fixSamples } from './sample-fixer.js';
|
|
10
12
|
const IMPROVE_SYSTEM_PROMPT = `你是一个 AI 提示词改进专家。你的任务是分析评测结果中的薄弱环节,针对性地改进 skill(系统提示词),使其在评测中获得更高的分数。
|
|
11
13
|
|
|
@@ -121,9 +123,11 @@ function readSkillName(skillPath) {
|
|
|
121
123
|
return null;
|
|
122
124
|
}
|
|
123
125
|
}
|
|
124
|
-
export function extractWeakSamples(report, variantKey, count = 5) {
|
|
126
|
+
export function extractWeakSamples(report, variantKey, count = 5, sampleIdFilter) {
|
|
125
127
|
const weakSamples = [];
|
|
126
128
|
for (const r of report.results) {
|
|
129
|
+
if (sampleIdFilter && !sampleIdFilter.has(r.sample_id))
|
|
130
|
+
continue;
|
|
127
131
|
const v = r.variants[variantKey];
|
|
128
132
|
if (!v || typeof v.compositeScore !== 'number')
|
|
129
133
|
continue;
|
|
@@ -153,6 +157,134 @@ export function extractWeakSamples(report, variantKey, count = 5) {
|
|
|
153
157
|
.sort((a, b) => a.compositeScore - b.compositeScore)
|
|
154
158
|
.slice(0, count);
|
|
155
159
|
}
|
|
160
|
+
/** Below this many samples on any side, a split is too small to be meaningful —
|
|
161
|
+
* evolve falls back to full-set scoring and warns. */
|
|
162
|
+
const MIN_HOLDOUT_SUBSET = 3;
|
|
163
|
+
/** Below this many decision (val) samples the bootstrap diff CI almost never
|
|
164
|
+
* excludes 0 for realistic effect sizes, so the significance gate would reject
|
|
165
|
+
* every candidate. Under that floor evolve degrades to the point-estimate accept
|
|
166
|
+
* and flags `gate.underpowered`. */
|
|
167
|
+
export const MIN_GATE_SAMPLES = 8;
|
|
168
|
+
/** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
|
|
169
|
+
* picked subset is representative of the ordering and stable across rounds/runs. */
|
|
170
|
+
function pickByStride(ids, count) {
|
|
171
|
+
const picked = new Set();
|
|
172
|
+
if (count <= 0)
|
|
173
|
+
return picked;
|
|
174
|
+
const stride = ids.length / count;
|
|
175
|
+
for (let k = 0; k < count; k++)
|
|
176
|
+
picked.add(ids[Math.floor(k * stride)]);
|
|
177
|
+
return picked;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
181
|
+
* held out). Holdout members are picked at an even stride so the partition is
|
|
182
|
+
* representative of the ordering, and the split is stable across rounds and runs
|
|
183
|
+
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
184
|
+
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
185
|
+
*/
|
|
186
|
+
export function splitHoldout(sampleIds, ratio) {
|
|
187
|
+
if (!(ratio > 0) || sampleIds.length === 0)
|
|
188
|
+
return null;
|
|
189
|
+
const holdoutCount = Math.round(sampleIds.length * ratio);
|
|
190
|
+
const trainCount = sampleIds.length - holdoutCount;
|
|
191
|
+
if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
|
|
192
|
+
return null;
|
|
193
|
+
const holdoutIds = pickByStride(sampleIds, holdoutCount);
|
|
194
|
+
const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
|
|
195
|
+
return { trainIds, holdoutIds };
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* Deterministically split sample ids into train / val / test. `val` is carved
|
|
199
|
+
* first at an even stride; `test` is carved at an even stride over what remains,
|
|
200
|
+
* so the three sets are disjoint and stable across rounds/runs (no RNG). Returns
|
|
201
|
+
* null when either ratio ≤ 0 or any of the three sides would drop below
|
|
202
|
+
* MIN_HOLDOUT_SUBSET — the caller then degrades to a 2-way (or full-set) split.
|
|
203
|
+
*/
|
|
204
|
+
export function splitTrainValTest(sampleIds, valRatio, testRatio) {
|
|
205
|
+
if (!(valRatio > 0) || !(testRatio > 0) || sampleIds.length === 0)
|
|
206
|
+
return null;
|
|
207
|
+
const valCount = Math.round(sampleIds.length * valRatio);
|
|
208
|
+
const testCount = Math.round(sampleIds.length * testRatio);
|
|
209
|
+
const trainCount = sampleIds.length - valCount - testCount;
|
|
210
|
+
if (valCount < MIN_HOLDOUT_SUBSET || testCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
|
|
211
|
+
return null;
|
|
212
|
+
const valIds = pickByStride(sampleIds, valCount);
|
|
213
|
+
const remaining = sampleIds.filter((id) => !valIds.has(id));
|
|
214
|
+
const testIds = pickByStride(remaining, testCount);
|
|
215
|
+
const trainIds = new Set(sampleIds.filter((id) => !valIds.has(id) && !testIds.has(id)));
|
|
216
|
+
return { trainIds, valIds, testIds };
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* Mean composite over the subset of a report's results whose sample_id is in
|
|
220
|
+
* `ids`, using the same aggregation as the full-run summary
|
|
221
|
+
* (`buildVariantSummary`) so train / holdout scores stay comparable to the
|
|
222
|
+
* headline composite. Returns 0 when the subset has no scorable entries.
|
|
223
|
+
*/
|
|
224
|
+
function subsetCompositeScore(report, variantKey, ids) {
|
|
225
|
+
const entries = [];
|
|
226
|
+
for (const r of report.results) {
|
|
227
|
+
if (!ids.has(r.sample_id))
|
|
228
|
+
continue;
|
|
229
|
+
const v = r.variants[variantKey];
|
|
230
|
+
if (v)
|
|
231
|
+
entries.push(v);
|
|
232
|
+
}
|
|
233
|
+
if (entries.length === 0)
|
|
234
|
+
return 0;
|
|
235
|
+
return buildVariantSummary(entries).avgCompositeScore ?? 0;
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* Per-sample composite scores over the subset of a report's results whose
|
|
239
|
+
* sample_id is in `ids`, in result order. Feeds `bootstrapDiffCI` for the
|
|
240
|
+
* significance accept gate — the array (not the mean) is what the bootstrap
|
|
241
|
+
* resamples. Entries without a numeric compositeScore are skipped.
|
|
242
|
+
*/
|
|
243
|
+
function perSampleComposite(report, variantKey, ids) {
|
|
244
|
+
const scores = [];
|
|
245
|
+
for (const r of report.results) {
|
|
246
|
+
if (!ids.has(r.sample_id))
|
|
247
|
+
continue;
|
|
248
|
+
const v = r.variants[variantKey];
|
|
249
|
+
if (v && typeof v.compositeScore === 'number')
|
|
250
|
+
scores.push(v.compositeScore);
|
|
251
|
+
}
|
|
252
|
+
return scores;
|
|
253
|
+
}
|
|
254
|
+
/**
|
|
255
|
+
* The accept decision for one round. With the significance gate on and enough
|
|
256
|
+
* decision samples, a candidate is accepted only when it is **significantly** above
|
|
257
|
+
* the current best's fresh re-eval (`bootstrapDiffCI(...).significant && estimate > 0`)
|
|
258
|
+
* AND its decision score actually beats the recorded best (`pointCand > pointBest`).
|
|
259
|
+
* The second clause preserves evolve's monotonic invariant: `bestScore` never
|
|
260
|
+
* decreases. Without it, an unlucky (noise-low) re-eval of the current best could let
|
|
261
|
+
* a candidate that is significantly above that re-eval — yet still below the recorded
|
|
262
|
+
* best — win and overwrite the best downward. Off, or under-powered, the gate degrades
|
|
263
|
+
* to the legacy point-estimate comparison alone. Pure (modulo the seeded bootstrap) so
|
|
264
|
+
* the core behavior is unit-testable.
|
|
265
|
+
*/
|
|
266
|
+
export function decideAccept(bestScores, candScores, pointBest, pointCand, opts) {
|
|
267
|
+
const powered = bestScores.length >= MIN_GATE_SAMPLES && candScores.length >= MIN_GATE_SAMPLES;
|
|
268
|
+
if (opts.significanceGate && powered) {
|
|
269
|
+
const diff = bootstrapDiffCI(bestScores, candScores, opts.alpha, DEFAULT_BOOTSTRAP_SAMPLES, opts.seed);
|
|
270
|
+
return {
|
|
271
|
+
accepted: diff.significant && diff.estimate > 0 && pointCand > pointBest,
|
|
272
|
+
diffCI: { low: diff.low, high: diff.high, estimate: diff.estimate, significant: diff.significant },
|
|
273
|
+
underpowered: false,
|
|
274
|
+
};
|
|
275
|
+
}
|
|
276
|
+
return { accepted: pointCand > pointBest, underpowered: opts.significanceGate && !powered };
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* A view of `report` whose results are restricted to `sampleIds`. Used to keep the
|
|
280
|
+
* holdout split out of the sample-fixer: under an active holdout, only training-split
|
|
281
|
+
* samples may enter the --auto-fix-samples prompt or be rewritten — otherwise the
|
|
282
|
+
* skill's samples get tuned to the very samples that decide acceptance, reintroducing
|
|
283
|
+
* the leak holdout exists to prevent.
|
|
284
|
+
*/
|
|
285
|
+
export function restrictReportToSamples(report, sampleIds) {
|
|
286
|
+
return { ...report, results: report.results.filter((r) => sampleIds.has(r.sample_id)) };
|
|
287
|
+
}
|
|
156
288
|
export function allNonTripwireAssertionsPass(report, variantKey) {
|
|
157
289
|
for (const entry of report.results) {
|
|
158
290
|
const variant = entry.variants[variantKey];
|
|
@@ -279,7 +411,36 @@ async function autoFixSamplesAfterSkillRound(opts) {
|
|
|
279
411
|
}
|
|
280
412
|
return { fixedCount: result.fixedCount, costUSD: result.costUSD };
|
|
281
413
|
}
|
|
282
|
-
|
|
414
|
+
/** Number of skill lines below which the edit budget never trips — so a tiny skill
|
|
415
|
+
* isn't frozen by a percentage threshold that a few lines already blow past. */
|
|
416
|
+
const EDIT_BUDGET_FLOOR_LINES = 10;
|
|
417
|
+
/**
|
|
418
|
+
* How a candidate differs from the current best, by trimmed non-empty line sets.
|
|
419
|
+
* `ratio` drives the edit budget; `summary` feeds the rejected-edit memory so the
|
|
420
|
+
* improver doesn't re-propose changes that already failed. Order-insensitive and
|
|
421
|
+
* O(n) over small skill files.
|
|
422
|
+
*/
|
|
423
|
+
export function computeEditDelta(before, after, maxSummaryLines = 12) {
|
|
424
|
+
const beforeArr = before.split('\n').map((l) => l.trim()).filter(Boolean);
|
|
425
|
+
const afterArr = after.split('\n').map((l) => l.trim()).filter(Boolean);
|
|
426
|
+
const beforeSet = new Set(beforeArr);
|
|
427
|
+
const afterSet = new Set(afterArr);
|
|
428
|
+
const added = [...afterSet].filter((l) => !beforeSet.has(l));
|
|
429
|
+
const removed = [...beforeSet].filter((l) => !afterSet.has(l));
|
|
430
|
+
const changedLines = added.length + removed.length;
|
|
431
|
+
const ratio = changedLines / Math.max(beforeArr.length, 1);
|
|
432
|
+
const parts = [];
|
|
433
|
+
for (const l of added.slice(0, maxSummaryLines))
|
|
434
|
+
parts.push(`+ ${l}`);
|
|
435
|
+
if (added.length > maxSummaryLines)
|
|
436
|
+
parts.push(`+ …(其余 +${added.length - maxSummaryLines} 行)`);
|
|
437
|
+
for (const l of removed.slice(0, maxSummaryLines))
|
|
438
|
+
parts.push(`- ${l}`);
|
|
439
|
+
if (removed.length > maxSummaryLines)
|
|
440
|
+
parts.push(`- …(其余 -${removed.length - maxSummaryLines} 行)`);
|
|
441
|
+
return { ratio, changedLines, summary: parts.join('\n') || '(无文本差异)' };
|
|
442
|
+
}
|
|
443
|
+
export function buildImprovementPrompt(skillContent, score, weakSamples, rejectedEdits) {
|
|
283
444
|
const weakDetails = weakSamples.map((s) => {
|
|
284
445
|
const parts = [`### ${s.sample_id}(${s.compositeScore}/5.0)`];
|
|
285
446
|
if (s.llmReason)
|
|
@@ -308,13 +469,16 @@ export function buildImprovementPrompt(skillContent, score, weakSamples) {
|
|
|
308
469
|
}
|
|
309
470
|
return parts.join('\n');
|
|
310
471
|
}).join('\n\n');
|
|
472
|
+
const rejectedSection = rejectedEdits && rejectedEdits.length > 0
|
|
473
|
+
? `\n\n## 已试过且未带来显著提升的改法(不要重复)\n\n${rejectedEdits.join('\n\n')}`
|
|
474
|
+
: '';
|
|
311
475
|
return `## 当前 Skill(平均分: ${score.toFixed(2)}/5.0)
|
|
312
476
|
|
|
313
477
|
${skillContent}
|
|
314
478
|
|
|
315
479
|
## 低分用例分析
|
|
316
480
|
|
|
317
|
-
${weakDetails || '(无低分用例)'}`;
|
|
481
|
+
${weakDetails || '(无低分用例)'}${rejectedSection}`;
|
|
318
482
|
}
|
|
319
483
|
function buildImprovementSuffix(mode, candidatePath) {
|
|
320
484
|
if (mode === 'agent') {
|
|
@@ -377,7 +541,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
377
541
|
}
|
|
378
542
|
const runId = `evolve-${skillName}-${generateRunId([skillName]).split('-').slice(-2).join('-')}`;
|
|
379
543
|
const report = {
|
|
380
|
-
|
|
544
|
+
reportKind: 'evaluation',
|
|
381
545
|
id: runId,
|
|
382
546
|
meta: {
|
|
383
547
|
...firstReport.meta,
|
|
@@ -401,7 +565,7 @@ export function mergeEvolveReports(roundReports, skillName, totalCostUSD, sample
|
|
|
401
565
|
report.analysis = analyzeResults(report, { samples });
|
|
402
566
|
return report;
|
|
403
567
|
}
|
|
404
|
-
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, onProgress = null, onRoundProgress = null, }) {
|
|
568
|
+
export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target = null, stopOnAssertionsPass = false, autoFixSamples = false, sampleFixMaxAttempts = 2, reuseLatestEval = false, model = DEFAULT_MODEL, judgeModels, improveModel = DEFAULT_MODEL, improveMode = 'agent', executorName = 'claude', concurrency = 1, timeoutMs, skipConnectivity = false, effort, noDiagnostic, skipDoctor, holdoutRatio = 0, significanceGate = true, significanceAlpha = DEFAULT_BOOTSTRAP_ALPHA, testRatio = 0, editBudget = 0.2, rejectMemory = true, onProgress = null, onRoundProgress = null, }) {
|
|
405
569
|
if (judgeModels && judgeModels.length > 1) {
|
|
406
570
|
throw new Error('evolveSkill does not support multi-judge ensemble (received '
|
|
407
571
|
+ `${judgeModels.length} judges). Pass a single-judge array, e.g. `
|
|
@@ -421,6 +585,72 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
421
585
|
if (!existsSync(absSamplesPath))
|
|
422
586
|
throw new Error(`samples file not found: ${absSamplesPath}`);
|
|
423
587
|
mkdirSync(evolveDir, { recursive: true });
|
|
588
|
+
// Split (opt-in). Computed once over the canonical sample order so it's stable
|
|
589
|
+
// across rounds. `val` drives the accept decision; weak-sample extraction and the
|
|
590
|
+
// sample-fixer only ever see `train`; `test` is locked away — never seen during the
|
|
591
|
+
// loop — and read once at the end for an unbiased generalization score. With no
|
|
592
|
+
// holdout the decision runs on the full set (legacy). A too-small 3-way split
|
|
593
|
+
// degrades to 2-way, then to full-set.
|
|
594
|
+
const allSampleIds = loadSamples(absSamplesPath).samples.map((s) => s.sample_id);
|
|
595
|
+
const threeWay = (testRatio > 0 && holdoutRatio > 0) ? splitTrainValTest(allSampleIds, holdoutRatio, testRatio) : null;
|
|
596
|
+
const twoWay = (!threeWay && holdoutRatio > 0) ? splitHoldout(allSampleIds, holdoutRatio) : null;
|
|
597
|
+
const split = threeWay
|
|
598
|
+
? { trainIds: threeWay.trainIds, valIds: threeWay.valIds, testIds: threeWay.testIds }
|
|
599
|
+
: twoWay
|
|
600
|
+
? { trainIds: twoWay.trainIds, valIds: twoWay.holdoutIds, testIds: null }
|
|
601
|
+
: null;
|
|
602
|
+
const holdoutInfo = holdoutRatio > 0
|
|
603
|
+
? {
|
|
604
|
+
ratio: holdoutRatio,
|
|
605
|
+
trainCount: split?.trainIds.size ?? allSampleIds.length,
|
|
606
|
+
holdoutCount: split?.valIds.size ?? 0,
|
|
607
|
+
...(split ? {} : { disabled: true }),
|
|
608
|
+
}
|
|
609
|
+
: undefined;
|
|
610
|
+
// test 被请求(配了 --holdout-ratio)但 3-way 太小回退 → 标 disabled,别让用户
|
|
611
|
+
// 以为拿到了 locked-test 泛化分。
|
|
612
|
+
const testRequested = testRatio > 0 && holdoutRatio > 0;
|
|
613
|
+
const testInfo = threeWay
|
|
614
|
+
? { ratio: testRatio, count: threeWay.testIds.size }
|
|
615
|
+
: testRequested ? { ratio: testRatio, count: 0, disabled: true } : undefined;
|
|
616
|
+
// Deterministic seed so the gate's CIs are reproducible across reruns (and
|
|
617
|
+
// assertable in tests). Derived from skill identity + sample count, parsed to a uint32.
|
|
618
|
+
const gateSeed = parseInt(hashString(`${skillName}:${allSampleIds.length}`).slice(0, 8), 16) >>> 0;
|
|
619
|
+
let gateUnderpowered = false;
|
|
620
|
+
// Accept-decision score for a report's variant: val composite when a split is
|
|
621
|
+
// active, otherwise the full-set composite (legacy behavior).
|
|
622
|
+
const decisionScore = (report, key) => split
|
|
623
|
+
? subsetCompositeScore(report, key, split.valIds)
|
|
624
|
+
: (report.summary[key]?.avgCompositeScore ?? 0);
|
|
625
|
+
const trainScoreOf = (report, key) => split ? subsetCompositeScore(report, key, split.trainIds) : undefined;
|
|
626
|
+
// Per-round trajectory tail: train / holdout breakdown, only when split active.
|
|
627
|
+
const splitScores = (report, key, decision) => split ? { trainScore: trainScoreOf(report, key), holdoutScore: decision } : {};
|
|
628
|
+
// Unbiased generalization: the best skill's composite on the locked test set,
|
|
629
|
+
// read once at the very end. Present only under a valid 3-way split; the test
|
|
630
|
+
// set never influenced selection or weak-sample extraction.
|
|
631
|
+
const buildGeneralization = () => {
|
|
632
|
+
if (!testInfo)
|
|
633
|
+
return {};
|
|
634
|
+
if (!threeWay || !split?.testIds)
|
|
635
|
+
return { test: testInfo }; // requested but degraded → disabled, no score
|
|
636
|
+
const best = roundReports.find((r) => r.round === bestRound)?.report;
|
|
637
|
+
if (!best)
|
|
638
|
+
return { test: testInfo };
|
|
639
|
+
const key = Object.keys(best.summary)[0];
|
|
640
|
+
return { test: testInfo, generalizationScore: Number(subsetCompositeScore(best, key, split.testIds).toFixed(4)) };
|
|
641
|
+
};
|
|
642
|
+
const gateInfo = () => ({ enabled: significanceGate, alpha: significanceAlpha, ...(gateUnderpowered ? { underpowered: true } : {}) });
|
|
643
|
+
// Rejected-edit memory (most recent K). Fed back into the next round's improvement
|
|
644
|
+
// prompt so the improver doesn't re-propose changes that already failed to help.
|
|
645
|
+
const rejectedEdits = [];
|
|
646
|
+
const REJECT_MEMORY_K = 3;
|
|
647
|
+
const rememberRejected = (round, summary, reason) => {
|
|
648
|
+
if (!rejectMemory)
|
|
649
|
+
return;
|
|
650
|
+
rejectedEdits.push(`【第 ${round} 轮被拒(${reason})】\n${summary}`);
|
|
651
|
+
if (rejectedEdits.length > REJECT_MEMORY_K)
|
|
652
|
+
rejectedEdits.shift();
|
|
653
|
+
};
|
|
424
654
|
// Save original as r0
|
|
425
655
|
let currentBest = readFileSync(absSkillPath, 'utf-8').trim();
|
|
426
656
|
const r0Path = join(evolveDir, `${skillName}.r0.md`);
|
|
@@ -461,13 +691,13 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
461
691
|
});
|
|
462
692
|
}
|
|
463
693
|
const baselineVariantKey = Object.keys(baselineReport.summary)[0];
|
|
464
|
-
bestScore = baselineReport
|
|
694
|
+
bestScore = decisionScore(baselineReport, baselineVariantKey);
|
|
465
695
|
const baselineCost = baselineReused ? 0 : baselineReport.meta.totalCostUSD;
|
|
466
696
|
totalCostUSD += baselineCost;
|
|
467
697
|
const baselineCostReported = baselineReused || !reportHasUnreportedCost(baselineReport);
|
|
468
698
|
if (!baselineCostReported)
|
|
469
699
|
totalCostReported = false;
|
|
470
|
-
trajectory.push({ round: 0, score: bestScore, delta: 0, accepted: true, costUSD: baselineCost });
|
|
700
|
+
trajectory.push({ round: 0, score: bestScore, delta: 0, accepted: true, costUSD: baselineCost, ...splitScores(baselineReport, baselineVariantKey, bestScore) });
|
|
471
701
|
roundReports.push({ round: 0, accepted: true, report: baselineReport });
|
|
472
702
|
if (onRoundProgress)
|
|
473
703
|
onRoundProgress({ round: 0, totalRounds: rounds, phase: 'baseline', score: bestScore, costUSD: baselineCost, costReported: baselineCostReported, reused: baselineReused });
|
|
@@ -487,6 +717,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
487
717
|
...(sampleFixes.length > 0 && { sampleFixes }),
|
|
488
718
|
...(reusedBaselineReportId && { reusedBaselineReportId }),
|
|
489
719
|
...(totalCostReported ? {} : { costReported: false }),
|
|
720
|
+
...(holdoutInfo ? { holdout: holdoutInfo } : {}),
|
|
721
|
+
...buildGeneralization(),
|
|
722
|
+
gate: gateInfo(),
|
|
490
723
|
trajectory,
|
|
491
724
|
bestSkillPath: allVersions[bestRound],
|
|
492
725
|
allVersions,
|
|
@@ -513,10 +746,10 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
513
746
|
totalCostReported = false;
|
|
514
747
|
}
|
|
515
748
|
const lastVariantKey = Object.keys(lastReport.summary)[0];
|
|
516
|
-
const weakSamples = extractWeakSamples(lastReport, lastVariantKey);
|
|
749
|
+
const weakSamples = extractWeakSamples(lastReport, lastVariantKey, 5, split?.trainIds);
|
|
517
750
|
// Generate improvement
|
|
518
751
|
const candidatePath = join(evolveDir, `${skillName}.r${round}.md`);
|
|
519
|
-
const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples);
|
|
752
|
+
const basePrompt = buildImprovementPrompt(currentBest, bestScore, weakSamples, rejectMemory ? rejectedEdits : undefined);
|
|
520
753
|
const executor = createExecutor(executorName);
|
|
521
754
|
let candidateContent;
|
|
522
755
|
let improveCostUSD;
|
|
@@ -570,12 +803,32 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
570
803
|
if (!improveCostReported)
|
|
571
804
|
totalCostReported = false;
|
|
572
805
|
allVersions.push(candidatePath);
|
|
806
|
+
// Edit budget: reject oversized rewrites BEFORE paying for evaluation. The
|
|
807
|
+
// improvement prompt already asks for ≤ editBudget of lines changed — this
|
|
808
|
+
// enforces it. A small floor still lets tiny skills change a handful of lines.
|
|
809
|
+
const editDelta = computeEditDelta(currentBest, candidateContent);
|
|
810
|
+
if (editBudget > 0 && editDelta.ratio > editBudget && editDelta.changedLines > EDIT_BUDGET_FLOOR_LINES) {
|
|
811
|
+
totalCostUSD += improveCostUSD;
|
|
812
|
+
consecutiveRejects++;
|
|
813
|
+
const reason = `改动过大 ${(editDelta.ratio * 100).toFixed(0)}%(预算 ${(editBudget * 100).toFixed(0)}%),评测前判拒`;
|
|
814
|
+
rememberRejected(round, editDelta.summary, reason);
|
|
815
|
+
trajectory.push({ round, score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, editRatio: Number(editDelta.ratio.toFixed(4)), rejectedPreEval: true });
|
|
816
|
+
if (onRoundProgress)
|
|
817
|
+
onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: bestScore, delta: 0, accepted: false, costUSD: improveCostUSD, costReported: improveCostReported });
|
|
818
|
+
if (consecutiveRejects >= 2) {
|
|
819
|
+
stopReason = 'consecutive-rejects';
|
|
820
|
+
break;
|
|
821
|
+
}
|
|
822
|
+
continue;
|
|
823
|
+
}
|
|
573
824
|
let preEvalSampleFixCost = 0;
|
|
574
825
|
if (autoFixSamples) {
|
|
575
826
|
const sampleFix = await autoFixSamplesAfterSkillRound({
|
|
576
827
|
samplesPath: absSamplesPath,
|
|
577
828
|
skillContent: candidateContent,
|
|
578
|
-
|
|
829
|
+
// Under an active holdout, the sample-fixer may only see training-split samples —
|
|
830
|
+
// never the holdout samples that drive the accept decision (leak guard).
|
|
831
|
+
report: split ? restrictReportToSamples(lastReport, split.trainIds) : lastReport,
|
|
579
832
|
treatmentKey: lastVariantKey,
|
|
580
833
|
executorName,
|
|
581
834
|
model: improveModel,
|
|
@@ -593,13 +846,28 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
593
846
|
samplesPath: absSamplesPath, skillDir, model, judgeModels: effectiveJudgeModels, executorName, concurrency, timeoutMs, skipConnectivity, effort, noDiagnostic, skipDoctor, onProgress,
|
|
594
847
|
});
|
|
595
848
|
const candidateVariantKey = Object.keys(candidateReport.summary)[0];
|
|
596
|
-
const candidateScore = candidateReport
|
|
849
|
+
const candidateScore = decisionScore(candidateReport, candidateVariantKey);
|
|
597
850
|
const roundCost = improveCostUSD + preEvalSampleFixCost + candidateReport.meta.totalCostUSD;
|
|
598
851
|
const roundCostReported = improveCostReported && !reportHasUnreportedCost(candidateReport);
|
|
599
852
|
if (!roundCostReported)
|
|
600
853
|
totalCostReported = false;
|
|
601
854
|
totalCostUSD += improveCostUSD + candidateReport.meta.totalCostUSD;
|
|
602
|
-
|
|
855
|
+
// Significance accept gate: accept only when the candidate is *significantly*
|
|
856
|
+
// above the current best on the decision (val) set, not merely numerically higher
|
|
857
|
+
// — rejecting gains indistinguishable from judge noise. `lastReport` is the current
|
|
858
|
+
// best's fresh eval and `candidateReport` the candidate's, over the same samples;
|
|
859
|
+
// bootstrapDiffCI resamples the two arrays independently (conservative — not a paired
|
|
860
|
+
// bootstrap). Under-powered decision sets degrade to the legacy point-estimate accept
|
|
861
|
+
// (note: that path compares the prior-round best scalar, not this fresh re-eval) and
|
|
862
|
+
// flag `gate.underpowered`.
|
|
863
|
+
const valIds = split ? split.valIds : new Set(allSampleIds);
|
|
864
|
+
const bestScores = perSampleComposite(lastReport, lastVariantKey, valIds);
|
|
865
|
+
const candScores = perSampleComposite(candidateReport, candidateVariantKey, valIds);
|
|
866
|
+
const decision = decideAccept(bestScores, candScores, bestScore, candidateScore, { significanceGate, alpha: significanceAlpha, seed: gateSeed });
|
|
867
|
+
const accepted = decision.accepted;
|
|
868
|
+
const diffCI = decision.diffCI;
|
|
869
|
+
if (decision.underpowered)
|
|
870
|
+
gateUnderpowered = true;
|
|
603
871
|
if (accepted) {
|
|
604
872
|
currentBest = candidateContent;
|
|
605
873
|
bestScore = candidateScore;
|
|
@@ -608,13 +876,15 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
608
876
|
}
|
|
609
877
|
else {
|
|
610
878
|
consecutiveRejects++;
|
|
879
|
+
const reason = diffCI ? (diffCI.estimate > 0 ? '提升不显著' : '方向为负') : '未超过当前最优';
|
|
880
|
+
rememberRejected(round, editDelta.summary, reason);
|
|
611
881
|
}
|
|
612
882
|
if (accepted)
|
|
613
883
|
roundReports.push({ round, accepted, report: candidateReport });
|
|
614
884
|
const roundDelta = candidateScore - trajectory[trajectory.length - 1].score;
|
|
615
|
-
trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost });
|
|
885
|
+
trajectory.push({ round, score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, ...splitScores(candidateReport, candidateVariantKey, candidateScore), ...(diffCI ? { diffCI } : {}), editRatio: Number(editDelta.ratio.toFixed(4)) });
|
|
616
886
|
if (onRoundProgress)
|
|
617
|
-
onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported });
|
|
887
|
+
onRoundProgress({ round, totalRounds: rounds, phase: 'done', score: candidateScore, delta: roundDelta, accepted, costUSD: roundCost, costReported: roundCostReported, ...(diffCI ? { significant: diffCI.significant } : {}) });
|
|
618
888
|
// Early stop
|
|
619
889
|
if (stopOnAssertionsPass && accepted && allNonTripwireAssertionsPass(candidateReport, candidateVariantKey)) {
|
|
620
890
|
stopReason = 'assertions-pass';
|
|
@@ -652,6 +922,9 @@ export async function evolveSkill({ skillPath, samplesPath, rounds = 5, target =
|
|
|
652
922
|
...(sampleFixes.length > 0 && { sampleFixes }),
|
|
653
923
|
...(reusedBaselineReportId && { reusedBaselineReportId }),
|
|
654
924
|
...(totalCostReported ? {} : { costReported: false }),
|
|
925
|
+
...(holdoutInfo ? { holdout: holdoutInfo } : {}),
|
|
926
|
+
...buildGeneralization(),
|
|
927
|
+
gate: gateInfo(),
|
|
655
928
|
trajectory,
|
|
656
929
|
bestSkillPath: allVersions[bestRound],
|
|
657
930
|
allVersions,
|
|
@@ -678,4 +951,3 @@ async function evaluate(skillFilePath, { samplesPath, skillDir, model, judgeMode
|
|
|
678
951
|
});
|
|
679
952
|
return report;
|
|
680
953
|
}
|
|
681
|
-
//# sourceMappingURL=evolver.js.map
|