oh-my-knowledge 0.47.0 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -18
- package/README.zh.md +55 -23
- package/dist/analysis/coverage-analyzer.d.ts +1 -0
- package/dist/analysis/coverage-analyzer.js +125 -62
- package/dist/analysis/failure-clusterer.js +2 -1
- package/dist/analysis/gap-analyzer.d.ts +2 -2
- package/dist/analysis/gap-analyzer.js +13 -3
- package/dist/analysis/hedging-classifier.d.ts +2 -2
- package/dist/analysis/hedging-classifier.js +3 -4
- package/dist/analysis/report-diagnostics.js +9 -7
- package/dist/analysis/sample-diagnostics.js +6 -6
- package/dist/artifact-graph/doctor.js +15 -7
- package/dist/assets/agent-skills/omk/SKILL.md +27 -7
- package/dist/assets/agent-skills/omk/references/commands.md +19 -17
- package/dist/authoring/evolver.d.ts +10 -6
- package/dist/authoring/evolver.js +496 -83
- package/dist/authoring/generator.d.ts +3 -3
- package/dist/authoring/generator.js +5 -10
- package/dist/authoring/sample-fixer.d.ts +8 -6
- package/dist/authoring/sample-fixer.js +76 -5
- package/dist/cli/commands/doctor.js +31 -14
- package/dist/cli/commands/eval/index.d.ts +3 -0
- package/dist/cli/commands/eval/index.js +163 -17
- package/dist/cli/commands/evolve.d.ts +4 -4
- package/dist/cli/commands/evolve.js +27 -13
- package/dist/cli/commands/init.js +16 -3
- package/dist/cli/commands/observe/inbox.js +44 -21
- package/dist/cli/commands/observe/index.js +20 -11
- package/dist/cli/commands/observe/ingest.d.ts +3 -0
- package/dist/cli/commands/observe/ingest.js +35 -4
- package/dist/cli/commands/sample.d.ts +9 -3
- package/dist/cli/commands/sample.js +91 -74
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/codex-model-hint.d.ts +9 -0
- package/dist/cli/lib/codex-model-hint.js +45 -0
- package/dist/cli/lib/generation-failure-hint.d.ts +2 -0
- package/dist/cli/lib/generation-failure-hint.js +61 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +4 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +38 -6
- package/dist/cli/lib/i18n-dict/help.js +6 -6
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +13 -9
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +34 -2
- package/dist/cli/lib/llm-failure-classifier.d.ts +2 -0
- package/dist/cli/lib/llm-failure-classifier.js +8 -0
- package/dist/cli/lib/parse-run-config.d.ts +6 -5
- package/dist/cli/lib/parse-run-config.js +16 -9
- package/dist/cli/lib/runtime-defaults.d.ts +21 -0
- package/dist/cli/lib/runtime-defaults.js +79 -0
- package/dist/diagnosis/observe-mapper.js +14 -15
- package/dist/diagnosis/observe-producer.js +3 -1
- package/dist/diagnosis/studio-projection.js +14 -7
- package/dist/diagnosis/types.d.ts +2 -0
- package/dist/diagnosis/types.js +12 -0
- package/dist/doctor/endpoint-rule.js +2 -1
- package/dist/eval-core/artifact-file-names.js +18 -1
- package/dist/eval-core/artifact-index.d.ts +7 -11
- package/dist/eval-core/artifact-index.js +139 -80
- package/dist/eval-core/cache.d.ts +12 -3
- package/dist/eval-core/cache.js +89 -29
- package/dist/eval-core/comparability.js +10 -6
- package/dist/eval-core/evaluation-execution.d.ts +2 -1
- package/dist/eval-core/evaluation-execution.js +122 -37
- package/dist/eval-core/evaluation-job.d.ts +4 -1
- package/dist/eval-core/evaluation-job.js +4 -1
- package/dist/eval-core/evaluation-reporting.d.ts +15 -13
- package/dist/eval-core/evaluation-reporting.js +54 -52
- package/dist/eval-core/execution-strategy.d.ts +2 -0
- package/dist/eval-core/execution-strategy.js +11 -9
- package/dist/eval-core/fact-checker.js +15 -7
- package/dist/eval-core/holdout.js +3 -2
- package/dist/eval-core/judge-independence.d.ts +2 -2
- package/dist/eval-core/mock-hook.cjs +23 -6
- package/dist/eval-core/mocks-runtime.js +30 -8
- package/dist/eval-core/report-document.d.ts +12 -0
- package/dist/eval-core/report-document.js +1151 -0
- package/dist/eval-core/report-extensions.d.ts +4 -0
- package/dist/eval-core/report-extensions.js +500 -0
- package/dist/eval-core/report-file-migration.js +7 -2
- package/dist/eval-core/resume-compatibility.d.ts +31 -0
- package/dist/eval-core/resume-compatibility.js +141 -0
- package/dist/eval-core/sample-fingerprint.d.ts +12 -0
- package/dist/eval-core/sample-fingerprint.js +193 -0
- package/dist/eval-core/schema.js +86 -31
- package/dist/eval-core/verdict.d.ts +8 -4
- package/dist/eval-core/verdict.js +24 -10
- package/dist/eval-workflows/batch-evaluation-workflow.d.ts +2 -1
- package/dist/eval-workflows/batch-evaluation-workflow.js +25 -12
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.d.ts +10 -5
- package/dist/eval-workflows/evaluation-pipeline/preflight-warnings.js +58 -21
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +3 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +4 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +4 -1
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.d.ts +6 -5
- package/dist/eval-workflows/evaluation-pipeline/test-set-hash.js +17 -10
- package/dist/eval-workflows/evaluation-pipeline.js +12 -7
- package/dist/eval-workflows/run-evaluation.d.ts +9 -7
- package/dist/eval-workflows/run-evaluation.js +79 -51
- package/dist/executors/anthropic-api.js +65 -9
- package/dist/executors/claude-cli.js +16 -79
- package/dist/executors/claude-protocol.d.ts +28 -0
- package/dist/executors/claude-protocol.js +180 -0
- package/dist/executors/claude-sdk-trace.js +56 -28
- package/dist/executors/claude-sdk.d.ts +1 -0
- package/dist/executors/claude-sdk.js +39 -93
- package/dist/executors/codex-cli-trace.js +166 -31
- package/dist/executors/codex-cli.d.ts +6 -8
- package/dist/executors/codex-cli.js +49 -151
- package/dist/executors/codex-protocol.d.ts +24 -0
- package/dist/executors/codex-protocol.js +234 -0
- package/dist/executors/codex-sdk.js +68 -120
- package/dist/executors/gemini.js +88 -13
- package/dist/executors/index.d.ts +2 -3
- package/dist/executors/index.js +5 -3
- package/dist/executors/openai-api.js +70 -9
- package/dist/executors/runtime-fingerprint.js +88 -11
- package/dist/executors/script-command.d.ts +8 -0
- package/dist/executors/script-command.js +87 -0
- package/dist/executors/script.js +202 -29
- package/dist/executors/shared.d.ts +35 -3
- package/dist/executors/shared.js +113 -15
- package/dist/grading/assertions.d.ts +1 -1
- package/dist/grading/assertions.js +19 -9
- package/dist/grading/diagnostic.d.ts +9 -2
- package/dist/grading/diagnostic.js +25 -2
- package/dist/grading/index.js +10 -4
- package/dist/grading/judge.js +19 -6
- package/dist/grading/layered-scores.d.ts +2 -3
- package/dist/grading/layered-scores.js +2 -3
- package/dist/inputs/load-samples.d.ts +1 -2
- package/dist/inputs/load-samples.js +23 -1
- package/dist/inputs/mcp-resolver.js +6 -3
- package/dist/inputs/sample-document.d.ts +11 -0
- package/dist/inputs/sample-document.js +96 -0
- package/dist/managed/evidence.d.ts +1 -0
- package/dist/managed/evidence.js +1 -1
- package/dist/managed/store.js +200 -91
- package/dist/observability/codex-trace-adapter.d.ts +5 -0
- package/dist/observability/codex-trace-adapter.js +850 -0
- package/dist/observability/experience.d.ts +32 -6
- package/dist/observability/experience.js +2695 -459
- package/dist/observability/feedback-matchers.js +16 -1
- package/dist/observability/inbox-view-model.d.ts +2 -1
- package/dist/observability/inbox-view-model.js +20 -14
- package/dist/observability/inbox.d.ts +7 -1
- package/dist/observability/inbox.js +632 -124
- package/dist/observability/problem-patterns.js +2 -0
- package/dist/observability/review-state.d.ts +6 -0
- package/dist/observability/review-state.js +235 -63
- package/dist/observability/skill-chain-advisories.js +1 -1
- package/dist/observability/skill-chain.js +17 -4
- package/dist/observability/skill-health-analyzer.d.ts +32 -7
- package/dist/observability/skill-health-analyzer.js +194 -121
- package/dist/observability/skill-health-report.d.ts +10 -0
- package/dist/observability/skill-health-report.js +620 -0
- package/dist/observability/soft-standards/constants.d.ts +0 -1
- package/dist/observability/soft-standards/constants.js +0 -1
- package/dist/observability/soft-standards/index.d.ts +1 -1
- package/dist/observability/soft-standards/index.js +1 -1
- package/dist/observability/soft-standards/llm-extractor.js +8 -10
- package/dist/observability/soft-standards/skill-standards-store.d.ts +2 -1
- package/dist/observability/soft-standards/skill-standards-store.js +59 -18
- package/dist/observability/soft-standards/types.d.ts +2 -2
- package/dist/observability/trace-adapter.d.ts +12 -7
- package/dist/observability/trace-adapter.js +11 -9
- package/dist/observability/trace-attribution.d.ts +13 -5
- package/dist/observability/trace-attribution.js +315 -21
- package/dist/observability/trace-ingestion.d.ts +9 -0
- package/dist/observability/trace-ingestion.js +80 -0
- package/dist/observability/trace-ir.d.ts +113 -0
- package/dist/observability/trace-ir.js +87 -0
- package/dist/observability/trace-segmenter.d.ts +19 -6
- package/dist/observability/trace-segmenter.js +377 -196
- package/dist/observability/trace-session-index.d.ts +19 -0
- package/dist/observability/trace-session-index.js +68 -0
- package/dist/observability/trace-source.d.ts +12 -4
- package/dist/observability/trace-source.js +939 -215
- package/dist/renderer/html-renderer.js +37 -6
- package/dist/renderer/icons.js +3 -0
- package/dist/renderer/observation-inbox-renderer.js +227 -91
- package/dist/renderer/skill-detail-renderer.js +452 -109
- package/dist/renderer/skill-health-renderer.js +69 -12
- package/dist/renderer/summary.js +28 -7
- package/dist/renderer/table.js +21 -4
- package/dist/renderer/test-view.d.ts +1 -0
- package/dist/renderer/test-view.js +44 -9
- package/dist/server/indexed-report-store.js +14 -18
- package/dist/server/job-store.js +64 -26
- package/dist/server/report-server.js +190 -78
- package/dist/server/report-store.js +57 -80
- package/dist/server/skill-index.js +143 -49
- package/dist/server/skill-insights.js +44 -5
- package/dist/shared/artifact-graph.d.ts +3 -0
- package/dist/shared/artifact-graph.js +224 -0
- package/dist/shared/assertion-types.d.ts +8 -0
- package/dist/shared/assertion-types.js +46 -0
- package/dist/shared/atomic-json.d.ts +8 -0
- package/dist/shared/atomic-json.js +33 -0
- package/dist/shared/diagnosis-schema.d.ts +9 -0
- package/dist/shared/diagnosis-schema.js +181 -0
- package/dist/shared/doctor-report.d.ts +3 -0
- package/dist/shared/doctor-report.js +103 -0
- package/dist/shared/evaluation-job.d.ts +6 -0
- package/dist/shared/evaluation-job.js +217 -0
- package/dist/shared/executor-result.d.ts +17 -0
- package/dist/shared/executor-result.js +221 -0
- package/dist/shared/file-lock.d.ts +12 -0
- package/dist/shared/file-lock.js +129 -0
- package/dist/shared/json-value.d.ts +5 -0
- package/dist/shared/json-value.js +36 -0
- package/dist/shared/keyed-mutex.d.ts +7 -0
- package/dist/shared/keyed-mutex.js +24 -0
- package/dist/shared/record-count.d.ts +8 -0
- package/dist/shared/record-count.js +43 -0
- package/dist/shared/sample-contract.d.ts +3 -0
- package/dist/shared/sample-contract.js +332 -0
- package/dist/shared/shell-quote.d.ts +2 -0
- package/dist/shared/shell-quote.js +7 -0
- package/dist/shared/timestamp.d.ts +6 -0
- package/dist/shared/timestamp.js +64 -0
- package/dist/shared/token-usage.d.ts +19 -0
- package/dist/shared/token-usage.js +50 -0
- package/dist/shared/tool-call-status.d.ts +8 -0
- package/dist/shared/tool-call-status.js +28 -0
- package/dist/shared/tool-identity.d.ts +21 -0
- package/dist/shared/tool-identity.js +84 -0
- package/dist/shared/tool-search.js +73 -16
- package/dist/shared/trace-projection.d.ts +5 -0
- package/dist/shared/trace-projection.js +20 -0
- package/dist/shared/trace-source-kind.d.ts +3 -0
- package/dist/shared/trace-source-kind.js +12 -0
- package/dist/types/diagnosis.d.ts +2 -0
- package/dist/types/eval.d.ts +4 -0
- package/dist/types/executor.d.ts +32 -5
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.js +1 -0
- package/dist/types/judge.d.ts +2 -0
- package/dist/types/observability.d.ts +116 -9
- package/dist/types/report.d.ts +58 -6
- package/dist/types/skill-index.d.ts +7 -0
- package/dist/types/trace.d.ts +2 -0
- package/dist/types/trace.js +1 -0
- package/package.json +9 -5
|
@@ -11,19 +11,23 @@ export const initDict = {
|
|
|
11
11
|
// 跑出第一份报告是冷启动最该先发生的事;「换成你自己的」放到跑通之后。这也消除了
|
|
12
12
|
// 主 README「不用改任何文件」与旧 init「先编辑」的矛盾。
|
|
13
13
|
'cli.init.next_step_run': {
|
|
14
|
-
zh: ' 1. 直接跑通(无需先改任何文件):
|
|
15
|
-
en: ' 1. Run it as-is (no edits needed):
|
|
14
|
+
zh: ' 1. 直接跑通(无需先改任何文件):{command}',
|
|
15
|
+
en: ' 1. Run it as-is (no edits needed): {command}',
|
|
16
|
+
},
|
|
17
|
+
'cli.init.next_step_report': {
|
|
18
|
+
zh: ' 2. 看报告里的 verdict 和“下一步”:PROGRESS 才能发布;UNDERPOWERED / NOISE 先扩样到约 20 条以上后重跑。',
|
|
19
|
+
en: ' 2. Read the report verdict and Next line: PROGRESS can ship; UNDERPOWERED / NOISE means grow to roughly 20+ samples and re-run.',
|
|
16
20
|
},
|
|
17
21
|
'cli.init.next_step_executor': {
|
|
18
|
-
zh: '
|
|
19
|
-
en: ' The
|
|
22
|
+
zh: ' executor / judge 会按运行环境选择;Codex 任务自动使用本机 Codex 配置。也可用 OMK_EXECUTOR / OMK_MODEL 固定环境偏好,详见 https://oh-my-knowledge.pages.dev/zh/reference/executors。',
|
|
23
|
+
en: ' The executor / judge follow the runtime environment; Codex tasks use the local Codex configuration automatically. OMK_EXECUTOR / OMK_MODEL pin environment preferences. See https://oh-my-knowledge.pages.dev/reference/executors.',
|
|
20
24
|
},
|
|
21
25
|
'cli.init.next_step_customize': {
|
|
22
|
-
zh: '
|
|
23
|
-
en: '
|
|
26
|
+
zh: ' 3. 跑通后,替换为你自己的 skill 和 eval-samples.json;还没有用例时先运行 omk sample <skill-path>。',
|
|
27
|
+
en: ' 3. Once it runs, replace the starter skills and eval-samples.json with your own; if you have no samples yet, run omk sample <skill-path>.',
|
|
24
28
|
},
|
|
25
|
-
'cli.init.
|
|
26
|
-
zh: '
|
|
27
|
-
en: '
|
|
29
|
+
'cli.init.note_skill_injection': {
|
|
30
|
+
zh: ' 注:omk eval 会把 SKILL.md 作为 system prompt 注入;模板 frontmatter 只是方便同一目录复用为 agent skill。',
|
|
31
|
+
en: ' Note: omk eval injects SKILL.md as the system prompt; the starter frontmatter only helps reuse the same directory as an agent skill.',
|
|
28
32
|
},
|
|
29
33
|
};
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
import type { CliMessage } from './types.js';
|
|
2
|
-
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_verdict_next_step' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.skip_connectivity_warning';
|
|
2
|
+
export type RunMessageKey = 'cli.progress.preflight_starting' | 'cli.progress.sample_retry' | 'cli.progress.sample_error' | 'cli.progress.sample_executing' | 'cli.progress.sample_exec_done' | 'cli.progress.output_preview' | 'cli.progress.judging' | 'cli.progress.judged' | 'cli.progress.skipped' | 'cli.progress.sample_done' | 'cli.progress.sample_failed_done' | 'cli.run.invalid_repeat' | 'cli.run.invalid_holdout_ratio' | 'cli.run.invalid_judge_repeat' | 'cli.run.no_debias_length_active' | 'cli.run.invalid_bootstrap_samples' | 'cli.run.bootstrap_samples_too_large' | 'cli.run.dry_run_no_scores' | 'cli.run.skill_section' | 'cli.run.run_section' | 'cli.run.batch_complete' | 'cli.run.batch_verdict_header' | 'cli.run.batch_verdict_next_step' | 'cli.run.batch_child_report_missing' | 'cli.run.eval_complete' | 'cli.run.tally' | 'cli.run.report_saved' | 'cli.run.evidence_recorded' | 'cli.run.evidence_recorded_promotable' | 'cli.run.evidence_recorded_unbound' | 'cli.run.report_only_gate_skipped' | 'cli.run.report_server_running' | 'cli.run.report_server_view' | 'cli.run.report_server_stop' | 'cli.run.no_serve_in_non_tty' | 'cli.run.no_serve_view_hint' | 'cli.run.gold_load_failed' | 'cli.run.gold_load_issue' | 'cli.run.contamination_warning' | 'cli.run.codex_fallback_hint' | 'cli.run.codex_auth_hint' | 'cli.run.codex_model_hint' | 'cli.run.openai_api_auth_hint' | 'cli.run.openai_api_model_hint' | 'cli.run.anthropic_api_auth_hint' | 'cli.run.anthropic_api_model_hint' | 'cli.run.skip_connectivity_warning';
|
|
3
3
|
export declare const runDict: Record<RunMessageKey, CliMessage>;
|
|
@@ -68,8 +68,8 @@ export const runDict = {
|
|
|
68
68
|
en: '⚠ --bootstrap-samples {n} is large and may take several seconds. 1000 is the industry standard and usually sufficient.\n',
|
|
69
69
|
},
|
|
70
70
|
'cli.run.dry_run_no_scores': {
|
|
71
|
-
zh: 'eval dry-run
|
|
72
|
-
en: 'Eval dry-run: no scores to
|
|
71
|
+
zh: 'eval dry-run:仅预览任务,不检查分数。下一步:确认任务无误后,去掉 --dry-run 运行正式评测。',
|
|
72
|
+
en: 'Eval dry-run: no scores checked. Next: remove --dry-run to run the eval.',
|
|
73
73
|
},
|
|
74
74
|
'cli.run.skill_section': {
|
|
75
75
|
zh: '\n=== [{i}/{n}] Skill: {skill} ===\n',
|
|
@@ -111,6 +111,10 @@ export const runDict = {
|
|
|
111
111
|
zh: '🔖 已为受管 skill「{name}」记录评测证据 → measurable\n',
|
|
112
112
|
en: '🔖 Recorded eval evidence for managed skill "{name}" → measurable\n',
|
|
113
113
|
},
|
|
114
|
+
'cli.run.evidence_recorded_promotable': {
|
|
115
|
+
zh: '🔖 已为受管 skill「{name}」记录评测证据 → measurable。运行 {command} 接受当前版本。\n',
|
|
116
|
+
en: '🔖 Recorded eval evidence for managed skill "{name}" → measurable. Run {command} to accept this version.\n',
|
|
117
|
+
},
|
|
114
118
|
'cli.run.evidence_recorded_unbound': {
|
|
115
119
|
zh: '🔖 受管 skill「{name}」:评测内容与当前安装版本指纹不一致,证据已留存但不绑当前版本\n',
|
|
116
120
|
en: '🔖 Managed skill "{name}": eval content differs from the installed version; evidence kept but not bound to current\n',
|
|
@@ -151,6 +155,34 @@ export const runDict = {
|
|
|
151
155
|
zh: '\n⚠ {warning}\n',
|
|
152
156
|
en: '\n⚠ {warning}\n',
|
|
153
157
|
},
|
|
158
|
+
'cli.run.codex_fallback_hint': {
|
|
159
|
+
zh: '\n提示:当前失败的是 Claude 系列执行器。先确认 Claude Code 已登录;如果你在 Codex 环境里,也可以把模型运行参数改为:{flags}。{codexModelHint}codex 执行器目前不会报告 costUSD。',
|
|
160
|
+
en: '\nHint: the failing runtime is Claude-based. First confirm Claude Code is authenticated; in a Codex environment, you can also switch the model runtime flags to: {flags}. {codexModelHint} The codex executor does not report costUSD yet.',
|
|
161
|
+
},
|
|
162
|
+
'cli.run.codex_auth_hint': {
|
|
163
|
+
zh: '\n提示:当前失败的是 Codex 系列执行器。先确认 Codex CLI / SDK 已安装并完成登录;如果你有 Claude Code 可用,可以改走 Claude:{claudeFlags};如果要继续走 OpenAI API,可以改为:{openaiFlags},并设置 OPENAI_API_KEY。openai-api 会按 API 响应记录 token / cost。',
|
|
164
|
+
en: '\nHint: the failing runtime is Codex-based. First confirm the Codex CLI / SDK is installed and authenticated; if Claude Code is available, switch to Claude: {claudeFlags}; to stay on the OpenAI API path, switch to: {openaiFlags}, and set OPENAI_API_KEY. openai-api records token / cost from API responses.',
|
|
165
|
+
},
|
|
166
|
+
'cli.run.codex_model_hint': {
|
|
167
|
+
zh: '\n提示:当前失败的是 Codex 系列执行器,但模型名看起来不可用。可以先按本机 Codex 配置重试:{codexFlags}({codexModelHint});也可以先运行 `{codexExec}` 验证模型是否可用。若只是想先跑通,可以改走 Claude:{claudeFlags};或继续走 OpenAI API:{openaiFlags},并设置 OPENAI_API_KEY。',
|
|
168
|
+
en: '\nHint: the failing runtime is Codex-based, but the model name appears unavailable. Retry with the local Codex config model: {codexFlags} ({codexModelHint}); you can also run `{codexExec}` to verify the model. To just get a first run through, switch to Claude: {claudeFlags}; or stay on the OpenAI API path: {openaiFlags}, and set OPENAI_API_KEY.',
|
|
169
|
+
},
|
|
170
|
+
'cli.run.openai_api_auth_hint': {
|
|
171
|
+
zh: '\n提示:当前失败的是 OpenAI API 执行器。请检查 OPENAI_API_KEY / OPENAI_BASE_URL 是否可用,并确认模型名对当前端点可用。',
|
|
172
|
+
en: '\nHint: the failing runtime is the OpenAI API executor. Check OPENAI_API_KEY / OPENAI_BASE_URL and confirm the model is available on that endpoint.',
|
|
173
|
+
},
|
|
174
|
+
'cli.run.openai_api_model_hint': {
|
|
175
|
+
zh: '\n提示:当前失败的是 OpenAI API 执行器,但模型名看起来对当前端点不可用。请检查 --model / --judge-models、OPENAI_BASE_URL 与账号权限是否匹配。',
|
|
176
|
+
en: '\nHint: the failing runtime is the OpenAI API executor, but the model name appears unavailable on the current endpoint. Check --model / --judge-models, OPENAI_BASE_URL, and account access.',
|
|
177
|
+
},
|
|
178
|
+
'cli.run.anthropic_api_auth_hint': {
|
|
179
|
+
zh: '\n提示:当前失败的是 Anthropic API 执行器。请检查 ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL 是否可用,并确认模型名对当前端点可用。',
|
|
180
|
+
en: '\nHint: the failing runtime is the Anthropic API executor. Check ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL and confirm the model is available on that endpoint.',
|
|
181
|
+
},
|
|
182
|
+
'cli.run.anthropic_api_model_hint': {
|
|
183
|
+
zh: '\n提示:当前失败的是 Anthropic API 执行器,但模型名看起来对当前端点不可用。请检查 --model / --judge-models、ANTHROPIC_BASE_URL 与账号权限是否匹配。',
|
|
184
|
+
en: '\nHint: the failing runtime is the Anthropic API executor, but the model name appears unavailable on the current endpoint. Check --model / --judge-models, ANTHROPIC_BASE_URL, and account access.',
|
|
185
|
+
},
|
|
154
186
|
'cli.run.skip_connectivity_warning': {
|
|
155
187
|
zh: '⚠️ --skip-connectivity 已启用: 跳过 LLM 模型连通性检测。请确保 executor / judge 已通过其他方式验证可达。',
|
|
156
188
|
en: '⚠️ --skip-connectivity enabled: LLM connectivity check skipped. Verify executor / judge are reachable by other means.',
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
const SETUP_FAILURE_RE = /auth|login|credential|API[_ ]?KEY|BASE_URL|API error|ENOENT|not found|ECONN|ENOTFOUND|ETIMEDOUT|timeout|timed out|401|403|404|invalid_request_error/i;
|
|
2
|
+
const MODEL_UNAVAILABLE_RE = /model .*not supported|model is not supported|unsupported model|invalid_request_error.*model|model.*invalid_request_error|model_not_found|model .*not found|model .*does not exist|no such model|not have access to (the )?model|model .*not available|model .*is not available/i;
|
|
3
|
+
export function looksLikeModelUnavailableFailure(message) {
|
|
4
|
+
return MODEL_UNAVAILABLE_RE.test(message);
|
|
5
|
+
}
|
|
6
|
+
export function looksLikeLlmSetupFailure(message) {
|
|
7
|
+
return SETUP_FAILURE_RE.test(message) || looksLikeModelUnavailableFailure(message);
|
|
8
|
+
}
|
|
@@ -18,20 +18,21 @@
|
|
|
18
18
|
* 反而失去「精度链一目了然」的可读性。
|
|
19
19
|
*/
|
|
20
20
|
import type { EvalConfig, VariantSpec, JudgeConfig, EvalBudget, ProgressCallback } from '../../types/index.js';
|
|
21
|
+
import { type RuntimeResolutionOptions } from './runtime-defaults.js';
|
|
21
22
|
export { parseJudgeModelsArg, parseJudgeModelsArgOrExit } from './parse-run-config/judge-models.js';
|
|
22
23
|
export interface RunConfig {
|
|
23
24
|
samplesPath: string;
|
|
24
25
|
skillDir: string;
|
|
25
26
|
variantSpecs: VariantSpec[];
|
|
26
|
-
model: string
|
|
27
|
+
model: string;
|
|
27
28
|
outputDir: string;
|
|
28
29
|
noJudge: boolean | undefined;
|
|
29
30
|
noCache: boolean | undefined;
|
|
30
31
|
dryRun: boolean | undefined;
|
|
31
32
|
concurrency: number;
|
|
32
33
|
timeoutMs: number;
|
|
33
|
-
executorName: string
|
|
34
|
-
/** 跳过 LLM
|
|
34
|
+
executorName: string;
|
|
35
|
+
/** 跳过 LLM 模型连通性检测。仅当 --resume 报告通过完整契约校验时自动 true。 */
|
|
35
36
|
skipConnectivity: boolean | undefined;
|
|
36
37
|
/** 跳过 doctor 健康检查门禁(--skip-doctor)。escape hatch — 默认 false。
|
|
37
38
|
* 开启后 doctor 整段不跑(节省静态检查时间);doctor 失败也不再阻断 eval。
|
|
@@ -51,7 +52,7 @@ export interface RunConfig {
|
|
|
51
52
|
/** --judge-repeat N. Calls LLM judge N times per (sample × dimension). Default 1. */
|
|
52
53
|
judgeRepeat?: number;
|
|
53
54
|
/** Unified judge config. Always non-empty; 1 entry = single judge, ≥ 2 = ensemble.
|
|
54
|
-
*
|
|
55
|
+
* Defaults to the selected runtime's judge model. */
|
|
55
56
|
judgeModels: JudgeConfig[];
|
|
56
57
|
/** --bootstrap. Adds bootstrap CI to summary (per-variant mean + pairwise diff). */
|
|
57
58
|
bootstrap?: boolean;
|
|
@@ -90,4 +91,4 @@ export interface ParseRunConfigResult {
|
|
|
90
91
|
* 上游对未知 flag 拦截 exit 2,这里不再 parseArgs。eval-runner 等业务 caller
|
|
91
92
|
* 把 oclif flags 当 values 喂进来。
|
|
92
93
|
*/
|
|
93
|
-
export declare function parseRunConfig(values: Record<string, unknown
|
|
94
|
+
export declare function parseRunConfig(values: Record<string, unknown>, runtimeOptions?: RuntimeResolutionOptions): ParseRunConfigResult;
|
|
@@ -20,17 +20,18 @@
|
|
|
20
20
|
import { resolve } from 'node:path';
|
|
21
21
|
import { projectReportsDir, globalReportsDir } from '../../eval-core/measurement-dirs.js';
|
|
22
22
|
import { loadEvalConfig } from '../../inputs/eval-config.js';
|
|
23
|
-
import {
|
|
23
|
+
import { setOwnRecordValue } from '../../shared/record-count.js';
|
|
24
24
|
import { parseJudgeModelsArgOrExit } from './parse-run-config/judge-models.js';
|
|
25
25
|
import { discoverSamplesPath } from './parse-run-config/samples-discovery.js';
|
|
26
26
|
import { resolveVariantSpecs } from './parse-run-config/variant-resolution.js';
|
|
27
|
+
import { envJudgeModels, resolveRuntimeSelection, } from './runtime-defaults.js';
|
|
27
28
|
export { parseJudgeModelsArg, parseJudgeModelsArgOrExit } from './parse-run-config/judge-models.js';
|
|
28
29
|
/**
|
|
29
30
|
* 接 typed flags(来自 oclif Command.parse() 输出)。oclif strict 模式已经在
|
|
30
31
|
* 上游对未知 flag 拦截 exit 2,这里不再 parseArgs。eval-runner 等业务 caller
|
|
31
32
|
* 把 oclif flags 当 values 喂进来。
|
|
32
33
|
*/
|
|
33
|
-
export function parseRunConfig(values) {
|
|
34
|
+
export function parseRunConfig(values, runtimeOptions = {}) {
|
|
34
35
|
if (values.variants !== undefined) {
|
|
35
36
|
throw new Error(`--variants 已在 v0.16 废除,请改用 --control <expr> 与 --treatment <v1,v2,...>\n`
|
|
36
37
|
+ ` 迁移示例:--variants baseline,my-skill → --control baseline --treatment my-skill\n`
|
|
@@ -59,18 +60,24 @@ export function parseRunConfig(values) {
|
|
|
59
60
|
// 3) Resolve variantSpecs: CLI > config > batch > error。dedup 在 helper 里。
|
|
60
61
|
const variantSpecs = resolveVariantSpecs(values, evalConfig, skillDir);
|
|
61
62
|
// 4) Apply CLI > config > hard-coded default for all other fields.
|
|
62
|
-
const
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
63
|
+
const runtime = resolveRuntimeSelection({
|
|
64
|
+
executor: values.executor ?? evalConfig?.executor,
|
|
65
|
+
model: values.model ?? evalConfig?.model,
|
|
66
|
+
}, runtimeOptions);
|
|
67
|
+
const executorName = runtime.executor;
|
|
68
|
+
const model = runtime.model;
|
|
66
69
|
// judgeModels: unified judge config. Parse --judge-models (CLI) or evalConfig.judgeModels (yaml).
|
|
67
70
|
// 1 entry = single judge, ≥ 2 entries = ensemble. Format `executor:model[,executor:model]`.
|
|
68
|
-
// 出口 RunConfig.judgeModels
|
|
71
|
+
// 出口 RunConfig.judgeModels 保证非空。Codex 默认沿用任务模型,避免把 Claude 的
|
|
72
|
+
// `haiku` alias 误传给 Codex。
|
|
69
73
|
const cliJudgesRaw = values['judge-models'];
|
|
70
74
|
const parsedJudges = cliJudgesRaw !== undefined ? parseJudgeModelsArgOrExit(cliJudgesRaw) : undefined;
|
|
75
|
+
const envJudgesRaw = envJudgeModels(runtimeOptions.env);
|
|
76
|
+
const parsedEnvJudges = envJudgesRaw !== undefined ? parseJudgeModelsArgOrExit(envJudgesRaw) : undefined;
|
|
71
77
|
const judgeModels = parsedJudges
|
|
72
78
|
?? evalConfig?.judgeModels
|
|
73
|
-
??
|
|
79
|
+
?? parsedEnvJudges
|
|
80
|
+
?? [{ executor: executorName, model: runtime.judgeModel }];
|
|
74
81
|
// 报告默认落项目 `.omk/reports`(绑用例集,construct validity);--global 写全局;--output-dir 最高优先。
|
|
75
82
|
// 同 observe / doctor 写入侧口径。读取侧(studio / resume / gold-compare / 复用)走 overlay 项目→全局兜底。
|
|
76
83
|
const outputDir = resolve(values['output-dir']
|
|
@@ -110,7 +117,7 @@ export function parseRunConfig(values) {
|
|
|
110
117
|
if (evalConfig?.variants) {
|
|
111
118
|
for (const v of evalConfig.variants) {
|
|
112
119
|
if (v.allowedSkills !== undefined) {
|
|
113
|
-
variantAllowedSkills
|
|
120
|
+
setOwnRecordValue(variantAllowedSkills, v.name, v.allowedSkills);
|
|
114
121
|
}
|
|
115
122
|
}
|
|
116
123
|
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
export interface RuntimeResolutionOptions {
|
|
2
|
+
env?: NodeJS.ProcessEnv;
|
|
3
|
+
commandExists?: (command: string, env: NodeJS.ProcessEnv) => boolean;
|
|
4
|
+
lang?: 'zh' | 'en';
|
|
5
|
+
}
|
|
6
|
+
export interface RuntimeSelection {
|
|
7
|
+
executor: string;
|
|
8
|
+
model: string;
|
|
9
|
+
judgeModel: string;
|
|
10
|
+
}
|
|
11
|
+
export declare function commandExistsOnPath(command: string, env?: NodeJS.ProcessEnv): boolean;
|
|
12
|
+
export declare function isCodexHost(env?: NodeJS.ProcessEnv): boolean;
|
|
13
|
+
export declare function isCodexExecutor(executor: string): boolean;
|
|
14
|
+
export declare function resolveCliExecutor(explicitExecutor?: string, options?: RuntimeResolutionOptions): string;
|
|
15
|
+
export declare function resolveCliModel(executor: string, explicitModel?: string, options?: RuntimeResolutionOptions): string;
|
|
16
|
+
export declare function defaultJudgeModel(executor: string, taskModel: string): string;
|
|
17
|
+
export declare function resolveRuntimeSelection(input: {
|
|
18
|
+
executor?: string;
|
|
19
|
+
model?: string;
|
|
20
|
+
}, options?: RuntimeResolutionOptions): RuntimeSelection;
|
|
21
|
+
export declare function envJudgeModels(env?: NodeJS.ProcessEnv): string | undefined;
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import { accessSync, constants } from 'node:fs';
|
|
2
|
+
import { delimiter, join } from 'node:path';
|
|
3
|
+
import { getCodexModelSuggestion } from './codex-model-hint.js';
|
|
4
|
+
import { DEFAULT_MODEL, JUDGE_MODEL } from '../../executors/shared.js';
|
|
5
|
+
function nonEmpty(value) {
|
|
6
|
+
const trimmed = value?.trim();
|
|
7
|
+
return trimmed ? trimmed : undefined;
|
|
8
|
+
}
|
|
9
|
+
export function commandExistsOnPath(command, env = process.env) {
|
|
10
|
+
const pathEntries = (env.PATH ?? '').split(delimiter).filter(Boolean);
|
|
11
|
+
const extensions = process.platform === 'win32'
|
|
12
|
+
? (env.PATHEXT ?? '.EXE;.CMD;.BAT;.COM').split(';')
|
|
13
|
+
: [''];
|
|
14
|
+
return pathEntries.some((entry) => extensions.some((extension) => {
|
|
15
|
+
try {
|
|
16
|
+
accessSync(join(entry, `${command}${extension}`), constants.X_OK);
|
|
17
|
+
return true;
|
|
18
|
+
}
|
|
19
|
+
catch {
|
|
20
|
+
return false;
|
|
21
|
+
}
|
|
22
|
+
}));
|
|
23
|
+
}
|
|
24
|
+
export function isCodexHost(env = process.env) {
|
|
25
|
+
return Boolean(nonEmpty(env.CODEX_THREAD_ID)
|
|
26
|
+
|| nonEmpty(env.CODEX_INTERNAL_ORIGINATOR_OVERRIDE)
|
|
27
|
+
|| nonEmpty(env.CODEX_CI));
|
|
28
|
+
}
|
|
29
|
+
export function isCodexExecutor(executor) {
|
|
30
|
+
return executor === 'codex' || executor === 'codex-sdk';
|
|
31
|
+
}
|
|
32
|
+
export function resolveCliExecutor(explicitExecutor, options = {}) {
|
|
33
|
+
const env = options.env ?? process.env;
|
|
34
|
+
const explicit = nonEmpty(explicitExecutor);
|
|
35
|
+
if (explicit)
|
|
36
|
+
return explicit;
|
|
37
|
+
const envExecutor = nonEmpty(env.OMK_EXECUTOR);
|
|
38
|
+
if (envExecutor)
|
|
39
|
+
return envExecutor;
|
|
40
|
+
if (isCodexHost(env))
|
|
41
|
+
return 'codex';
|
|
42
|
+
const commandExists = options.commandExists ?? commandExistsOnPath;
|
|
43
|
+
if (commandExists('codex', env) && !commandExists('claude', env))
|
|
44
|
+
return 'codex';
|
|
45
|
+
return 'claude';
|
|
46
|
+
}
|
|
47
|
+
export function resolveCliModel(executor, explicitModel, options = {}) {
|
|
48
|
+
const env = options.env ?? process.env;
|
|
49
|
+
const explicit = nonEmpty(explicitModel);
|
|
50
|
+
if (explicit)
|
|
51
|
+
return explicit;
|
|
52
|
+
const envModel = nonEmpty(env.OMK_MODEL);
|
|
53
|
+
if (envModel)
|
|
54
|
+
return envModel;
|
|
55
|
+
if (!isCodexExecutor(executor))
|
|
56
|
+
return DEFAULT_MODEL;
|
|
57
|
+
const suggestion = getCodexModelSuggestion(env);
|
|
58
|
+
if (suggestion.fromConfig)
|
|
59
|
+
return suggestion.model;
|
|
60
|
+
const lang = options.lang ?? 'zh';
|
|
61
|
+
throw new Error(lang === 'zh'
|
|
62
|
+
? `Codex 执行器需要明确模型。请用 --model <model>、设置 OMK_MODEL,或在 ${suggestion.configPath} 配置顶层 model。`
|
|
63
|
+
: `The Codex executor needs an explicit model. Pass --model <model>, set OMK_MODEL, or configure a top-level model in ${suggestion.configPath}.`);
|
|
64
|
+
}
|
|
65
|
+
export function defaultJudgeModel(executor, taskModel) {
|
|
66
|
+
return isCodexExecutor(executor) ? taskModel : JUDGE_MODEL;
|
|
67
|
+
}
|
|
68
|
+
export function resolveRuntimeSelection(input, options = {}) {
|
|
69
|
+
const executor = resolveCliExecutor(input.executor, options);
|
|
70
|
+
const model = resolveCliModel(executor, input.model, options);
|
|
71
|
+
return {
|
|
72
|
+
executor,
|
|
73
|
+
model,
|
|
74
|
+
judgeModel: defaultJudgeModel(executor, model),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
export function envJudgeModels(env = process.env) {
|
|
78
|
+
return nonEmpty(env.OMK_JUDGE_MODELS);
|
|
79
|
+
}
|
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { createHash } from 'node:crypto';
|
|
2
|
+
import { maxDiagnosisLifecycle } from './types.js';
|
|
3
|
+
import { ownRecordValue, setOwnRecordValue, sumRecordCounts, } from '../shared/record-count.js';
|
|
2
4
|
const OBSERVE_ONLY_COVERAGE = { observe: true, doctor: false, eval: false };
|
|
3
5
|
export function buildObserveDiagnostics(input) {
|
|
4
6
|
const byStableKey = new Map();
|
|
@@ -23,8 +25,9 @@ export function buildObserveDiagnostics(input) {
|
|
|
23
25
|
}
|
|
24
26
|
const bySkill = {};
|
|
25
27
|
for (const diagnosis of byStableKey.values()) {
|
|
26
|
-
bySkill
|
|
27
|
-
|
|
28
|
+
const group = ownRecordValue(bySkill, diagnosis.skillName) ?? [];
|
|
29
|
+
group.push(diagnosis);
|
|
30
|
+
setOwnRecordValue(bySkill, diagnosis.skillName, group);
|
|
28
31
|
}
|
|
29
32
|
for (const diagnoses of Object.values(bySkill)) {
|
|
30
33
|
diagnoses.sort((a, b) => severityRank(b.severity) - severityRank(a.severity) || a.signal.localeCompare(b.signal));
|
|
@@ -158,7 +161,7 @@ function makeDiagnosis(generatedAt, input) {
|
|
|
158
161
|
`target:${input.target}`,
|
|
159
162
|
].join('|');
|
|
160
163
|
const occurrence = {
|
|
161
|
-
id: hashParts('occurrence', stableKey, input.sourceKind, input.sourceId),
|
|
164
|
+
id: hashParts('occurrence', stableKey, input.sourceKind, input.sourceId, input.timestamp ?? generatedAt),
|
|
162
165
|
diagnosisStableKey: stableKey,
|
|
163
166
|
source: 'observe',
|
|
164
167
|
sourceId: input.sourceId,
|
|
@@ -198,10 +201,14 @@ function upsertDiagnosis(byStableKey, next) {
|
|
|
198
201
|
byStableKey.set(next.stableKey, next);
|
|
199
202
|
return;
|
|
200
203
|
}
|
|
201
|
-
existing.occurrences.
|
|
202
|
-
|
|
204
|
+
const existingOccurrenceIds = new Set(existing.occurrences.map((occurrence) => occurrence.id));
|
|
205
|
+
const hasOverlap = next.occurrences.some((occurrence) => existingOccurrenceIds.has(occurrence.id));
|
|
206
|
+
existing.occurrences.push(...next.occurrences.filter((occurrence) => !existingOccurrenceIds.has(occurrence.id)));
|
|
207
|
+
existing.occurrenceCount = hasOverlap
|
|
208
|
+
? Math.max(existing.occurrenceCount, next.occurrenceCount)
|
|
209
|
+
: sumRecordCounts(existing.occurrenceCount, next.occurrenceCount);
|
|
203
210
|
existing.severity = maxSeverity(existing.severity, next.severity);
|
|
204
|
-
existing.lifecycle =
|
|
211
|
+
existing.lifecycle = maxDiagnosisLifecycle(existing.lifecycle, next.lifecycle);
|
|
205
212
|
}
|
|
206
213
|
function typeForPattern(bucket, signal) {
|
|
207
214
|
if (bucket === 'definition_gap')
|
|
@@ -251,15 +258,7 @@ export function maxLifecycle(a, b) {
|
|
|
251
258
|
// 自动 detected 信号,应保留 resolved 而不是把 confirmation 擦掉。
|
|
252
259
|
// 「resolved 后新 occurrence 进来 reopen 到 detected」属于跨时间的 lifecycle 状态机,
|
|
253
260
|
// 由上层 review-state store 决定,不在此处处理。
|
|
254
|
-
|
|
255
|
-
resolved: 6,
|
|
256
|
-
rejected: 5,
|
|
257
|
-
detected: 4,
|
|
258
|
-
candidate: 3,
|
|
259
|
-
confirmed: 2,
|
|
260
|
-
stale: 1,
|
|
261
|
-
};
|
|
262
|
-
return rank[a] >= rank[b] ? a : b;
|
|
261
|
+
return maxDiagnosisLifecycle(a, b);
|
|
263
262
|
}
|
|
264
263
|
function severityRank(severity) {
|
|
265
264
|
if (severity === 'high')
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { buildObservationSkillChain, buildObservationSkillChains } from '../observability/skill-chain.js';
|
|
2
2
|
import { getSkillChainAdvisory, resolveAdvisoryCommand } from '../observability/skill-chain-advisories.js';
|
|
3
3
|
import { buildObserveDiagnostics } from './observe-mapper.js';
|
|
4
|
+
import { setOwnRecordValue } from '../shared/record-count.js';
|
|
4
5
|
export function buildObserveDiagnosticsFromReport(report, options = {}) {
|
|
5
6
|
const experienceReports = report.experience ? [report.experience] : [];
|
|
6
7
|
const skillNames = skillNamesFromReport(report);
|
|
@@ -33,7 +34,7 @@ function resolveSkillChains(report, skillNames, experienceReports, options) {
|
|
|
33
34
|
for (const skillName of skillNames) {
|
|
34
35
|
const cwd = cwdBySkill.get(skillName);
|
|
35
36
|
if (cwd) {
|
|
36
|
-
chains
|
|
37
|
+
setOwnRecordValue(chains, skillName, buildObservationSkillChain(skillName, cwd, experienceReports));
|
|
37
38
|
}
|
|
38
39
|
}
|
|
39
40
|
return chains;
|
|
@@ -184,6 +185,7 @@ function evidenceRef(ref) {
|
|
|
184
185
|
return {
|
|
185
186
|
id: ref.id,
|
|
186
187
|
kind: ref.kind,
|
|
188
|
+
traceId: ref.traceId,
|
|
187
189
|
sourceTrace: ref.sourceTrace,
|
|
188
190
|
sessionId: ref.sessionId,
|
|
189
191
|
messageIndex: ref.messageIndex,
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { isActiveDiagnosisLifecycle } from './types.js';
|
|
1
|
+
import { isActiveDiagnosisLifecycle, maxDiagnosisLifecycle, } from './types.js';
|
|
2
|
+
import { incrementRecordCount, ownRecordValue, setOwnRecordValue, sumRecordCounts, } from '../shared/record-count.js';
|
|
2
3
|
export function buildStudioDiagnosisSummary(bundle) {
|
|
3
4
|
const sourceCoverage = bundle?.sourceCoverage ?? { observe: false, doctor: false, eval: false };
|
|
4
5
|
const diagnostics = Object.values(bundle?.bySkill ?? {}).flat();
|
|
@@ -7,8 +8,8 @@ export function buildStudioDiagnosisSummary(bundle) {
|
|
|
7
8
|
const byAudience = {};
|
|
8
9
|
for (const diagnosis of diagnostics) {
|
|
9
10
|
bySeverity[diagnosis.severity] += 1;
|
|
10
|
-
bySkill
|
|
11
|
-
byAudience
|
|
11
|
+
incrementRecordCount(bySkill, diagnosis.skillName);
|
|
12
|
+
incrementRecordCount(byAudience, diagnosis.audience);
|
|
12
13
|
}
|
|
13
14
|
return {
|
|
14
15
|
sourceCoverage,
|
|
@@ -40,21 +41,27 @@ export function mergeDiagnosisBundles(bundles, generatedAt = new Date().toISOStr
|
|
|
40
41
|
});
|
|
41
42
|
continue;
|
|
42
43
|
}
|
|
43
|
-
existing.occurrences.
|
|
44
|
+
const existingOccurrenceIds = new Set(existing.occurrences.map((occurrence) => occurrence.id));
|
|
45
|
+
const hasOverlap = diagnosis.occurrences.some((occurrence) => existingOccurrenceIds.has(occurrence.id));
|
|
46
|
+
existing.occurrences.push(...diagnosis.occurrences.filter((occurrence) => !existingOccurrenceIds.has(occurrence.id)));
|
|
44
47
|
// 累加而不是用 occurrences.length 覆盖:聚合型 Diagnosis(如 problemPattern)的
|
|
45
48
|
// occurrenceCount 是「N 次真实发生」的语义,不是「N 条 source occurrence 条数」。
|
|
46
49
|
// 用 length 会把 `3 次 + 4 次` 真实发生压成 `2`(两条 source 条目),Studio 排序和
|
|
47
50
|
// affectedCount 显示偏小。
|
|
48
|
-
existing.occurrenceCount =
|
|
51
|
+
existing.occurrenceCount = hasOverlap
|
|
52
|
+
? Math.max(existing.occurrenceCount, diagnosis.occurrenceCount)
|
|
53
|
+
: sumRecordCounts(existing.occurrenceCount, diagnosis.occurrenceCount);
|
|
49
54
|
existing.severity = severityRank(existing.severity) >= severityRank(diagnosis.severity)
|
|
50
55
|
? existing.severity
|
|
51
56
|
: diagnosis.severity;
|
|
57
|
+
existing.lifecycle = maxDiagnosisLifecycle(existing.lifecycle, diagnosis.lifecycle);
|
|
52
58
|
}
|
|
53
59
|
}
|
|
54
60
|
const bySkill = {};
|
|
55
61
|
for (const diagnosis of byStableKey.values()) {
|
|
56
|
-
bySkill
|
|
57
|
-
|
|
62
|
+
const group = ownRecordValue(bySkill, diagnosis.skillName) ?? [];
|
|
63
|
+
group.push(diagnosis);
|
|
64
|
+
setOwnRecordValue(bySkill, diagnosis.skillName, group);
|
|
58
65
|
}
|
|
59
66
|
for (const values of Object.values(bySkill)) {
|
|
60
67
|
values.sort((a, b) => severityRank(b.severity) - severityRank(a.severity) || a.signal.localeCompare(b.signal));
|
|
@@ -1,3 +1,5 @@
|
|
|
1
1
|
export type { Diagnosis, DiagnosisAudience, DiagnosisBundle, DiagnosisEvidenceRef, DiagnosisLifecycle, DiagnosisOccurrence, DiagnosisPatch, DiagnosisScope, DiagnosisSeverity, DiagnosisSource, DiagnosisSourceCoverage, DiagnosisType, StudioDiagnosisSummary, } from '../types/diagnosis.js';
|
|
2
2
|
import type { DiagnosisLifecycle } from '../types/diagnosis.js';
|
|
3
3
|
export declare function isActiveDiagnosisLifecycle(lifecycle: DiagnosisLifecycle): boolean;
|
|
4
|
+
/** Merge lifecycle states without making the result depend on bundle order. */
|
|
5
|
+
export declare function maxDiagnosisLifecycle(a: DiagnosisLifecycle, b: DiagnosisLifecycle): DiagnosisLifecycle;
|
package/dist/diagnosis/types.js
CHANGED
|
@@ -17,3 +17,15 @@ const ACTIVE_DIAGNOSIS_LIFECYCLES = new Set([
|
|
|
17
17
|
export function isActiveDiagnosisLifecycle(lifecycle) {
|
|
18
18
|
return ACTIVE_DIAGNOSIS_LIFECYCLES.has(lifecycle);
|
|
19
19
|
}
|
|
20
|
+
/** Merge lifecycle states without making the result depend on bundle order. */
|
|
21
|
+
export function maxDiagnosisLifecycle(a, b) {
|
|
22
|
+
const rank = {
|
|
23
|
+
resolved: 6,
|
|
24
|
+
rejected: 5,
|
|
25
|
+
detected: 4,
|
|
26
|
+
candidate: 3,
|
|
27
|
+
confirmed: 2,
|
|
28
|
+
stale: 1,
|
|
29
|
+
};
|
|
30
|
+
return rank[a] >= rank[b] ? a : b;
|
|
31
|
+
}
|
|
@@ -25,6 +25,7 @@
|
|
|
25
25
|
*/
|
|
26
26
|
import { existsSync, readFileSync, readdirSync, lstatSync, openSync, readSync, closeSync } from 'node:fs';
|
|
27
27
|
import { join } from 'node:path';
|
|
28
|
+
import { setOwnRecordValue } from '../shared/record-count.js';
|
|
28
29
|
const DEFAULT_MAX_FILE_BYTES = 200 * 1024;
|
|
29
30
|
const DEFAULT_MAX_TOTAL_BYTES = 2 * 1024 * 1024;
|
|
30
31
|
/** 二进制 / 大文件不可读时跳过;只收文本。简单按扩展名 + 内容嗅探。 */
|
|
@@ -125,7 +126,7 @@ function collectFiles(skillRoot, maxFileBytes, maxTotalBytes) {
|
|
|
125
126
|
const over = buf.length > cap;
|
|
126
127
|
const slice = over ? buf.subarray(0, cap) : buf;
|
|
127
128
|
const text = slice.toString('utf-8') + (over ? '\n…[truncated]' : '');
|
|
128
|
-
files
|
|
129
|
+
setOwnRecordValue(files, relPath, text);
|
|
129
130
|
total += Buffer.byteLength(text, 'utf-8');
|
|
130
131
|
}
|
|
131
132
|
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
1
2
|
import { join } from 'node:path';
|
|
2
3
|
export const REPORT_FILE_SUFFIX = '.report.json';
|
|
3
4
|
export const GRAPH_FILE_SUFFIX = '.graph.json';
|
|
@@ -41,5 +42,21 @@ export function stripDomainPrefix(id, domain) {
|
|
|
41
42
|
return safeId.startsWith(prefix) ? safeId.slice(prefix.length) : safeId;
|
|
42
43
|
}
|
|
43
44
|
export function doctorReportFileStem(skillName, reportId) {
|
|
44
|
-
|
|
45
|
+
const reportSuffix = reportId.startsWith('doctor-')
|
|
46
|
+
? reportId.slice('doctor-'.length)
|
|
47
|
+
: reportId;
|
|
48
|
+
return [
|
|
49
|
+
collisionResistantFileComponent(skillName),
|
|
50
|
+
collisionResistantFileComponent(reportSuffix),
|
|
51
|
+
].join('-');
|
|
52
|
+
}
|
|
53
|
+
function collisionResistantFileComponent(value) {
|
|
54
|
+
const safe = safeArtifactFileStem(value);
|
|
55
|
+
if (safe === value)
|
|
56
|
+
return safe;
|
|
57
|
+
const fingerprint = createHash('sha256')
|
|
58
|
+
.update(value)
|
|
59
|
+
.digest('hex')
|
|
60
|
+
.slice(0, 12);
|
|
61
|
+
return `${safe}-${fingerprint}`;
|
|
45
62
|
}
|
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
import type { ReportDocument, ReportIndexCard } from '../types/index.js';
|
|
2
|
-
import type { SkillDoctorSnapshot } from '../types/skill-index.js';
|
|
3
2
|
import type { DoctorSkillStatus } from '../types/doctor.js';
|
|
4
3
|
export type ArtifactDomain = 'report' | 'doctor' | 'observe-health';
|
|
5
4
|
/** 某域的索引目录 `<root>/<domain>/`。 */
|
|
@@ -17,15 +16,11 @@ export declare function cardTargetSentinel(domain: ArtifactDomain): string;
|
|
|
17
16
|
* persistReport 的索引钩子:报告落盘后 best-effort 追加卡片。永不抛、永不阻断报告落盘。
|
|
18
17
|
* 入参故意收 `{id}` 宽松(同 persistReport 的 PersistableReport):非完整报告(无 canonical kind)防御式跳过。
|
|
19
18
|
*/
|
|
20
|
-
export declare function indexReportWrite(report:
|
|
21
|
-
id: string;
|
|
22
|
-
}, sourcePath: string, outputDir: string): void;
|
|
19
|
+
export declare function indexReportWrite(report: ReportDocument, sourcePath: string, outputDir: string): void;
|
|
23
20
|
/** 读 report 域全部卡片(跳过坏文件 / 缺字段 / 坏 kind)。 */
|
|
24
21
|
export declare function listReportCards(): ReportIndexCard[];
|
|
25
22
|
/** report 卡片(过滤悬空真身):供 studio 机器级 list / findBy 展示。 */
|
|
26
23
|
export declare function listLiveReportCards(): ReportIndexCard[];
|
|
27
|
-
/** 卡片 → ReportDocument(results:[]):供 studio list / buildSkillIndex / trend 消费(它们不读 results)。 */
|
|
28
|
-
export declare function cardToReportDocument(card: ReportIndexCard): ReportDocument;
|
|
29
24
|
/** 删 report 域某 id 的卡片(DELETE 报告时连卡片一起删,使其从机器级 list 消失)。幂等、best-effort。 */
|
|
30
25
|
export declare function removeReportCard(id: string): boolean;
|
|
31
26
|
/** doctor 卡片:per-skill 文件(`{skillName}-{reportId}.json`)一张。id = 文件名 stem(唯一),
|
|
@@ -49,22 +44,23 @@ export declare function indexDoctorWrite(card: Omit<DoctorIndexCard, 'domain'>,
|
|
|
49
44
|
export declare function listDoctorCards(): DoctorIndexCard[];
|
|
50
45
|
/** doctor 卡片(过滤悬空真身):供 buildSkillIndex 机器级合并。 */
|
|
51
46
|
export declare function listLiveDoctorCards(): DoctorIndexCard[];
|
|
52
|
-
/** doctor 卡片 → (skillName, SkillDoctorSnapshot)。results:[](逐规则详情需回源项目看)。 */
|
|
53
|
-
export declare function cardToDoctorSnapshot(card: DoctorIndexCard): {
|
|
54
|
-
skillName: string;
|
|
55
|
-
snap: SkillDoctorSnapshot;
|
|
56
|
-
};
|
|
57
47
|
/** 删 doctor 域某 id(文件 stem)的卡片。doctor 历史按 50/skill prune,删正文时必须连卡片一起删,
|
|
58
48
|
* 否则被 prune 掉的报告会经卡片合并在本项目 studio「复活」。 */
|
|
59
49
|
export declare function removeDoctorCard(id: string): boolean;
|
|
60
50
|
/** observe 报告投影到卡片所需的结构子集(不 import observability/SkillHealthReport,保持 eval-core 解耦)。 */
|
|
61
51
|
interface ObserveCardSkill {
|
|
62
52
|
toolFailureRate: number;
|
|
53
|
+
toolFailureCount?: number;
|
|
54
|
+
toolCallCount?: number;
|
|
55
|
+
toolResolvedCount?: number;
|
|
56
|
+
toolCancelledCount?: number;
|
|
57
|
+
toolUnknownCount?: number;
|
|
63
58
|
segmentCount: number;
|
|
64
59
|
gap?: {
|
|
65
60
|
weightedGapRate?: number;
|
|
66
61
|
};
|
|
67
62
|
confidence?: 'high' | 'low' | 'underpowered';
|
|
63
|
+
stability?: 'stable' | 'unstable' | 'very-unstable' | 'unknown';
|
|
68
64
|
}
|
|
69
65
|
interface ObserveSource {
|
|
70
66
|
meta: {
|