oh-my-knowledge 0.38.0 → 0.40.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/agent-skills/omk/references/commands.md +2 -0
- package/dist/authoring/evolver.js +9 -5
- package/dist/cli/commands/list.d.ts +4 -1
- package/dist/cli/commands/list.js +12 -4
- package/dist/cli/commands/observe/index.d.ts +9 -0
- package/dist/cli/commands/observe/index.js +74 -1
- package/dist/cli/commands/sample.d.ts +13 -0
- package/dist/cli/commands/sample.js +113 -16
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +8 -0
- package/dist/cli/lib/i18n-dict/list.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/list.js +4 -0
- package/dist/cli/lib/record-evolve-outcome.js +19 -10
- package/dist/eval-core/bootstrap.d.ts +53 -2
- package/dist/eval-core/bootstrap.js +82 -5
- package/dist/eval-core/evaluation-reporting.js +37 -15
- package/dist/eval-core/execution-strategy.js +1 -0
- package/dist/eval-core/schema.js +5 -3
- package/dist/eval-core/verdict.d.ts +30 -0
- package/dist/eval-core/verdict.js +71 -21
- package/dist/grading/assertions.js +28 -0
- package/dist/grading/debias-validate.js +6 -2
- package/dist/grading/index.js +6 -1
- package/dist/grading/layered-scores.d.ts +15 -0
- package/dist/grading/layered-scores.js +64 -38
- package/dist/inputs/load-samples.d.ts +5 -0
- package/dist/inputs/load-samples.js +4 -2
- package/dist/inputs/skill-loader.d.ts +8 -0
- package/dist/inputs/skill-loader.js +23 -0
- package/dist/managed/evidence.d.ts +2 -1
- package/dist/managed/evidence.js +15 -2
- package/dist/managed/index.d.ts +2 -0
- package/dist/managed/index.js +2 -0
- package/dist/managed/list-view.d.ts +15 -1
- package/dist/managed/list-view.js +9 -1
- package/dist/managed/observe-feedback.d.ts +60 -0
- package/dist/managed/observe-feedback.js +49 -0
- package/dist/managed/store.d.ts +38 -1
- package/dist/managed/store.js +116 -3
- package/dist/managed/version-scores.d.ts +33 -0
- package/dist/managed/version-scores.js +70 -0
- package/dist/observability/skill-health-analyzer.d.ts +5 -0
- package/dist/observability/skill-health-analyzer.js +3 -2
- package/dist/renderer/managed-history-renderer.d.ts +2 -2
- package/dist/renderer/managed-history-renderer.js +189 -7
- package/dist/renderer/summary.js +23 -22
- package/dist/server/report-server.js +11 -2
- package/dist/types/eval.d.ts +2 -0
- package/dist/types/judge.d.ts +8 -0
- package/dist/types/managed.d.ts +40 -0
- package/dist/types/report.d.ts +10 -0
- package/package.json +3 -3
|
@@ -1,35 +1,57 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
'
|
|
18
|
-
'
|
|
19
|
-
'
|
|
20
|
-
'
|
|
21
|
-
'
|
|
22
|
-
'
|
|
23
|
-
'
|
|
24
|
-
'
|
|
25
|
-
'
|
|
26
|
-
'
|
|
27
|
-
'
|
|
28
|
-
'
|
|
29
|
-
|
|
30
|
-
'
|
|
31
|
-
'
|
|
32
|
-
|
|
1
|
+
/**
|
|
2
|
+
* 每个 assertion 类型归到事实层 / 行为层。**单一来源**:`computeLayeredScores` 据此把 assertion 明细拆进
|
|
3
|
+
* fact / behavior 两层算分。
|
|
4
|
+
* - **事实层(fact)**:测输出内容对不对 —— 命中 / 匹配 / 结构合法 / 与参考的文本相似度。
|
|
5
|
+
* - **行为层(behavior)**:测做事的方式 —— 长度 / 成本 / 轮次 / 工具调用 / 必经里程碑,不看内容本身。
|
|
6
|
+
*
|
|
7
|
+
* 不变量:**runner 支持的每个 assertion 类型都必须能被归层**,否则该类型的 pass/fail 会被 computeLayeredScores
|
|
8
|
+
* 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
|
|
9
|
+
* bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
|
|
10
|
+
* 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
|
|
11
|
+
* 解析(同层→归层、混层→不计),结果落 detail.layer。`test/grading/layered-scores-exhaustiveness.test.ts` 扫
|
|
12
|
+
* runner 源(evalAssertion 的 case ∪ `assertion.type ===` 组合器 ∪ ASYNC_ASSERTION_TYPES)守住:新增类型既不在本
|
|
13
|
+
* 映射、又不是已知组合器,即 CI 失败。
|
|
14
|
+
*/
|
|
15
|
+
export const ASSERTION_LAYER = {
|
|
16
|
+
contains: 'fact',
|
|
17
|
+
not_contains: 'fact',
|
|
18
|
+
regex: 'fact',
|
|
19
|
+
json_valid: 'fact',
|
|
20
|
+
json_schema: 'fact',
|
|
21
|
+
equals: 'fact',
|
|
22
|
+
not_equals: 'fact',
|
|
23
|
+
contains_all: 'fact',
|
|
24
|
+
contains_any: 'fact',
|
|
25
|
+
semantic_similarity: 'fact',
|
|
26
|
+
tool_output_contains: 'fact',
|
|
27
|
+
tool_input_contains: 'fact',
|
|
28
|
+
tool_input_not_contains: 'fact',
|
|
29
|
+
// 文本相似度指标:衡量输出与参考答案的内容贴合度 = 内容保真 → 事实层。
|
|
30
|
+
rouge_n_min: 'fact',
|
|
31
|
+
bleu_min: 'fact',
|
|
32
|
+
levenshtein_max: 'fact',
|
|
33
|
+
// RAG 评委指标:都评输出内容质量(忠实度 / 答非所问 / 关键事实召回),= 内容正确性 → 事实层。
|
|
34
|
+
faithfulness: 'fact',
|
|
35
|
+
answer_relevancy: 'fact',
|
|
36
|
+
context_recall: 'fact',
|
|
37
|
+
starts_with: 'behavior',
|
|
38
|
+
ends_with: 'behavior',
|
|
39
|
+
min_length: 'behavior',
|
|
40
|
+
max_length: 'behavior',
|
|
41
|
+
word_count_min: 'behavior',
|
|
42
|
+
word_count_max: 'behavior',
|
|
43
|
+
cost_max: 'behavior',
|
|
44
|
+
latency_max: 'behavior',
|
|
45
|
+
turns_min: 'behavior',
|
|
46
|
+
turns_max: 'behavior',
|
|
47
|
+
tools_called: 'behavior',
|
|
48
|
+
tools_not_called: 'behavior',
|
|
49
|
+
tools_count_min: 'behavior',
|
|
50
|
+
tools_count_max: 'behavior',
|
|
51
|
+
custom: 'behavior',
|
|
52
|
+
// 必经工具 / 步骤里程碑(value 形如 "Tool:N"):测"有没有走到那一步" = 做事方式 → 行为层(类同 tools_called)。
|
|
53
|
+
mock_hit: 'behavior',
|
|
54
|
+
};
|
|
33
55
|
function ratioToScore(ratio) {
|
|
34
56
|
return Number((1 + ratio * 4).toFixed(2));
|
|
35
57
|
}
|
|
@@ -45,15 +67,19 @@ function scoreFromDetails(details) {
|
|
|
45
67
|
export function computeLayeredScores(results) {
|
|
46
68
|
const layered = {};
|
|
47
69
|
if (results.assertions?.details) {
|
|
48
|
-
|
|
49
|
-
|
|
70
|
+
// 优先用 detail.layer(组合器如 assert-set 在 grading 期按 children 解析出的层;混层组合器无 layer → 两层都不计,
|
|
71
|
+
// 见 assertions.ts resolveAssertSetLayer);叶子断言无 layer,退回静态 ASSERTION_LAYER[type]。
|
|
72
|
+
const layerOf = (d) => d.layer ?? ASSERTION_LAYER[d.type];
|
|
73
|
+
const factDetails = results.assertions.details.filter((d) => layerOf(d) === 'fact');
|
|
74
|
+
const behaviorDetails = results.assertions.details.filter((d) => layerOf(d) === 'behavior');
|
|
50
75
|
layered.factScore = scoreFromDetails(factDetails) ?? undefined;
|
|
51
76
|
layered.behaviorScore = scoreFromDetails(behaviorDetails) ?? undefined;
|
|
52
77
|
}
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
|
|
78
|
+
// 评委分量表是 1-5(见 judge.ts prompt),`score <= 0` 是失败哨兵(非 JSON / parse / executor 错),
|
|
79
|
+
// 是**缺测**不是「0 分内容」—— 当缺层排除,绝不让基础设施失败当内容分污染 composite。上游(grading/index.ts
|
|
80
|
+
// 单评委路径 + 多维度 filter(s>0) + 多轮 validSamples)已保证失败不进 llmScore;这里再 `> 0` 防御一层,
|
|
81
|
+
// 与全管线"0=失败=排除"口径一致(fact/behavior 经 ratioToScore 恒 ≥ 1,有效评委分亦 ≥ 1)。
|
|
82
|
+
if (typeof results.llmScore === 'number' && results.llmScore > 0) {
|
|
57
83
|
layered.judgeScore = results.llmScore;
|
|
58
84
|
}
|
|
59
85
|
const scores = [layered.factScore, layered.behaviorScore, layered.judgeScore].filter((s) => s != null);
|
|
@@ -32,4 +32,9 @@ export interface LoadSamplesResult {
|
|
|
32
32
|
* - `requires` from each file unioned together
|
|
33
33
|
*/
|
|
34
34
|
export declare function loadSamples(samplesPath: string): LoadSamplesResult;
|
|
35
|
+
/** Pull `.json/.yaml/.yml` siblings out of a directory, skipping omk's own report/health
|
|
36
|
+
* artifacts and any underscore-prefixed file (the convention for "not a sample").
|
|
37
|
+
* Exported so `omk sample --append` picks its target with the same sorted/filtered order
|
|
38
|
+
* that directory-mode loading merges by (deterministic, predictable). */
|
|
39
|
+
export declare function listSampleFilesInDir(dir: string): string[];
|
|
35
40
|
export declare function validateSamples(samples: Sample[]): void;
|
|
@@ -38,8 +38,10 @@ export function loadSamples(samplesPath) {
|
|
|
38
38
|
};
|
|
39
39
|
}
|
|
40
40
|
/** Pull `.json/.yaml/.yml` siblings out of a directory, skipping omk's own report/health
|
|
41
|
-
* artifacts and any underscore-prefixed file (the convention for "not a sample").
|
|
42
|
-
|
|
41
|
+
* artifacts and any underscore-prefixed file (the convention for "not a sample").
|
|
42
|
+
* Exported so `omk sample --append` picks its target with the same sorted/filtered order
|
|
43
|
+
* that directory-mode loading merges by (deterministic, predictable). */
|
|
44
|
+
export function listSampleFilesInDir(dir) {
|
|
43
45
|
const RESERVED = /^(report|health|_)/i;
|
|
44
46
|
return readdirSync(dir)
|
|
45
47
|
.filter((f) => /\.(json|ya?ml)$/i.test(f))
|
|
@@ -5,6 +5,14 @@ export declare function gitShowFile(ref: string, filePath: string, cwd?: string)
|
|
|
5
5
|
* 同 gitShowFile 用 `cat-file blob`:对目录会非零退出,不会把树清单字节当文件内容物化。
|
|
6
6
|
*/
|
|
7
7
|
export declare function gitShowBytes(ref: string, filePath: string, cwd?: string): Buffer | null;
|
|
8
|
+
/**
|
|
9
|
+
* 把一个 ref(branch / tag / HEAD / 缩写或完整 SHA)解析到它指向的 commit SHA(#234/#236 还原坐标)。
|
|
10
|
+
* `^{commit}` 会把 annotated tag 也 peel 到 commit;`--verify --quiet` 解析不出时静默非零退出 → 落 null。
|
|
11
|
+
* 不存在 / 出错 → null(best-effort,绝不抛、不阻断 eval)。`<ref>` 物化内容时用的就是它,故这是被测字节的定点
|
|
12
|
+
* (与工作树是否 dirty 无关 —— 内容从 object DB 按 ref 取)。dash-ref(前缀 `-`)直接挡,不得被当 git 选项
|
|
13
|
+
* (rev-parse 的 rev 不能放 `--` 之后,故显式前置守卫,与 gitShowBytes 的 `--` 同口径 fail-closed)。
|
|
14
|
+
*/
|
|
15
|
+
export declare function gitResolveCommit(ref: string, cwd?: string): string | null;
|
|
8
16
|
export interface GitTreeEntry {
|
|
9
17
|
/** git 文件模式:100644/100755=普通文件,120000=软链,160000=submodule。 */
|
|
10
18
|
mode: string;
|
|
@@ -73,6 +73,24 @@ export function gitShowBytes(ref, filePath, cwd = process.cwd()) {
|
|
|
73
73
|
return null;
|
|
74
74
|
}
|
|
75
75
|
}
|
|
76
|
+
/**
|
|
77
|
+
* 把一个 ref(branch / tag / HEAD / 缩写或完整 SHA)解析到它指向的 commit SHA(#234/#236 还原坐标)。
|
|
78
|
+
* `^{commit}` 会把 annotated tag 也 peel 到 commit;`--verify --quiet` 解析不出时静默非零退出 → 落 null。
|
|
79
|
+
* 不存在 / 出错 → null(best-effort,绝不抛、不阻断 eval)。`<ref>` 物化内容时用的就是它,故这是被测字节的定点
|
|
80
|
+
* (与工作树是否 dirty 无关 —— 内容从 object DB 按 ref 取)。dash-ref(前缀 `-`)直接挡,不得被当 git 选项
|
|
81
|
+
* (rev-parse 的 rev 不能放 `--` 之后,故显式前置守卫,与 gitShowBytes 的 `--` 同口径 fail-closed)。
|
|
82
|
+
*/
|
|
83
|
+
export function gitResolveCommit(ref, cwd = process.cwd()) {
|
|
84
|
+
if (!ref || ref.startsWith('-'))
|
|
85
|
+
return null;
|
|
86
|
+
try {
|
|
87
|
+
const out = execFileSync('git', ['rev-parse', '--verify', '--quiet', `${ref}^{commit}`], { cwd, encoding: 'utf-8', stdio: GIT_PROBE_STDIO }).trim();
|
|
88
|
+
return out || null;
|
|
89
|
+
}
|
|
90
|
+
catch {
|
|
91
|
+
return null;
|
|
92
|
+
}
|
|
93
|
+
}
|
|
76
94
|
/**
|
|
77
95
|
* 递归列出 `<ref>:<treePath>` 子树下的叶子条目(blob / 软链 / submodule)。tree 不存在返回 []。
|
|
78
96
|
* 用 `-z`(NUL 分隔):git 不会对含换行 / 非 ASCII 的路径做 C-quote,路径原样可回喂 git show。
|
|
@@ -632,6 +650,9 @@ export function resolveArtifacts(skillDir, variants, opts = {}) {
|
|
|
632
650
|
if (!resolved) {
|
|
633
651
|
throw new Error(`skill not found in git ${ref}: ${name}.md or ${name}/SKILL.md`);
|
|
634
652
|
}
|
|
653
|
+
// #234/#236:把 variant 的 ref 解析到实际 commit 当还原坐标。是 ref(可能 branch/tag/HEAD)而非进程
|
|
654
|
+
// cwd 的 HEAD —— 内容从 object DB 按这个 ref 物化,坐标必须对齐它(否则在别的分支跑会记错版本)。
|
|
655
|
+
const resolvedCommit = gitResolveCommit(ref, gitCtx.repoRoot) ?? undefined;
|
|
635
656
|
if (resolved.isDir) {
|
|
636
657
|
// git 目录-skill 忠实执行:物化整树到临时目录 → 落地内容寻址隔离副本 → executor cwd 锚副本,
|
|
637
658
|
// agent 读得到 references/ 资产、资产成为真实运行时输入。整树指纹与 install 受管记录的
|
|
@@ -652,6 +673,7 @@ export function resolveArtifacts(skillDir, variants, opts = {}) {
|
|
|
652
673
|
contentHash: isolated.contentHash,
|
|
653
674
|
locator: name,
|
|
654
675
|
ref,
|
|
676
|
+
...(resolvedCommit ? { resolvedCommit } : {}),
|
|
655
677
|
cwd: variantCwd,
|
|
656
678
|
...(isolated.execRoot ? { execRoot: isolated.execRoot } : {}),
|
|
657
679
|
});
|
|
@@ -672,6 +694,7 @@ export function resolveArtifacts(skillDir, variants, opts = {}) {
|
|
|
672
694
|
contentHash: hashBytes(skillMdBytes),
|
|
673
695
|
locator: name,
|
|
674
696
|
ref,
|
|
697
|
+
...(resolvedCommit ? { resolvedCommit } : {}),
|
|
675
698
|
cwd: variantCwd,
|
|
676
699
|
});
|
|
677
700
|
continue;
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { EvaluationReport, ManagedEvidenceRef } from '../types/index.js';
|
|
2
2
|
/**
|
|
3
3
|
* 为某个变体组装一条 evidence ref;变体无真实内容(baseline / no-skill / 缺 hash)→ null。
|
|
4
|
-
* `verdict` 由调用方传入(CLI 已 computeVerdict,避免在此重算)。
|
|
4
|
+
* `verdict` 由调用方传入(CLI 已 computeVerdict,避免在此重算)。git 还原坐标从该 variant 的
|
|
5
|
+
* `resolvedCommit` 取(见 variantResolvedCommit),无则不记。
|
|
5
6
|
*/
|
|
6
7
|
export declare function buildEvidenceRef(report: EvaluationReport, variant: string, verdict: string, recordedAt: string): ManagedEvidenceRef | null;
|
|
7
8
|
export interface RecordedEvidence {
|
package/dist/managed/evidence.js
CHANGED
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
*/
|
|
28
28
|
import { basename, dirname } from 'node:path';
|
|
29
29
|
import { hashString } from '../eval-core/evaluation-reporting.js';
|
|
30
|
-
import { loadAllManagedRecords, appendManagedEvidence, managedDir, resolveManagedDir } from './store.js';
|
|
30
|
+
import { loadAllManagedRecords, appendManagedEvidence, managedDir, resolveManagedDir, isShaLike } from './store.js';
|
|
31
31
|
/** baseline / 无 skill 变体的 artifactHash 哨兵(见 report.ts artifactHashes 注释)——不产证据。 */
|
|
32
32
|
const NO_SKILL = 'no-skill';
|
|
33
33
|
/** 样本集覆盖摘要:report 的 sampleHashes 排序后取一个稳定 digest(同一样本集 ⇒ 同 hash)。
|
|
@@ -39,9 +39,20 @@ function sampleCoverage(report) {
|
|
|
39
39
|
const entries = Object.entries(sh).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
|
|
40
40
|
return { count: entries.length, hash: hashString(JSON.stringify(entries)) };
|
|
41
41
|
}
|
|
42
|
+
/**
|
|
43
|
+
* 该版的 git 还原坐标(#234/#236):被测 variant 物化时 `<ref>` 解析出的实际 commit(resolveArtifacts 在
|
|
44
|
+
* 物化本地 git variant 时算好,经 `variantConfigs[].resolvedCommit` 透传)。**不是**进程 cwd 的 HEAD ——
|
|
45
|
+
* variant 内容从 object DB 按它自己的 ref 取,在别的分支 / 用显式旧 SHA 跑时 cwd HEAD 会是错版本。
|
|
46
|
+
* 远端 / file variant 无 resolvedCommit → 不记(诚实留空)。读到的值再过一道 SHA 形态守卫,脏报告不污染证据。
|
|
47
|
+
*/
|
|
48
|
+
function variantResolvedCommit(report, variant) {
|
|
49
|
+
const cfg = report.meta?.variantConfigs?.find((c) => c.variant === variant);
|
|
50
|
+
return isShaLike(cfg?.resolvedCommit) ? cfg.resolvedCommit : undefined;
|
|
51
|
+
}
|
|
42
52
|
/**
|
|
43
53
|
* 为某个变体组装一条 evidence ref;变体无真实内容(baseline / no-skill / 缺 hash)→ null。
|
|
44
|
-
* `verdict` 由调用方传入(CLI 已 computeVerdict,避免在此重算)。
|
|
54
|
+
* `verdict` 由调用方传入(CLI 已 computeVerdict,避免在此重算)。git 还原坐标从该 variant 的
|
|
55
|
+
* `resolvedCommit` 取(见 variantResolvedCommit),无则不记。
|
|
45
56
|
*/
|
|
46
57
|
export function buildEvidenceRef(report, variant, verdict, recordedAt) {
|
|
47
58
|
const contentHash = report.meta?.artifactHashes?.[variant];
|
|
@@ -49,6 +60,7 @@ export function buildEvidenceRef(report, variant, verdict, recordedAt) {
|
|
|
49
60
|
return null;
|
|
50
61
|
const meta = report.meta;
|
|
51
62
|
const cov = sampleCoverage(report);
|
|
63
|
+
const gitCommit = variantResolvedCommit(report, variant);
|
|
52
64
|
return {
|
|
53
65
|
reportId: report.id,
|
|
54
66
|
contentHash,
|
|
@@ -60,6 +72,7 @@ export function buildEvidenceRef(report, variant, verdict, recordedAt) {
|
|
|
60
72
|
...(meta.judgePromptHash ? { judgePromptHash: meta.judgePromptHash } : {}),
|
|
61
73
|
...(meta.debiasMode ? { debiasMode: meta.debiasMode } : {}),
|
|
62
74
|
},
|
|
75
|
+
...(gitCommit ? { gitCommit } : {}),
|
|
63
76
|
};
|
|
64
77
|
}
|
|
65
78
|
/**
|
package/dist/managed/index.d.ts
CHANGED
|
@@ -4,7 +4,9 @@
|
|
|
4
4
|
*/
|
|
5
5
|
export * from './store.js';
|
|
6
6
|
export * from './evidence.js';
|
|
7
|
+
export * from './observe-feedback.js';
|
|
7
8
|
export * from './list-view.js';
|
|
9
|
+
export * from './version-scores.js';
|
|
8
10
|
export * from './list-query.js';
|
|
9
11
|
export * from './source-probe.js';
|
|
10
12
|
export * from './promote-gate.js';
|
package/dist/managed/index.js
CHANGED
|
@@ -4,7 +4,9 @@
|
|
|
4
4
|
*/
|
|
5
5
|
export * from './store.js';
|
|
6
6
|
export * from './evidence.js';
|
|
7
|
+
export * from './observe-feedback.js';
|
|
7
8
|
export * from './list-view.js';
|
|
9
|
+
export * from './version-scores.js';
|
|
8
10
|
export * from './list-query.js';
|
|
9
11
|
export * from './source-probe.js';
|
|
10
12
|
export * from './promote-gate.js';
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* reachable 时才走 `deriveManagedState`(哈不等 → stale)。verdict / 可比性取**当前有效证据**
|
|
9
9
|
* (contentHash == record.contentHash)里 recordedAt 最新那条 —— 旧内容的证据不冒充当前。
|
|
10
10
|
*/
|
|
11
|
-
import type { ArtifactKind, ManagedArtifactRecord, ManagedLifecycleLabel } from '../types/index.js';
|
|
11
|
+
import type { ArtifactKind, ManagedArtifactRecord, ManagedLifecycleLabel, ManagedObservation } from '../types/index.js';
|
|
12
12
|
/** 当前源探测结果(三态)。`reachable:false` = 不可达 / 解析失败 / 拒读,**不等于**已 drift。 */
|
|
13
13
|
export interface SourceProbe {
|
|
14
14
|
reachable: boolean;
|
|
@@ -36,6 +36,20 @@ export interface ManagedListRow {
|
|
|
36
36
|
judgePromptHash?: string;
|
|
37
37
|
debiasMode?: Array<'length' | 'position'>;
|
|
38
38
|
};
|
|
39
|
+
/** 当前 promoted 版本是否经 `--force` override 采用(只读审计标);非越门 / 已 rollback / 未采用 → undefined。
|
|
40
|
+
* override 的写仍只在 CLI(`promote --force`),Studio 只读展示 —— 见 spec §9(#238)。 */
|
|
41
|
+
override?: {
|
|
42
|
+
verdict: string;
|
|
43
|
+
overriddenBlocks?: string[];
|
|
44
|
+
};
|
|
45
|
+
/** observe 生产盲区 marker(#235):仅当 `deriveProductionGap` 取最新观测为**确诊盲区**(红 + 统计够力)时填。
|
|
46
|
+
* 与 `state` / `drifted` **正交** —— 它是版本无关的生产信号(observe 量线上部署版),绝不参与生命周期、不翻
|
|
47
|
+
* stale(见 store.deriveProductionGap)。underpowered / 非红观测不在此 surface(留 unknown / 信息态,避免噪声)。 */
|
|
48
|
+
productionGap?: {
|
|
49
|
+
healthBand: ManagedObservation['healthBand'];
|
|
50
|
+
confidence: ManagedObservation['confidence'];
|
|
51
|
+
gapByType: ManagedObservation['gapByType'];
|
|
52
|
+
};
|
|
39
53
|
/** 最新当前证据的记录时间。 */
|
|
40
54
|
recordedAt?: string;
|
|
41
55
|
/** 当前有效证据数 / 全部证据数(含旧内容的历史证据)。 */
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { deriveManagedState, isCurrentlyPromoted } from './store.js';
|
|
1
|
+
import { deriveManagedState, isCurrentlyPromoted, currentPromoteOverride, deriveProductionGap } from './store.js';
|
|
2
2
|
/** 当前有效证据(contentHash == record.contentHash)里 recordedAt 最新那条;无则 undefined。
|
|
3
3
|
* list(展示最新 verdict)与 promote(门禁取证)共用同一口径——旧内容的证据不冒充当前。 */
|
|
4
4
|
export function latestCurrentEvidence(record) {
|
|
@@ -35,6 +35,10 @@ export function buildManagedListRow(record, probe) {
|
|
|
35
35
|
drifted = false;
|
|
36
36
|
}
|
|
37
37
|
const latest = latestCurrentEvidence(record);
|
|
38
|
+
const override = currentPromoteOverride(record);
|
|
39
|
+
// 生产盲区是读时 marker、与生命周期正交:在 reachable / unreachable / 任意 state 下都照常计算并 surface
|
|
40
|
+
// (observe 量的是线上部署版,与本机能否 probe 源、源是否漂移无关)。只 surface 确诊盲区(marker==='gap')。
|
|
41
|
+
const gap = deriveProductionGap(record);
|
|
38
42
|
return {
|
|
39
43
|
id: record.id,
|
|
40
44
|
name: record.name,
|
|
@@ -47,6 +51,10 @@ export function buildManagedListRow(record, probe) {
|
|
|
47
51
|
...(latest?.verdict ? { latestVerdict: latest.verdict } : {}),
|
|
48
52
|
...(latest?.comparability ? { comparability: latest.comparability } : {}),
|
|
49
53
|
...(latest?.recordedAt ? { recordedAt: latest.recordedAt } : {}),
|
|
54
|
+
...(override ? { override } : {}),
|
|
55
|
+
...(gap.marker === 'gap' && gap.latest
|
|
56
|
+
? { productionGap: { healthBand: gap.latest.healthBand, confidence: gap.latest.confidence, gapByType: gap.latest.gapByType } }
|
|
57
|
+
: {}),
|
|
50
58
|
currentEvidenceCount,
|
|
51
59
|
totalEvidenceCount: record.evidence.length,
|
|
52
60
|
distributionCount: record.distribution.length,
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* observe → managed 反哺(#235)。一次 `omk observe` 跑完,把每个 skill 的生产健康(盲区率 / 严重度加权 /
|
|
3
|
+
* 统计功效 / 盲区类型计数)落成一条 `ManagedObservation` 追加进**已纳管**的同名 skill 记录,供读时派生
|
|
4
|
+
* production-gap marker(`deriveProductionGap`)与「建议补样本」提示。写入仿 `evidence.ts` 的 `recordEvalEvidence`。
|
|
5
|
+
*
|
|
6
|
+
* 三条与 eval 写证据对齐的取舍:
|
|
7
|
+
* - **触发**:observe 完成自动写,但只写**已存在**的受管记录(install 是显式 opt-in,未纳管 skill 永不被
|
|
8
|
+
* 凭空建记录,零副作用惊吓);CLI 另给 `--no-feedback` 关。
|
|
9
|
+
* - **去重**:append-only + 按 `reportId` 去重(observe 无 contentHash,一份报告对一条观测;新窗口是新报告
|
|
10
|
+
* → 新条目 = 时间序)。
|
|
11
|
+
* - **匹配**:按 skill **名** + `kind==='skill'` 绑 —— observe 报告只带名、无 contentHash。observe 的 skillName
|
|
12
|
+
* 是 trace 调用名(已 `normalizeSkillName` 去插件前缀),`record.name` 是 install 名,约定相等;skill 的
|
|
13
|
+
* frontmatter `name:` ≠ 目录名时静默不匹配(fail-safe,见 spec §7 已知局限)。
|
|
14
|
+
*
|
|
15
|
+
* 版本无关:观测的是线上**正在跑那一版**(无 skill contentHash),故产读时 marker、**绝不翻 stale**
|
|
16
|
+
* (§6.1 只有内容漂移翻 stale;observe 是信号源不是受控 eval)。
|
|
17
|
+
*/
|
|
18
|
+
import type { ManagedObservation } from '../types/index.js';
|
|
19
|
+
/** 结构化最小入参(仿 `version-scores.ts` 的 `ReportScoreView`):observe CLI 侧从 `SkillHealthReport` 抽出
|
|
20
|
+
* 这几样传入,`managed/` 不 import `observability/`(避免跨支柱反向依赖)。`healthBand` 由 CLI 用 observability
|
|
21
|
+
* 自己的 `healthBandOf` 算好传入 —— 阈值单一来源,managed 不复制阈值、不伪造 observe 不出的 per-skill band。 */
|
|
22
|
+
export interface ObservedSkillHealthView {
|
|
23
|
+
skillName: string;
|
|
24
|
+
segmentCount: number;
|
|
25
|
+
gapRate: number;
|
|
26
|
+
weightedGapRate: number;
|
|
27
|
+
confidence: 'high' | 'low' | 'underpowered';
|
|
28
|
+
healthBand: 'green' | 'yellow' | 'red';
|
|
29
|
+
gapByType: {
|
|
30
|
+
failed_search: number;
|
|
31
|
+
explicit_marker: number;
|
|
32
|
+
hedging: number;
|
|
33
|
+
repeated_failure: number;
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
export interface ObserveReportView {
|
|
37
|
+
/** observe-health 报告 id —— 观测去重主键。 */
|
|
38
|
+
reportId: string;
|
|
39
|
+
/** 被观测**流量窗口的结束时刻**(report.meta.timeRange.to,CLI 侧 buildObserveReportView 取,空则退
|
|
40
|
+
* generatedAt)——供 deriveProductionGap 的 latest-wins,不是报告生成的「此刻」(generatedAt 恒约等于
|
|
41
|
+
* now,拿它比会让所有观测一样新)。无版本闸门:观测是版本无关的生产信号(见 ManagedObservation.observedAt)。 */
|
|
42
|
+
observedAt: string;
|
|
43
|
+
skills: ObservedSkillHealthView[];
|
|
44
|
+
}
|
|
45
|
+
export interface RecordedObservation {
|
|
46
|
+
recordId: string;
|
|
47
|
+
name: string;
|
|
48
|
+
healthBand: 'green' | 'yellow' | 'red';
|
|
49
|
+
confidence: 'high' | 'low' | 'underpowered';
|
|
50
|
+
/** 该观测是否构成确诊生产盲区(red 且够力)—— CLI 据此选「盲区警示」还是「已记录」文案。 */
|
|
51
|
+
isProductionGap: boolean;
|
|
52
|
+
gapByType: ManagedObservation['gapByType'];
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* 驱动:对每个能按 name(限 kind==='skill')匹配到的**已纳管**记录追加一条生产健康观测。返回实际写入清单
|
|
56
|
+
* (供 CLI 提示)。无任何记录匹配 → 返回空(常见的非管理用户场景,静默无副作用)。
|
|
57
|
+
*/
|
|
58
|
+
export declare function recordObserveHealth(report: ObserveReportView, opts?: {
|
|
59
|
+
dir?: string;
|
|
60
|
+
}): RecordedObservation[];
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { loadAllManagedRecords, appendManagedObservation, managedDir, resolveManagedDir, isProductionGapObservation, } from './store.js';
|
|
2
|
+
/**
|
|
3
|
+
* 驱动:对每个能按 name(限 kind==='skill')匹配到的**已纳管**记录追加一条生产健康观测。返回实际写入清单
|
|
4
|
+
* (供 CLI 提示)。无任何记录匹配 → 返回空(常见的非管理用户场景,静默无副作用)。
|
|
5
|
+
*/
|
|
6
|
+
export function recordObserveHealth(report, opts = {}) {
|
|
7
|
+
const out = [];
|
|
8
|
+
if (report.skills.length === 0)
|
|
9
|
+
return out;
|
|
10
|
+
// 写回读方实际取记录的同一目录(project→global 同口径)。
|
|
11
|
+
const dir = resolveManagedDir(opts.dir ?? managedDir());
|
|
12
|
+
const records = loadAllManagedRecords(dir);
|
|
13
|
+
if (records.length === 0)
|
|
14
|
+
return out;
|
|
15
|
+
// 按 name 索引,仅 kind==='skill':同名跨 kind 的 prompt/agent 记录不接 observe 信号(observe 测的是 skill)。
|
|
16
|
+
// id = hash(kind,name) 保证每个 (skill,name) 至多一条记录,故 name→record 唯一,无需唯一性闸门。
|
|
17
|
+
const byName = new Map();
|
|
18
|
+
for (const r of records)
|
|
19
|
+
if (r.kind === 'skill')
|
|
20
|
+
byName.set(r.name, { id: r.id, name: r.name });
|
|
21
|
+
for (const s of report.skills) {
|
|
22
|
+
const rec = byName.get(s.skillName);
|
|
23
|
+
if (!rec)
|
|
24
|
+
continue;
|
|
25
|
+
const observation = {
|
|
26
|
+
observationKind: 'production-health',
|
|
27
|
+
reportId: report.reportId,
|
|
28
|
+
observedAt: report.observedAt,
|
|
29
|
+
gapRate: s.gapRate,
|
|
30
|
+
weightedGapRate: s.weightedGapRate,
|
|
31
|
+
confidence: s.confidence,
|
|
32
|
+
healthBand: s.healthBand,
|
|
33
|
+
segmentCount: s.segmentCount,
|
|
34
|
+
gapByType: s.gapByType,
|
|
35
|
+
};
|
|
36
|
+
const merged = appendManagedObservation(dir, rec.id, observation);
|
|
37
|
+
if (!merged)
|
|
38
|
+
continue;
|
|
39
|
+
out.push({
|
|
40
|
+
recordId: rec.id,
|
|
41
|
+
name: rec.name,
|
|
42
|
+
healthBand: s.healthBand,
|
|
43
|
+
confidence: s.confidence,
|
|
44
|
+
isProductionGap: isProductionGapObservation(observation),
|
|
45
|
+
gapByType: s.gapByType,
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
return out;
|
|
49
|
+
}
|
package/dist/managed/store.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { ArtifactKind, DeriveManagedStateInput, DerivedManagedState, ManagedArtifactRecord, ManagedArtifactSource, ManagedDecision, ManagedDistributionTarget, ManagedEvidenceRef } from '../types/index.js';
|
|
1
|
+
import type { ArtifactKind, DeriveManagedStateInput, DerivedManagedState, ManagedArtifactRecord, ManagedArtifactSource, ManagedDecision, ManagedDistributionTarget, ManagedEvidenceRef, ManagedObservation } from '../types/index.js';
|
|
2
2
|
/**
|
|
3
3
|
* 受管记录的 per-record 文件存储。一条记录一个 `.omk/managed/<id>.json`,镜像 report-store 的
|
|
4
4
|
* 成熟模式(原子 tmp+rename):每次 install 只碰自己那个文件,independent write、不丢别人、
|
|
@@ -15,6 +15,9 @@ export declare function recordPath(dir: string, id: string): string;
|
|
|
15
15
|
/** 稳定身份 = hash(kind, name)。源路径是可变属性、不进 id。kind 取自固定枚举(无 `|`),分隔可注入。 */
|
|
16
16
|
export declare function managedRecordId(kind: ArtifactKind, name: string): string;
|
|
17
17
|
export { hashArtifactSource, isDistributablePath, distributableCopyFilter } from '../inputs/content-hash.js';
|
|
18
|
+
/** git commit SHA 形态:7–64 位 hex。写入(evidence.ts)与读取校验共用同一判定,避免写读不对称
|
|
19
|
+
* (写时不校验、读时却要求 SHA → 自己写进去的值重载时被自己判脏)。 */
|
|
20
|
+
export declare function isShaLike(v: unknown): v is string;
|
|
18
21
|
export declare function loadManagedRecord(dir: string, id: string): ManagedArtifactRecord | null;
|
|
19
22
|
/** 读全部记录。项目目录空 → 兜底全局(镜像 observe inbox 的 project→global)。 */
|
|
20
23
|
export declare function loadAllManagedRecords(dir?: string): ManagedArtifactRecord[];
|
|
@@ -44,6 +47,13 @@ export declare function upsertManagedRecord(dir: string, record: ManagedArtifact
|
|
|
44
47
|
* null —— 这是"管理是 install 显式 opt-in"的体现:eval 绝不为未纳管的 skill 凭空建记录。
|
|
45
48
|
*/
|
|
46
49
|
export declare function appendManagedEvidence(dir: string, recordId: string, evidence: ManagedEvidenceRef): ManagedArtifactRecord | null;
|
|
50
|
+
/**
|
|
51
|
+
* 追加一条 observe 生产健康观测(append-only,#235)。同 evidence 不走 upsert:`mergeManagedRecord` 保留旧
|
|
52
|
+
* observations、丢弃 next 的(install 只写事实)。按 `reportId` 去重——observe 无 contentHash,一份 observe 报告
|
|
53
|
+
* 对一条观测;同报告重跑不堆重复,不同报告(新窗口)是新 reportId → 新条目。记录不存在(未 install / 名字不
|
|
54
|
+
* 匹配)返回 null —— 与 evidence 同口径:observe 绝不为未纳管 skill 凭空建记录(管理是 install 显式 opt-in)。
|
|
55
|
+
*/
|
|
56
|
+
export declare function appendManagedObservation(dir: string, recordId: string, observation: ManagedObservation): ManagedArtifactRecord | null;
|
|
47
57
|
/**
|
|
48
58
|
* 把受管记录的 drift 基线(contentHash)重锚到新源内容哈,保留 evidence / decisions / source /
|
|
49
59
|
* distribution 全不动。用于 `omk evolve` 把胜出版本写回 source 后,让记录跟上实际内容——否则记录的
|
|
@@ -65,6 +75,13 @@ export declare function rebaselineManagedContentHash(dir: string, recordId: stri
|
|
|
65
75
|
* (数组追加序 = 事件序)。
|
|
66
76
|
*/
|
|
67
77
|
export declare function isCurrentlyPromoted(record: ManagedArtifactRecord): boolean;
|
|
78
|
+
/**
|
|
79
|
+
* 当前 promoted 版本是否经 override(--force)采用 —— 返回该 override(verdict + 被绕过的门),否则 undefined。
|
|
80
|
+
* 仅当前内容最近一条决定是 promote 且带 override 才有值;rollback 之后(已撤销接受)返回 undefined。供 list /
|
|
81
|
+
* Studio **读时审计**:从总览一眼看出哪些当前采用是越门来的。override 的**写**仍只在 CLI(`promote --force`),
|
|
82
|
+
* Studio 不执行——见 evidence-gated-management.md §9(#238)。
|
|
83
|
+
*/
|
|
84
|
+
export declare function currentPromoteOverride(record: ManagedArtifactRecord): ManagedDecision['override'];
|
|
68
85
|
/**
|
|
69
86
|
* 追加一条人工管理决定(append-only,promote/reject/rollback 走此路)。与 evidence 同样**不能走 upsert**
|
|
70
87
|
* (`mergeManagedRecord` 刻意保留旧 decisions、丢弃 next.decisions),必须独立 load→push→原子重写。
|
|
@@ -103,3 +120,23 @@ export declare function recordManagedArtifact(record: ManagedArtifactRecord, opt
|
|
|
103
120
|
* 这正是 #203「证据必须跟 artifact 一起走」的读时保证。`'discovered'` 留给无分发的记录,此函数不产生。
|
|
104
121
|
*/
|
|
105
122
|
export declare function deriveManagedState(input: DeriveManagedStateInput): DerivedManagedState;
|
|
123
|
+
/** 一条观测是否构成「确诊生产盲区」:healthBand 红 **且** 统计够力。underpowered(数据不足)是 §6.1 的
|
|
124
|
+
* unknown —— 既不放行也不拦截,不当确诊盲区;yellow / green 也不算(留信息展示,避免低样本噪声里过度报警)。 */
|
|
125
|
+
export declare function isProductionGapObservation(obs: ManagedObservation): boolean;
|
|
126
|
+
export interface ProductionGapState {
|
|
127
|
+
/** `gap` = 确诊生产盲区(red + 够力);`underpowered` = 线上数据不足、不可知;`none` = 无观测 / 最新观测非红。 */
|
|
128
|
+
marker: 'none' | 'gap' | 'underpowered';
|
|
129
|
+
latest?: ManagedObservation;
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* observe 生产健康观测的**读时 marker**(#235)。取**最新一条**观测(latest-wins,按 observedAt = 被观测
|
|
133
|
+
* 流量窗口结束时刻)分类:red + 够力 → `gap`;underpowered → `underpowered`(数据不足);其余 → `none`。
|
|
134
|
+
*
|
|
135
|
+
* **版本无关的生产信号**:observe 量的是线上**部署版**的行为,而记录里没有可靠的「源码版 ↔ 部署版」时间锚
|
|
136
|
+
* (`evolve` 改写源、`rebaselineManagedContentHash` 只换 `contentHash`,不碰 distribution/installedAt;部署副本
|
|
137
|
+
* 仍是旧的)——所以这里**不按源码版归因、也不臆造版本闸门**(那只会给假精度、还会在源码 bump 后压掉仍然有效的
|
|
138
|
+
* 真盲区)。marker 反映的是部署版的近况,可能滞后于当前源码版,UI / spec 据此老实标注。
|
|
139
|
+
*
|
|
140
|
+
* **绝不翻 stale / 不动生命周期枚举**(§6.1:只有内容漂移翻 stale,observe 是信号源不是受控 eval)。
|
|
141
|
+
*/
|
|
142
|
+
export declare function deriveProductionGap(record: ManagedArtifactRecord): ProductionGapState;
|