@stalfh233/omc-cli 0.0.0-stage → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +304 -3
  3. package/coverage/m1-coverage-manifest-v1.json +1278 -0
  4. package/dist/approval-token.js +102 -0
  5. package/dist/args.js +25 -0
  6. package/dist/artifacts.js +119 -0
  7. package/dist/bench/call-face-eval.js +256 -0
  8. package/dist/bench/context-attribution.js +151 -0
  9. package/dist/bench/discovery-cost-eval.js +230 -0
  10. package/dist/bench/driver.js +81 -0
  11. package/dist/bench/evals.js +236 -0
  12. package/dist/bench/fake-http-server.js +65 -0
  13. package/dist/bench/instrument.js +87 -0
  14. package/dist/bench/intent-face-eval.js +343 -0
  15. package/dist/bench/run.js +166 -0
  16. package/dist/bench/scenario.js +343 -0
  17. package/dist/bench/types.js +76 -0
  18. package/dist/bi-wire.js +41 -0
  19. package/dist/bizservice-config.js +581 -0
  20. package/dist/call.js +153 -0
  21. package/dist/capability-absences.js +23 -0
  22. package/dist/capability-overview.js +492 -0
  23. package/dist/capability-shape.js +154 -0
  24. package/dist/cli-contract.js +70 -0
  25. package/dist/cli-output.js +73 -0
  26. package/dist/cli.js +1418 -0
  27. package/dist/code-rules.js +69 -0
  28. package/dist/command-transport.js +155 -0
  29. package/dist/config-store.js +195 -0
  30. package/dist/context.js +21 -0
  31. package/dist/contract-consistency.js +66 -0
  32. package/dist/contract-resources.js +62 -0
  33. package/dist/coverage-consistency.js +62 -0
  34. package/dist/coverage-registry.js +76 -0
  35. package/dist/coverage.js +130 -0
  36. package/dist/data-list-filter.js +114 -0
  37. package/dist/discovery.js +390 -0
  38. package/dist/endpoints.js +154 -0
  39. package/dist/environment-policy.js +26 -0
  40. package/dist/execution-metadata.js +1092 -0
  41. package/dist/fake/app.js +45 -0
  42. package/dist/fake/b2-registration.js +594 -0
  43. package/dist/fake/businessrule.js +242 -0
  44. package/dist/fake/datarule.js +72 -0
  45. package/dist/fake/dictionary.js +108 -0
  46. package/dist/fake/environment.js +30 -0
  47. package/dist/fake/field.js +167 -0
  48. package/dist/fake/form.js +101 -0
  49. package/dist/fake/index.js +121 -0
  50. package/dist/fake/list-view.js +289 -0
  51. package/dist/fake/model.js +136 -0
  52. package/dist/fake/online-js.js +18 -0
  53. package/dist/fake/report.js +253 -0
  54. package/dist/fake/routes.js +47 -0
  55. package/dist/fake/rule-lifecycle.js +37 -0
  56. package/dist/fake/runtime-data.js +367 -0
  57. package/dist/fake/state.js +67 -0
  58. package/dist/fake/workflow.js +364 -0
  59. package/dist/field-change.js +200 -0
  60. package/dist/field-families.js +896 -0
  61. package/dist/form-layout.js +111 -0
  62. package/dist/form-support.js +821 -0
  63. package/dist/goal-routes.js +468 -0
  64. package/dist/governed-execution.js +87 -0
  65. package/dist/human-summary.js +212 -0
  66. package/dist/identity.js +62 -0
  67. package/dist/intent/baseline.js +57 -0
  68. package/dist/intent/capabilities/bizservice.js +274 -0
  69. package/dist/intent/capabilities/businessrule.js +493 -0
  70. package/dist/intent/capabilities/datarule.js +187 -0
  71. package/dist/intent/capabilities/field.js +545 -0
  72. package/dist/intent/capabilities/form.js +136 -0
  73. package/dist/intent/capabilities/index.js +64 -0
  74. package/dist/intent/capabilities/listview.js +157 -0
  75. package/dist/intent/capabilities/model.js +100 -0
  76. package/dist/intent/capabilities/onlinejs.js +108 -0
  77. package/dist/intent/capabilities/report.js +355 -0
  78. package/dist/intent/capabilities/workflow.js +458 -0
  79. package/dist/intent/capability.js +6 -0
  80. package/dist/intent/cli.js +91 -0
  81. package/dist/intent/compare.js +56 -0
  82. package/dist/intent/compiler.js +79 -0
  83. package/dist/intent/dsl.js +129 -0
  84. package/dist/intent/plan-file.js +63 -0
  85. package/dist/intent/readback.js +65 -0
  86. package/dist/intent/schema.js +158 -0
  87. package/dist/intent/validation.js +30 -0
  88. package/dist/intent/yaml.js +315 -0
  89. package/dist/json-column.js +68 -0
  90. package/dist/lanes/app-contract.js +95 -0
  91. package/dist/lanes/app-coverage.js +16 -0
  92. package/dist/lanes/app.js +174 -0
  93. package/dist/lanes/apply-changes.js +231 -0
  94. package/dist/lanes/b2-registration-contract.js +292 -0
  95. package/dist/lanes/b2-registration-coverage.js +48 -0
  96. package/dist/lanes/b2-registration.js +1187 -0
  97. package/dist/lanes/businessrule-contract.js +232 -0
  98. package/dist/lanes/businessrule-coverage.js +16 -0
  99. package/dist/lanes/businessrule.js +221 -0
  100. package/dist/lanes/contract-support.js +65 -0
  101. package/dist/lanes/coverage-declaration.js +9 -0
  102. package/dist/lanes/datarule-contract.js +155 -0
  103. package/dist/lanes/datarule-coverage.js +19 -0
  104. package/dist/lanes/datarule-protocol.js +308 -0
  105. package/dist/lanes/datarule.js +818 -0
  106. package/dist/lanes/dictionary-contract.js +78 -0
  107. package/dist/lanes/dictionary-coverage.js +24 -0
  108. package/dist/lanes/dictionary.js +235 -0
  109. package/dist/lanes/environment-contract.js +70 -0
  110. package/dist/lanes/environment-coverage.js +14 -0
  111. package/dist/lanes/environment.js +163 -0
  112. package/dist/lanes/field-contract.js +223 -0
  113. package/dist/lanes/field-coverage.js +27 -0
  114. package/dist/lanes/field.js +374 -0
  115. package/dist/lanes/form-contract.js +97 -0
  116. package/dist/lanes/form-coverage.js +16 -0
  117. package/dist/lanes/form.js +185 -0
  118. package/dist/lanes/lane-ids.js +34 -0
  119. package/dist/lanes/list-view-contract.js +175 -0
  120. package/dist/lanes/list-view-coverage.js +20 -0
  121. package/dist/lanes/list-view-shapes.js +1207 -0
  122. package/dist/lanes/list-view.js +578 -0
  123. package/dist/lanes/meta-contract.js +85 -0
  124. package/dist/lanes/meta.js +255 -0
  125. package/dist/lanes/model-contract.js +168 -0
  126. package/dist/lanes/model-coverage.js +20 -0
  127. package/dist/lanes/model.js +986 -0
  128. package/dist/lanes/online-js-contract.js +99 -0
  129. package/dist/lanes/online-js-coverage.js +28 -0
  130. package/dist/lanes/online-js.js +127 -0
  131. package/dist/lanes/report-contract.js +144 -0
  132. package/dist/lanes/report-coverage.js +21 -0
  133. package/dist/lanes/report.js +476 -0
  134. package/dist/lanes/rule-graph.js +1846 -0
  135. package/dist/lanes/rule-lifecycle-contract.js +92 -0
  136. package/dist/lanes/rule-lifecycle-coverage.js +20 -0
  137. package/dist/lanes/rule-lifecycle.js +176 -0
  138. package/dist/lanes/runtime-data-contract.js +264 -0
  139. package/dist/lanes/runtime-data-coverage.js +25 -0
  140. package/dist/lanes/runtime-data.js +1054 -0
  141. package/dist/lanes/workflow-contract.js +223 -0
  142. package/dist/lanes/workflow-coverage.js +40 -0
  143. package/dist/lanes/workflow.js +1813 -0
  144. package/dist/online-js-layout.js +58 -0
  145. package/dist/online-js-source.js +276 -0
  146. package/dist/package-tool.js +51 -0
  147. package/dist/package.js +73 -0
  148. package/dist/plan.js +73 -0
  149. package/dist/read.js +144 -0
  150. package/dist/redact.js +28 -0
  151. package/dist/rule-support.js +587 -0
  152. package/dist/runtime-support.js +134 -0
  153. package/dist/server.js +92 -0
  154. package/dist/session-manager.js +30 -0
  155. package/dist/session.js +149 -0
  156. package/dist/skills.js +112 -0
  157. package/dist/support.js +98 -0
  158. package/dist/tool-types.js +127 -0
  159. package/dist/tools.js +59 -0
  160. package/dist/usage-log.js +197 -0
  161. package/dist/wire.js +287 -0
  162. package/dist/workflow-support.js +99 -0
  163. package/dist/write-lease.js +26 -0
  164. package/dist/write-lock.js +109 -0
  165. package/dist/write.js +176 -0
  166. package/dist/zip.js +156 -0
  167. package/docs/tool-surface-map.md +42 -0
  168. package/package.json +71 -6
  169. package/skills/omc-acceptance-criteria.md +47 -0
  170. package/skills/omc-business-configuration.md +257 -0
  171. package/skills/omc-capabilities.md +197 -0
  172. package/skills/omc-capability-scouting.md +55 -0
  173. package/skills/omc-config-draft-review.md +171 -0
  174. package/skills/omc-five-piece-flow.md +35 -0
  175. package/skills/omc-glossary.md +86 -0
  176. package/skills/omc-refusals.md +110 -0
  177. package/skills/omc-requirement-analysis.md +161 -0
  178. package/skills/omc-requirement-vocabulary.md +51 -0
  179. package/skills/omc-start-here.md +82 -0
  180. package/skills/omc-tool-selection.md +119 -0
  181. package/skills/omc-write-hazards.md +87 -0
@@ -0,0 +1,87 @@
1
+ import { Buffer } from "node:buffer";
2
+ export class WireTap {
3
+ records = [];
4
+ wrap(http) {
5
+ return {
6
+ request: async (input) => {
7
+ const response = await http.request(input);
8
+ if (process.env.OMC_BENCH_TRACE)
9
+ console.error(`[trace] ${input.method ?? "GET"} ${input.path} q=${JSON.stringify(input.query ?? {})} b=${input.body ? JSON.stringify(input.body).slice(0, 120) : ""}`);
10
+ this.records.push({
11
+ path: input.path,
12
+ method: input.method,
13
+ status: response.status,
14
+ responseBytes: Buffer.byteLength(JSON.stringify(response.body) ?? "null"),
15
+ });
16
+ return response;
17
+ },
18
+ };
19
+ }
20
+ total() {
21
+ return this.records.length;
22
+ }
23
+ duplicateCalls() {
24
+ // Extra calls to a path already hit before (second-and-later occurrences).
25
+ const seen = new Set();
26
+ let duplicates = 0;
27
+ for (const record of this.records) {
28
+ if (seen.has(record.path))
29
+ duplicates += 1;
30
+ else
31
+ seen.add(record.path);
32
+ }
33
+ return duplicates;
34
+ }
35
+ pathStats() {
36
+ const byPath = new Map();
37
+ for (const record of this.records) {
38
+ const entry = byPath.get(record.path) ?? { count: 0, responseBytes: 0 };
39
+ entry.count += 1;
40
+ entry.responseBytes += record.responseBytes;
41
+ byPath.set(record.path, entry);
42
+ }
43
+ return [...byPath.entries()]
44
+ .map(([path, entry]) => ({ path, count: entry.count, responseBytes: entry.responseBytes }))
45
+ .sort((a, b) => b.count - a.count || a.path.localeCompare(b.path));
46
+ }
47
+ }
48
+ export class ToolTap {
49
+ records = [];
50
+ /** Wrap a registry so every handler invocation is counted. */
51
+ wrapRegistry(registry) {
52
+ return registry.map((tool) => ({
53
+ ...tool,
54
+ handler: async (ctx, args, self) => {
55
+ const started = performance.now();
56
+ const result = (await tool.handler(ctx, args, self));
57
+ const record = {
58
+ tool: tool.name,
59
+ status: String(result?.status ?? "ok"),
60
+ ok: result?.status === undefined || result.status === "ok",
61
+ durationMs: performance.now() - started,
62
+ responseBytes: Buffer.byteLength(JSON.stringify(result) ?? "null"),
63
+ };
64
+ this.records.push(record);
65
+ return result;
66
+ },
67
+ }));
68
+ }
69
+ responseBytesTotal() {
70
+ return this.records.reduce((sum, record) => sum + record.responseBytes, 0);
71
+ }
72
+ }
73
+ /** The tools-list payload size for a registry (name + description + inputSchema, JSON serialized). */
74
+ export function toolsListBytes(registry) {
75
+ const payload = {
76
+ tools: registry.map((tool) => ({
77
+ name: tool.name,
78
+ description: tool.description,
79
+ inputSchema: {
80
+ type: "object",
81
+ properties: Object.fromEntries(Object.entries(tool.inputSchema).map(([key, schema]) => [key, { description: schema.description }])),
82
+ },
83
+ })),
84
+ };
85
+ return Buffer.byteLength(JSON.stringify(payload));
86
+ }
87
+ //# sourceMappingURL=instrument.js.map
@@ -0,0 +1,343 @@
1
+ import { spawn } from "node:child_process";
2
+ import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
3
+ import { join } from "node:path";
4
+ import { fileURLToPath } from "node:url";
5
+ import { startFakeHttpServer } from "./fake-http-server.js";
6
+ /**
7
+ * Intent-face eval (Spec 107 acceptance): the SAME scripted reference-agent
8
+ * task set as the call-face eval, driven through the declarative face
9
+ * (intent validate/plan/apply/readback) with `omc call` as the escape
10
+ * hatch for uncovered capabilities. Metrics are measured with the SAME
11
+ * accounting (subprocess per invocation, bytes/4 token proxy) so the two
12
+ * faces compare apples-to-apples against eval/call-face-baseline.json.
13
+ */
14
+ const packageRoot = fileURLToPath(new URL("../..", import.meta.url));
15
+ const cli = join(packageRoot, "dist", "cli.js");
16
+ function ok(condition, message) {
17
+ if (!condition)
18
+ throw new Error(message);
19
+ }
20
+ async function makeAgent(home, metric) {
21
+ const env = { ...process.env, OMC_CONFIG_HOME: home };
22
+ const base = {
23
+ env: env,
24
+ async run(args) {
25
+ metric.cliInvocations += 1;
26
+ if (args[0] === "call")
27
+ metric.callInvocations += 1;
28
+ else if (args[0] === "intent")
29
+ metric.intentInvocations += 1;
30
+ else
31
+ metric.discoveryInvocations += 1;
32
+ const argsJson = args.find((token) => token.startsWith("{"));
33
+ metric.requestBytes += argsJson ? Buffer.byteLength(argsJson) : args.filter((token) => !token.startsWith("--")).join(" ").length;
34
+ return await new Promise((resolveRun, rejectRun) => {
35
+ const child = spawn(process.execPath, [cli, ...args], { cwd: home, env, stdio: ["pipe", "pipe", "pipe"] });
36
+ let stdout = "";
37
+ let stderr = "";
38
+ child.stdout.on("data", (chunk) => (stdout += chunk.toString("utf8")));
39
+ child.stderr.on("data", (chunk) => (stderr += chunk.toString("utf8")));
40
+ child.on("error", rejectRun);
41
+ child.on("close", (code) => {
42
+ metric.responseBytes += Buffer.byteLength(stdout);
43
+ let envelope;
44
+ try {
45
+ envelope = JSON.parse(stdout);
46
+ }
47
+ catch (error) {
48
+ rejectRun(new Error(`stdout is not a single JSON envelope (${String(code)}): ${stdout.slice(0, 200)}\nstderr=${stderr.slice(0, 200)} ${error.message}`));
49
+ return;
50
+ }
51
+ if (code === 13)
52
+ metric.expectedApprovalExits += 1;
53
+ else if (code !== 0)
54
+ metric.errorExits += 1;
55
+ resolveRun({ code, envelope });
56
+ });
57
+ });
58
+ },
59
+ };
60
+ const agent = {
61
+ ...base,
62
+ async write(args) {
63
+ const tool = args.__tool;
64
+ const payload = { ...args };
65
+ delete payload.__tool;
66
+ const argsFlag = JSON.stringify(payload);
67
+ const callArgs = ["call", tool, "--env", "eval", "--args", argsFlag];
68
+ const executed = await base.run(callArgs);
69
+ ok(executed.envelope.result?.status === "ok", `execute phase (${tool}): ${JSON.stringify(executed.envelope).slice(0, 260)}`);
70
+ return executed.envelope.result;
71
+ },
72
+ async read(tool, args) {
73
+ const result = await base.run(["call", tool, "--env", "eval", "--args", JSON.stringify(args)]);
74
+ ok(result.envelope.ok === true, `read ${tool}: ${JSON.stringify(result.envelope).slice(0, 220)}`);
75
+ return result.envelope.result;
76
+ },
77
+ async discovery(kind, value) {
78
+ const result = await base.run(value === undefined ? [kind, ...(kind === "commands" ? ["--json"] : [])] : [kind, value]);
79
+ ok(result.envelope.ok === true, `discovery ${kind}: ${JSON.stringify(result.envelope).slice(0, 220)}`);
80
+ return result.envelope.result;
81
+ },
82
+ writeIntent(name, yaml) {
83
+ // The agent authors the declarative file itself — its bytes are agent
84
+ // output, so they count into the token proxy exactly like composed
85
+ // --args payloads did on the call face.
86
+ metric.requestBytes += Buffer.byteLength(yaml);
87
+ const file = join(home, name);
88
+ writeFileSync(file, yaml);
89
+ return file;
90
+ },
91
+ async intentValidate(file) {
92
+ const result = await base.run(["intent", "validate", file]);
93
+ return result.envelope.result;
94
+ },
95
+ async intentPlan(file) {
96
+ // The documented lean route (intent/cli.ts lifecycle): `--plan-out` writes
97
+ // the byte-identical ~11 KB plan to a file in the config root and returns
98
+ // only the decision fields (~1 KB). Inlining the plan charged the agent's
99
+ // window on every later turn; the reference agent must use the cheap face.
100
+ const planOut = join(home, `plan-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.json`);
101
+ const result = await base.run(["intent", "plan", file, "--env", "eval", "--plan-out", planOut]);
102
+ ok(result.envelope.ok === true, `intent plan: ${JSON.stringify(result.envelope).slice(0, 300)}`);
103
+ return { plan: result.envelope.result, planFile: planOut };
104
+ },
105
+ async intentApply(planFile) {
106
+ const result = await base.run(["intent", "apply", planFile]);
107
+ return result.envelope.result;
108
+ },
109
+ async intentReadback(file) {
110
+ const result = await base.run(["intent", "readback", file, "--env", "eval"]);
111
+ ok(result.envelope.ok === true, `intent readback: ${JSON.stringify(result.envelope).slice(0, 300)}`);
112
+ return result.envelope.result;
113
+ },
114
+ counts: () => metric,
115
+ };
116
+ return agent;
117
+ }
118
+ function freshHome(url) {
119
+ const home = mkdtempSync(join(packageRoot, ".intent-face-eval-"));
120
+ mkdirSync(home, { recursive: true });
121
+ writeFileSync(join(home, "omc.config.json"), JSON.stringify({ environments: { eval: { baseUrl: url, applicationCode: "test", isDefault: true, writable: true } } }));
122
+ writeFileSync(join(home, "credentials"), JSON.stringify({ eval: { username: "tester", password: "eval-face-password" } }));
123
+ return home;
124
+ }
125
+ const LADDER_INTENT = `intentVersion: 1
126
+ models:
127
+ - code: eval_m1
128
+ name: 评测建模
129
+ fields:
130
+ - code: f_title
131
+ name: 标题
132
+ type: SHORT_TEXT
133
+ - code: f_amount
134
+ name: 金额
135
+ type: NUMERICAL
136
+ options:
137
+ scale: 3
138
+ - code: f_when
139
+ name: 日期
140
+ type: DATE
141
+ form:
142
+ publish: true
143
+ `;
144
+ export const INTENT_FACE_TASKS = [
145
+ {
146
+ id: "usable-model-ladder",
147
+ description: "model + 3 fields + field publish + form draft/publish + verified readback (Spec 107 first slice, via the intent face)",
148
+ async run(agent) {
149
+ const inventory = await agent.discovery("commands");
150
+ ok(inventory.intent?.intentVersion === 1, "commands inventory lacks the intent face block");
151
+ const file = agent.writeIntent("intent.yaml", LADDER_INTENT);
152
+ const validated = await agent.intentValidate(file);
153
+ ok(validated.valid === true, `validate: ${JSON.stringify(validated).slice(0, 220)}`);
154
+ const { plan, planFile } = await agent.intentPlan(file);
155
+ ok(plan.changesCount === 7 && plan.riskClass === "lifecycle", `plan changes=${JSON.stringify(plan.changeSummary)}`);
156
+ const applied = await agent.intentApply(planFile);
157
+ ok(applied.status === "ok" && applied.applied === 7, `apply: ${JSON.stringify(applied).slice(0, 260)}`);
158
+ const readback = await agent.intentReadback(file);
159
+ ok(readback.equivalent === true && readback.differences.length === 0, `readback: ${JSON.stringify(readback.differences).slice(0, 260)}`);
160
+ },
161
+ },
162
+ {
163
+ id: "rule-backfill-ladder",
164
+ description: "two models via intent + DYNAMIC backfill rule + enable + trigger row + verified target readback (mixed face: intent + escape hatch)",
165
+ async run(agent, server) {
166
+ const modelsIntent = agent.writeIntent("models.yaml", `intentVersion: 1
167
+ models:
168
+ - code: eval_rs
169
+ name: 回填源
170
+ - code: eval_rt
171
+ name: 回填目标
172
+ `);
173
+ await agent.intentValidate(modelsIntent);
174
+ const { plan, planFile } = await agent.intentPlan(modelsIntent);
175
+ const applied = await agent.intentApply(planFile);
176
+ ok(applied.status === "ok" && applied.applied === 2, `models apply: ${JSON.stringify(applied).slice(0, 260)}`);
177
+ // Uncovered capability: the businessrule graph goes through the escape
178
+ // hatch (omc call) under the same governance.
179
+ await agent.discovery("contract", "B1-businessrule");
180
+ server.fake.seedDataRow("eval_rt", { latestPeriod: "" }, "eval_rt_row");
181
+ await agent.write({
182
+ __tool: "rule.save",
183
+ schemaCode: "eval_rs",
184
+ ruleCode: "Create",
185
+ name: "回填",
186
+ nodeGraph: {
187
+ nodes: [
188
+ { nodeCode: "Start", nodeType: "START" },
189
+ {
190
+ nodeCode: "UPDATE_1",
191
+ nodeType: "UPDATE",
192
+ targetObjectCode: "eval_rt",
193
+ inputParam: "_input",
194
+ filterCondition: { mainConditions: [[{ targetSchemaDataItem: "id", ruleConditionType: "EQ", ruleValueType: "DYNAMIC", currentSchemaDataItem: "project" }]], sheetConditions: [] },
195
+ dataActions: [{ targetDataItem: "latestPeriod", ruleValueType: "DYNAMIC", value: "", ruleActionType: "EQUALS", currentDataItemValue: "declarePeriod" }],
196
+ },
197
+ { nodeCode: "End", nodeType: "END" },
198
+ ],
199
+ routes: [{ preNode: "Start", postNode: "UPDATE_1" }, { preNode: "UPDATE_1", postNode: "End" }],
200
+ },
201
+ });
202
+ await agent.write({ __tool: "rule.enable", schemaCode: "eval_rs", ruleCode: "Create" });
203
+ await agent.write({ __tool: "data.save", schemaCode: "eval_rs", formData: { declarePeriod: "202612", project: { id: "eval_rt_row" } } });
204
+ const target = await agent.read("data.load", { schemaCode: "eval_rt", bizObjectId: "eval_rt_row" });
205
+ ok(target.data?.latestPeriod === "202612", `backfill=${String(target.data?.latestPeriod)}`);
206
+ },
207
+ },
208
+ {
209
+ id: "input-error-recovery",
210
+ description: "bad intent -> validate exit 10 with source location -> fix -> plan/apply -> idempotent re-plan/re-apply with zero duplicate wires",
211
+ async run(agent, server) {
212
+ const bad = agent.writeIntent("intent.yaml", `intentVersion: 1
213
+ models:
214
+ - code: eval_idem
215
+ name: 幂等模型
216
+ fields:
217
+ - code: f_bad
218
+ name: 坏类型
219
+ type: HOLOGRAM
220
+ `);
221
+ const rejected = await agent.run(["intent", "validate", bad]);
222
+ ok(rejected.code === 10 && rejected.envelope.error?.code === "invalid-input", `expected invalid-input exit 10, got ${String(rejected.code)}`);
223
+ const first = rejected.envelope.result.errors[0];
224
+ ok(first.code === "intent-field-type-unknown" && typeof first.line === "number", `located error: ${JSON.stringify(first).slice(0, 200)}`);
225
+ const fixed = agent.writeIntent("intent.yaml", `intentVersion: 1
226
+ models:
227
+ - code: eval_idem
228
+ name: 幂等模型
229
+ fields:
230
+ - code: f_bad
231
+ name: 改正的字段
232
+ type: SHORT_TEXT
233
+ `);
234
+ const validated = await agent.intentValidate(fixed);
235
+ ok(validated.valid === true, `fixed validate: ${JSON.stringify(validated).slice(0, 200)}`);
236
+ const { plan, planFile } = await agent.intentPlan(fixed);
237
+ const applied = await agent.intentApply(planFile);
238
+ ok(applied.status === "ok", `apply: ${JSON.stringify(applied).slice(0, 260)}`);
239
+ const createWires = server.fake.bodiesFor("/api/api/app/bizmodels/create").length;
240
+ ok(createWires === 1, `expected exactly one create wire, got ${String(createWires)}`);
241
+ // Idempotent convergence: re-plan -> zero changes; re-apply -> no-op.
242
+ const recheck = await agent.intentPlan(fixed);
243
+ ok(recheck.plan.changesCount === 0, `re-plan changes: ${JSON.stringify(recheck.plan.changeSummary)}`);
244
+ const reapplied = await agent.intentApply(recheck.planFile);
245
+ ok(reapplied.status === "ok" && reapplied.applied === 0, `re-apply: ${JSON.stringify(reapplied).slice(0, 260)}`);
246
+ ok(server.fake.bodiesFor("/api/api/app/bizmodels/create").length === 1, "idempotent re-apply must not send a second create wire");
247
+ const readback = await agent.intentReadback(fixed);
248
+ ok(readback.equivalent === true, `readback: ${JSON.stringify(readback.differences).slice(0, 200)}`);
249
+ },
250
+ },
251
+ {
252
+ id: "read-exploration",
253
+ description: "discovery-driven read exploration: commands -> describe -> contract -> filtered data.list projection (escape hatch reads)",
254
+ async run(agent, server) {
255
+ const inventory = await agent.discovery("commands");
256
+ const listTool = inventory.tools.find((tool) => tool.name === "data.list");
257
+ ok(listTool !== undefined, "data.list not discoverable");
258
+ await agent.discovery("describe", "data.list");
259
+ server.fake.seedField("eval_ro", "name", 0);
260
+ server.fake.seedDataRow("eval_ro", { name: "命中", state: "DRAFT" }, "eval_ro_hit");
261
+ server.fake.seedDataRow("eval_ro", { name: "未命中", state: "DRAFT" }, "eval_ro_miss");
262
+ await agent.write({ __tool: "listview.configure", schemaCode: "eval_ro", fieldCodes: ["name"] });
263
+ const listed = await agent.read("data.list", { schemaCode: "eval_ro", filter: { op: "eq", field: "name", value: "命中" } });
264
+ const items = (listed.items ?? []);
265
+ ok(items.length === 1 && items[0]?.fields?.name === "命中", `filtered projection wrong: ${JSON.stringify(items).slice(0, 160)}`);
266
+ },
267
+ },
268
+ ];
269
+ export async function runIntentFaceEvals(options = {}) {
270
+ const tasks = options.only ? INTENT_FACE_TASKS.filter((task) => task.id === options.only) : INTENT_FACE_TASKS;
271
+ if (tasks.length === 0)
272
+ throw new Error(`no intent-face task matches ${options.only}`);
273
+ const results = [];
274
+ for (const task of tasks) {
275
+ const server = await startFakeHttpServer();
276
+ const home = freshHome(server.url);
277
+ const metric = { cliInvocations: 0, discoveryInvocations: 0, intentInvocations: 0, callInvocations: 0, approvalRounds: 0, errorExits: 0, expectedApprovalExits: 0, requestBytes: 0, responseBytes: 0 };
278
+ const startedAt = Date.now();
279
+ let businessWireCalls = 0;
280
+ let failure;
281
+ try {
282
+ const agent = await makeAgent(home, metric);
283
+ await task.run(agent, server);
284
+ }
285
+ catch (error) {
286
+ failure = error instanceof Error ? error.message : String(error);
287
+ }
288
+ finally {
289
+ businessWireCalls = server.fake.state.callLog.filter((entry) => !entry.path.startsWith("/user/") && !entry.path.startsWith("/api/public/") && !entry.path.startsWith("/api/api/organization/")).length;
290
+ await server.close();
291
+ rmSync(home, { recursive: true, force: true });
292
+ }
293
+ results.push({
294
+ id: task.id,
295
+ passed: failure === undefined,
296
+ ...(failure ? { error: failure } : {}),
297
+ cliInvocations: metric.cliInvocations,
298
+ discoveryInvocations: metric.discoveryInvocations,
299
+ intentInvocations: metric.intentInvocations,
300
+ callInvocations: metric.callInvocations,
301
+ approvalRounds: metric.approvalRounds,
302
+ errorExits: metric.errorExits,
303
+ expectedApprovalExits: metric.expectedApprovalExits,
304
+ requestBytes: metric.requestBytes,
305
+ responseBytes: metric.responseBytes,
306
+ estimatedTokens: Math.round((metric.requestBytes + metric.responseBytes) / 4),
307
+ businessWireCalls,
308
+ wallClockMs: Date.now() - startedAt,
309
+ });
310
+ }
311
+ const sum = (pick) => results.reduce((total, entry) => total + pick(entry), 0);
312
+ return {
313
+ schema: "omc-intent-face-eval-v1",
314
+ face: "intent",
315
+ generatedAt: new Date().toISOString(),
316
+ tasks: results,
317
+ aggregate: {
318
+ tasksTotal: results.length,
319
+ tasksPassed: results.filter((entry) => entry.passed).length,
320
+ successRate: results.length === 0 ? 0 : Math.round((results.filter((entry) => entry.passed).length / results.length) * 10000) / 10000,
321
+ cliInvocations: sum((entry) => entry.cliInvocations),
322
+ discoveryInvocations: sum((entry) => entry.discoveryInvocations),
323
+ intentInvocations: sum((entry) => entry.intentInvocations),
324
+ callInvocations: sum((entry) => entry.callInvocations),
325
+ approvalRounds: sum((entry) => entry.approvalRounds),
326
+ errorExits: sum((entry) => entry.errorExits),
327
+ expectedApprovalExits: sum((entry) => entry.expectedApprovalExits),
328
+ requestBytes: sum((entry) => entry.requestBytes),
329
+ responseBytes: sum((entry) => entry.responseBytes),
330
+ estimatedTokens: sum((entry) => entry.estimatedTokens),
331
+ businessWireCalls: sum((entry) => entry.businessWireCalls),
332
+ },
333
+ };
334
+ }
335
+ export function formatIntentFaceReport(report) {
336
+ const lines = [
337
+ `omc intent-face eval — ${report.aggregate.tasksPassed}/${report.aggregate.tasksTotal} tasks passed (successRate=${report.aggregate.successRate})`,
338
+ ...report.tasks.map((task) => ` ${task.passed ? "PASS" : "FAIL"} ${task.id} — cli=${task.cliInvocations} (intent=${task.intentInvocations}/call=${task.callInvocations}/discovery=${task.discoveryInvocations}) approvals=${task.approvalRounds} errExits=${task.errorExits} wires=${task.businessWireCalls} tokens≈${task.estimatedTokens} ${task.wallClockMs}ms${task.error ? `\n ${task.error}` : ""}`),
339
+ `aggregate: cli=${report.aggregate.cliInvocations} (intent=${report.aggregate.intentInvocations}/call=${report.aggregate.callInvocations}/discovery=${report.aggregate.discoveryInvocations}) approvals=${report.aggregate.approvalRounds} errExits=${report.aggregate.errorExits} wires=${report.aggregate.businessWireCalls} tokens≈${report.aggregate.estimatedTokens}`,
340
+ ];
341
+ return lines.join("\n");
342
+ }
343
+ //# sourceMappingURL=intent-face-eval.js.map
@@ -0,0 +1,166 @@
1
+ import { mkdirSync, readFileSync, writeFileSync, existsSync, renameSync } from "node:fs";
2
+ import { flattenMetrics, computeDelta } from "./types.js";
3
+ import { makeBenchHarness } from "./driver.js";
4
+ import { runScenario } from "./scenario.js";
5
+ import { runGoldenEvals } from "./evals.js";
6
+ import { toolsListBytes } from "./instrument.js";
7
+ import { wireReadCacheStats } from "../wire.js";
8
+ /** Where the scoreboard lives; `previous.json` keeps the last run for deltas. */
9
+ export function benchDir(root = process.cwd()) {
10
+ return `${root}/.omc/bench`.replaceAll("\\\\", "/");
11
+ }
12
+ export async function runBench(options) {
13
+ // Golden evals always replay against a fresh fake — the intent metric must
14
+ // stay comparable across scenario modes.
15
+ const evals = await runGoldenEvals();
16
+ const harness = makeBenchHarness(options.mode, options.liveEnvironment);
17
+ let scenarioSteps = [];
18
+ try {
19
+ const outcome = await runScenario(harness);
20
+ scenarioSteps = outcome.steps;
21
+ }
22
+ finally {
23
+ harness.close();
24
+ }
25
+ const wallClockMs = scenarioSteps.reduce((sum, step) => sum + step.durationMs, 0);
26
+ const stepsOk = scenarioSteps.filter((step) => step.ok).length;
27
+ const assertionsPassed = scenarioSteps.reduce((sum, step) => sum + step.assertionsPassed, 0);
28
+ const assertionsTotal = scenarioSteps.reduce((sum, step) => sum + step.assertionsPassed + step.assertionsFailed, 0);
29
+ const responseBytes = scenarioSteps.reduce((sum, step) => sum + step.responseBytes, 0);
30
+ const listBytes = toolsListBytes(harness.registry);
31
+ // Repeat readback targets: governed reads cache by target key; a second
32
+ // identical read returns `unchanged` without a wire call. Wire-layer
33
+ // design-metadata read-cache hits (ticket 16) carry the same signal.
34
+ const readbackCacheHits = wireReadCacheStats().hits;
35
+ const wirePaths = harness.wireTap.pathStats();
36
+ const dimensions = {
37
+ speed: {
38
+ wallClockMs,
39
+ stepCount: scenarioSteps.length,
40
+ slowestStepMs: scenarioSteps.reduce((max, step) => Math.max(max, step.durationMs), 0),
41
+ },
42
+ efficiency: {
43
+ wireCalls: harness.wireTap.total(),
44
+ uniqueWirePaths: wirePaths.length,
45
+ duplicateWireCalls: harness.wireTap.duplicateCalls(),
46
+ repeatReadbackTargets: wirePaths.filter((entry) => entry.count > 1 && /query\/list|form\/load|get_draft|get_published|business_rule\/list/.test(entry.path)).length,
47
+ readbackCacheHits,
48
+ wirePaths,
49
+ },
50
+ quality: {
51
+ stepsOk,
52
+ stepsTotal: scenarioSteps.length,
53
+ assertionsPassed,
54
+ assertionsTotal,
55
+ assertionRate: assertionsTotal === 0 ? 0 : assertionsPassed / assertionsTotal,
56
+ },
57
+ token: {
58
+ responseBytes,
59
+ toolsListBytes: listBytes,
60
+ estimatedTokens: Math.round((responseBytes + listBytes) / 4),
61
+ },
62
+ intent: {
63
+ cases: evals.total,
64
+ passed: evals.passed,
65
+ intentAccuracy: evals.total === 0 ? 0 : evals.passed / evals.total,
66
+ caseResults: evals.cases,
67
+ },
68
+ };
69
+ let report = {
70
+ schema: "omc-bench-report-v2",
71
+ generatedAt: new Date().toISOString(),
72
+ mode: options.mode,
73
+ scenario: "customer-project-declaration-v1",
74
+ steps: scenarioSteps,
75
+ dimensions,
76
+ delta: null,
77
+ };
78
+ const dir = options.persist === false ? "" : benchDir(options.root);
79
+ const reportPath = dir ? `${dir}/bench-report.json` : "";
80
+ let delta = null;
81
+ if (reportPath && existsSync(reportPath)) {
82
+ try {
83
+ const previous = JSON.parse(readFileSync(reportPath, "utf8"));
84
+ if (previous?.dimensions && previous.mode === report.mode)
85
+ delta = computeDelta(previous, report);
86
+ }
87
+ catch {
88
+ delta = null;
89
+ }
90
+ report = { ...report, delta };
91
+ mkdirSync(dir, { recursive: true });
92
+ renameSync(reportPath, `${dir}/previous.json`);
93
+ writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`);
94
+ }
95
+ else if (reportPath) {
96
+ report = { ...report, delta };
97
+ mkdirSync(dir, { recursive: true });
98
+ writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`);
99
+ }
100
+ return report;
101
+ }
102
+ /** Human-facing five-dimension table + delta, printed by `omc bench`. */
103
+ export function formatBenchReport(report) {
104
+ const lines = [];
105
+ const metrics = flattenMetrics(report.dimensions);
106
+ lines.push(`omc bench — ${report.mode} @ ${report.generatedAt}`);
107
+ lines.push("");
108
+ lines.push("维度 指标 数值 上次 Δ% 判定");
109
+ const rows = [];
110
+ for (const [key, label] of Object.entries(METRIC_LABELS)) {
111
+ if (metrics[key] === undefined)
112
+ continue;
113
+ const delta = report.delta?.[key];
114
+ const deltaText = delta ? `${delta.changePct >= 0 ? "+" : ""}${delta.changePct.toFixed(1)}` : "—";
115
+ const verdict = delta ? (delta.verdict === "same" ? "=" : delta.verdict === "improved" ? "↑" : "↓") : "—";
116
+ const previousText = delta ? String(delta.previous) : "—";
117
+ rows.push([key, label, ""]);
118
+ lines.push(`${label[0]} ${label.padEnd(30)} ${String(metrics[key]).padStart(10)} ${previousText.padStart(11)} ${deltaText.padStart(8)} ${verdict}`);
119
+ }
120
+ void rows;
121
+ const failedSteps = report.steps.filter((step) => !step.ok);
122
+ if (failedSteps.length > 0) {
123
+ lines.push("");
124
+ lines.push(`失败步骤 (${failedSteps.length}):`);
125
+ for (const step of failedSteps)
126
+ lines.push(` - ${step.id} [${step.tool}] ${step.error ?? ""}`);
127
+ }
128
+ const failedAssertions = report.steps.filter((step) => step.assertionsFailed > 0);
129
+ if (failedAssertions.length > 0) {
130
+ lines.push("");
131
+ lines.push("失败断言:");
132
+ for (const step of failedAssertions)
133
+ lines.push(` - ${step.id}: ${step.error ?? ""}`);
134
+ }
135
+ const failedCases = report.dimensions.intent.caseResults.filter((entry) => !entry.passed);
136
+ if (failedCases.length > 0) {
137
+ lines.push("");
138
+ lines.push("金轨迹未过:");
139
+ for (const entry of failedCases)
140
+ lines.push(` - ${entry.id}: ${entry.error ?? ""}`);
141
+ }
142
+ lines.push("");
143
+ lines.push(`报告: ${benchDir()}/bench-report.json(上一轮保留在 previous.json)`);
144
+ return lines.join("\n");
145
+ }
146
+ const METRIC_LABELS = {
147
+ "speed.wallClockMs": "速度-剧本墙钟ms",
148
+ "speed.slowestStepMs": "速度-最慢步骤ms",
149
+ "efficiency.wireCalls": "效率-wire调用总数",
150
+ "efficiency.uniqueWirePaths": "效率-去重端点数",
151
+ "efficiency.duplicateWireCalls": "效率-重复wire调用",
152
+ "efficiency.repeatReadbackTargets": "效率-重复读回目标",
153
+ "efficiency.readbackCacheHits": "效率-缓存命中",
154
+ "quality.stepsOk": "质量-通过步骤",
155
+ "quality.stepsTotal": "质量-步骤总数",
156
+ "quality.assertionsPassed": "质量-通过断言",
157
+ "quality.assertionsTotal": "质量-断言总数",
158
+ "quality.assertionRate": "质量-断言通过率",
159
+ "token.responseBytes": "token-响应字节",
160
+ "token.toolsListBytes": "token-tools/list字节",
161
+ "token.estimatedTokens": "token-估算token",
162
+ "intent.cases": "意图-金轨迹数",
163
+ "intent.passed": "意图-通过数",
164
+ "intent.intentAccuracy": "意图-准确率",
165
+ };
166
+ //# sourceMappingURL=run.js.map