@worca/app 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (187) hide show
  1. package/README.md +85 -6
  2. package/agents/clarify.meta.json +1 -0
  3. package/agents/memoryDefragmenter.meta.json +2 -1
  4. package/agents/reviewer.meta.json +60 -0
  5. package/agents/worca-cc-code-reviewer.md +33 -0
  6. package/agents/worca-cc-memory-defragmenter.md +5 -3
  7. package/agents/workspaceScanner.meta.json +1 -0
  8. package/package.json +14 -10
  9. package/scripts/git-diff.mjs +25 -0
  10. package/scripts/gitDiff.meta.json +18 -0
  11. package/scripts/js-inline.mjs +11 -0
  12. package/scripts/js.meta.json +22 -0
  13. package/scripts/py-inline.py +27 -0
  14. package/scripts/py.meta.json +22 -0
  15. package/scripts/shell.meta.json +24 -0
  16. package/skills/worca/SKILL.md +3 -2
  17. package/src/cli/models.mjs +247 -0
  18. package/src/cli/render.mjs +72 -4
  19. package/src/cli/schedule.mjs +494 -0
  20. package/src/cli/worca-cc.mjs +1001 -22
  21. package/src/core/agent-registry.mjs +75 -23
  22. package/src/core/agent-store.mjs +51 -2
  23. package/src/core/artifacts.mjs +73 -10
  24. package/src/core/ask/events.mjs +119 -1
  25. package/src/core/ask/limits.mjs +32 -4
  26. package/src/core/ask/mcp-stdio.mjs +12 -0
  27. package/src/core/ask/model-deps.mjs +126 -0
  28. package/src/core/ask/model-proposal.mjs +370 -0
  29. package/src/core/ask/models.mjs +12 -0
  30. package/src/core/ask/policy-deps.mjs +124 -0
  31. package/src/core/ask/policy-proposal.mjs +363 -0
  32. package/src/core/ask/prompt.mjs +74 -9
  33. package/src/core/ask/proposal.mjs +54 -5
  34. package/src/core/ask/schedule-deps.mjs +83 -0
  35. package/src/core/ask/schedule-spec.mjs +310 -0
  36. package/src/core/ask/script-deps.mjs +357 -0
  37. package/src/core/ask/source-deps.mjs +52 -0
  38. package/src/core/ask/source-spec.mjs +157 -0
  39. package/src/core/ask/spawn.mjs +1 -0
  40. package/src/core/ask/store.mjs +6 -3
  41. package/src/core/ask/tool-deps.mjs +4 -0
  42. package/src/core/ask/tools.mjs +657 -37
  43. package/src/core/ask/turn.mjs +109 -2
  44. package/src/core/ask-files.mjs +406 -0
  45. package/src/core/ask-forms.mjs +195 -0
  46. package/src/core/ask-projection.mjs +72 -0
  47. package/src/core/bridge/errors.mjs +84 -0
  48. package/src/core/bridge/provider-ops.mjs +281 -0
  49. package/src/core/bridge/providers/copilot.mjs +269 -0
  50. package/src/core/bridge/providers/endpoint.mjs +257 -0
  51. package/src/core/bridge/registry.mjs +88 -0
  52. package/src/core/bridge/semaphore.mjs +73 -0
  53. package/src/core/bridge/server.mjs +184 -0
  54. package/src/core/bridge/telemetry.mjs +53 -0
  55. package/src/core/bridge/translate/request.mjs +252 -0
  56. package/src/core/bridge/translate/response.mjs +82 -0
  57. package/src/core/bridge/translate/stream.mjs +242 -0
  58. package/src/core/bridge/upstream.mjs +209 -0
  59. package/src/core/chat/command-router.mjs +58 -4
  60. package/src/core/chat/notifier.mjs +14 -1
  61. package/src/core/chat/renderers.mjs +35 -0
  62. package/src/core/claude-runner.mjs +126 -19
  63. package/src/core/config.mjs +212 -31
  64. package/src/core/cost-budget.mjs +3 -2
  65. package/src/core/db.mjs +169 -15
  66. package/src/core/failure-policy.mjs +10 -0
  67. package/src/core/fs-browse.mjs +16 -4
  68. package/src/core/git-info.mjs +22 -0
  69. package/src/core/graph/builtin-workflows.mjs +3 -1
  70. package/src/core/graph/exec-io.mjs +71 -0
  71. package/src/core/graph/executor.mjs +139 -70
  72. package/src/core/graph/human-evidence.mjs +131 -0
  73. package/src/core/graph/python-probe.mjs +172 -0
  74. package/src/core/graph/registry-ports.mjs +10 -6
  75. package/src/core/graph/scheduler.mjs +39 -24
  76. package/src/core/graph/script-child.mjs +81 -0
  77. package/src/core/graph/script-runner.mjs +597 -0
  78. package/src/core/graph/worca_script.py +207 -0
  79. package/src/core/guardrail-store.mjs +16 -0
  80. package/src/core/human-backfill.mjs +108 -0
  81. package/src/core/human-rate.mjs +17 -0
  82. package/src/core/index-html.mjs +6 -2
  83. package/src/core/memory-defrag-model.mjs +112 -0
  84. package/src/core/memory-store.mjs +70 -18
  85. package/src/core/memory-sync.mjs +22 -11
  86. package/src/core/metrics/read.mjs +4 -1
  87. package/src/core/metrics/record.mjs +50 -2
  88. package/src/core/metrics/sync.mjs +6 -4
  89. package/src/core/model-env.mjs +149 -0
  90. package/src/core/model-test.mjs +14 -1
  91. package/src/core/notifications.mjs +128 -0
  92. package/src/core/onboarding.mjs +8 -2
  93. package/src/core/orchestrator.mjs +357 -26
  94. package/src/core/phases.mjs +95 -6
  95. package/src/core/plugin-api.mjs +24 -7
  96. package/src/core/plugin-manifest.mjs +184 -18
  97. package/src/core/plugin-models.mjs +1 -0
  98. package/src/core/plugin-script-cases.mjs +118 -0
  99. package/src/core/plugin-store.mjs +163 -17
  100. package/src/core/plugin-workflows.mjs +71 -17
  101. package/src/core/policy/cache.mjs +116 -0
  102. package/src/core/policy/effective.mjs +175 -0
  103. package/src/core/policy/gate.mjs +91 -0
  104. package/src/core/policy/local.mjs +145 -0
  105. package/src/core/policy/registry.mjs +330 -0
  106. package/src/core/policy/scope.mjs +61 -0
  107. package/src/core/policy/state.mjs +79 -0
  108. package/src/core/policy/sync.mjs +513 -0
  109. package/src/core/protocol.mjs +43 -0
  110. package/src/core/run-harness.mjs +421 -72
  111. package/src/core/scheduler.mjs +980 -0
  112. package/src/core/script-bench.mjs +628 -0
  113. package/src/core/script-registry.mjs +116 -0
  114. package/src/core/script-store.mjs +563 -0
  115. package/src/core/settings.mjs +489 -14
  116. package/src/core/stats.mjs +33 -2
  117. package/src/core/workflow-export.mjs +94 -3
  118. package/src/core/workflow-share.mjs +67 -18
  119. package/src/core/workflows.mjs +47 -11
  120. package/src/core/workspaces.mjs +18 -12
  121. package/src/shared/forms/answer.mjs +164 -0
  122. package/src/shared/forms/catalog.mjs +91 -0
  123. package/src/shared/forms/form-def.mjs +290 -0
  124. package/src/shared/forms/layout.mjs +67 -0
  125. package/src/shared/forms/paths.mjs +47 -0
  126. package/src/shared/forms/project.mjs +309 -0
  127. package/src/shared/forms/schema.mjs +205 -0
  128. package/src/shared/graph/agent-meta.mjs +55 -5
  129. package/src/shared/graph/constants.mjs +14 -2
  130. package/src/shared/graph/flow-layout.mjs +2 -1
  131. package/src/shared/graph/isomorphic.mjs +5 -3
  132. package/src/shared/graph/manifest.mjs +22 -13
  133. package/src/shared/graph/ports.mjs +45 -19
  134. package/src/shared/graph/script-cases.mjs +257 -0
  135. package/src/shared/graph/script-icons.mjs +46 -0
  136. package/src/shared/graph/script-infer.mjs +259 -0
  137. package/src/shared/graph/script-meta.mjs +408 -0
  138. package/src/shared/graph/script-templates.mjs +201 -0
  139. package/src/shared/graph/template.mjs +4 -4
  140. package/src/shared/graph/validate.mjs +89 -16
  141. package/src/shared/human-estimate.mjs +100 -0
  142. package/src/shared/schedule/recurrence.mjs +353 -0
  143. package/src/shared/team-metrics/aggregate.mjs +51 -11
  144. package/{scripts → tools}/install.mjs +3 -3
  145. package/ui/public/app.js +4390 -683
  146. package/ui/public/artifact-picker.mjs +189 -0
  147. package/ui/public/ask/dom.mjs +121 -0
  148. package/ui/public/ask/form-preview.mjs +55 -0
  149. package/ui/public/ask/form-renderer.mjs +250 -0
  150. package/ui/public/ask/registry.mjs +53 -0
  151. package/ui/public/ask/widgets-display.mjs +370 -0
  152. package/ui/public/ask/widgets-input.mjs +624 -0
  153. package/ui/public/ask/widgets-layout.mjs +90 -0
  154. package/ui/public/ask-panel.mjs +401 -27
  155. package/ui/public/ask-run-card.mjs +1 -1
  156. package/ui/public/bridge-view.mjs +694 -0
  157. package/ui/public/chat-settings-view.mjs +24 -0
  158. package/ui/public/code-editor.mjs +181 -0
  159. package/ui/public/getting-started.mjs +34 -7
  160. package/ui/public/graph/composer.mjs +138 -10
  161. package/ui/public/graph/inspector.mjs +61 -58
  162. package/ui/public/graph/palette.mjs +27 -9
  163. package/ui/public/graph/run-decor.mjs +34 -17
  164. package/ui/public/graph/run-hosts.mjs +7 -1
  165. package/ui/public/graph/save-dialog.mjs +3 -0
  166. package/ui/public/graph/view.mjs +23 -7
  167. package/ui/public/guardrails-view.mjs +15 -3
  168. package/ui/public/guide-spot.mjs +87 -9
  169. package/ui/public/index.html +480 -141
  170. package/ui/public/memory-view.mjs +22 -4
  171. package/ui/public/models-view.mjs +162 -17
  172. package/ui/public/node-tunables.mjs +33 -4
  173. package/ui/public/plugins-view.mjs +23 -1
  174. package/ui/public/results-view.mjs +4 -2
  175. package/ui/public/schedule-sheet.mjs +430 -0
  176. package/ui/public/schedules-view.mjs +432 -0
  177. package/ui/public/script-bench-view.mjs +1154 -0
  178. package/ui/public/script-forms.mjs +282 -0
  179. package/ui/public/script-wizard.mjs +529 -0
  180. package/ui/public/scripts-view.mjs +868 -0
  181. package/ui/public/stats-view.mjs +159 -52
  182. package/ui/public/style.css +1461 -44
  183. package/ui/public/team-metrics-surfaces.mjs +77 -16
  184. package/ui/public/team-metrics-view.mjs +68 -5
  185. package/ui/public/team-policy-view.mjs +1402 -0
  186. package/ui/public/ui-level.mjs +237 -0
  187. package/ui/server.mjs +1827 -73
@@ -0,0 +1,628 @@
1
+ // src/core/script-bench.mjs
2
+ // The script test bench (workbench spec §4). ONE rule holds it together: the
3
+ // bench never has its own idea of how a script runs. It builds a SYNTHETIC
4
+ // execution context and calls the very `runScriptExecution` a pipeline run
5
+ // calls, so "passes in the bench" and "works in a run" cannot drift.
6
+ //
7
+ // Everything a run would put in a pipeline dir lands under
8
+ // <worcaHome()>/bench/<benchId>/ instead (W11): `pipeline/` for outputs, the
9
+ // verdict and the envelope, `in/` for the hand-filled inputs, `cwd/` for the
10
+ // scratch folder. Results OUTLIVE the run so the output tabs can read them;
11
+ // one live folder per script key, and anything older than 24 h is swept at
12
+ // server start. No DB row, no History entry, no ledger, no cost (W15).
13
+ import { EventEmitter } from 'node:events';
14
+ import { randomUUID } from 'node:crypto';
15
+ import { execFileSync } from 'node:child_process';
16
+ import { join, resolve } from 'node:path';
17
+ import { mkdir, writeFile, readdir, rm, stat, open } from 'node:fs/promises';
18
+ import { worcaHome, listProjects } from './projects.mjs';
19
+ import { loadScriptRegistry } from './script-registry.mjs';
20
+ import { loadAgentRegistry } from './agent-registry.mjs';
21
+ import {
22
+ readScript, sourceFileFor, userScriptsDir, stripNullKeys, programText, SCRIPT_KEY_RE, assertKeyAllowed,
23
+ } from './script-store.mjs';
24
+ import { runScriptExecution } from './graph/script-runner.mjs';
25
+ import { allocateOutputs, allocateVerdict } from './graph/executor.mjs';
26
+ import {
27
+ effectiveScriptParams, paramValueError, validateScriptMetaV2, normalizeScriptMeta,
28
+ resolvePlatformValue, scriptNodeCtx, DEFAULT_TIMEOUT_MS, MIN_TIMEOUT_MS, MAX_TIMEOUT_MS,
29
+ } from '../shared/graph/script-meta.mjs';
30
+ import { casePortSet, evaluateExpect, MAX_CASE_INPUT_BYTES } from '../shared/graph/script-cases.mjs';
31
+ import { firedOutputs } from '../shared/graph/ports.mjs';
32
+ import { hasBlocking } from '../shared/graph/verdict.mjs';
33
+
34
+ /** W14: the bench's OWN cap — bench runs are not run executions and must never
35
+ * starve, or be starved by, the scheduler's pools. */
36
+ export const BENCH_MAX_PARALLEL = 2;
37
+ export const BENCH_INLINE_BYTES = 262144;
38
+ export const BENCH_SWEEP_MS = 24 * 60 * 60 * 1000;
39
+ const TAIL_LINES = 20;
40
+
41
+ const isObject = (v) => !!v && typeof v === 'object' && !Array.isArray(v);
42
+ const benchError = (message, code) => Object.assign(new Error(message), { code });
43
+
44
+ // Live bench bookkeeping for THIS process (W14 + W11). A bench is not durable:
45
+ // a restart forgets them, and the 24 h sweep reclaims whatever they left.
46
+ const activeIds = new Set();
47
+ const activeByKey = new Map();
48
+ const dirByKey = new Map();
49
+
50
+ /** `<worcaHome()>/bench` — colon-free names (the Windows filename rule). */
51
+ export function benchRoot(home = worcaHome()) {
52
+ return join(home, 'bench');
53
+ }
54
+
55
+ /**
56
+ * Remove bench folders older than `maxAgeMs`. Called once at server start.
57
+ * Never throws: a missing root is an empty sweep, and a folder a just-killed
58
+ * child still holds on Windows is simply left for the next boot.
59
+ * @returns {Promise<string[]>} the folder names removed
60
+ */
61
+ export async function sweepBenchDirs(root, { maxAgeMs = BENCH_SWEEP_MS, now = Date.now() } = {}) {
62
+ const removed = [];
63
+ let entries;
64
+ try { entries = await readdir(root, { withFileTypes: true }); } catch { return removed; }
65
+ for (const e of entries) {
66
+ if (!e.isDirectory()) continue;
67
+ const dir = join(root, e.name);
68
+ let mtimeMs;
69
+ try { mtimeMs = (await stat(dir)).mtimeMs; } catch { continue; }
70
+ if (now - mtimeMs < maxAgeMs) continue;
71
+ try {
72
+ await rm(dir, { recursive: true, force: true, maxRetries: 3 });
73
+ removed.push(e.name);
74
+ } catch { /* still held (Windows): the next boot gets it */ }
75
+ }
76
+ return removed;
77
+ }
78
+
79
+ /**
80
+ * Write the hand-filled inputs and return the bindings a run would have built.
81
+ * A port that is not listed — and a void port whose box is unchecked — is
82
+ * UNBOUND and absent from the envelope, exactly as in a run (W2).
83
+ * @param {string} inDir @param {{inputs: Array}} ports @param {object} inputs
84
+ * @returns {Promise<Record<string, {seq:number, type:string, path?:string}>>}
85
+ */
86
+ export async function writeBenchInputs(inDir, ports, inputs) {
87
+ const given = isObject(inputs) ? inputs : {};
88
+ const declared = new Map((ports?.inputs || [])
89
+ .filter((p) => p && !p.synthetic && p.id !== 'await').map((p) => [p.id, p]));
90
+ for (const id of Object.keys(given)) {
91
+ if (!declared.has(id)) throw benchError(`bench input "${id}" is not a declared input port`, 'BAD_REQUEST');
92
+ }
93
+ await mkdir(inDir, { recursive: true });
94
+ const bindings = {};
95
+ let seq = 0;
96
+ for (const port of declared.values()) {
97
+ const spec = given[port.id];
98
+ if (!isObject(spec)) continue;
99
+ if (port.type === 'void') {
100
+ if (spec.fired !== true) continue;
101
+ seq += 1;
102
+ bindings[port.id] = { seq, type: 'void' };
103
+ continue;
104
+ }
105
+ if (typeof spec.text !== 'string') throw benchError(`bench input "${port.id}" must be { text }`, 'BAD_REQUEST');
106
+ if (Buffer.byteLength(spec.text, 'utf8') > MAX_CASE_INPUT_BYTES) {
107
+ throw benchError(`bench input "${port.id}" is over ${MAX_CASE_INPUT_BYTES} bytes`, 'BAD_REQUEST');
108
+ }
109
+ if (port.type === 'json') {
110
+ try { JSON.parse(spec.text); } catch { throw benchError(`bench input "${port.id}" is not valid JSON`, 'BAD_REQUEST'); }
111
+ }
112
+ const path = join(inDir, `${port.id}.${port.type === 'json' ? 'json' : 'md'}`);
113
+ await writeFile(path, spec.text, 'utf8');
114
+ seq += 1;
115
+ bindings[port.id] = { seq, type: port.type, path };
116
+ }
117
+ return bindings;
118
+ }
119
+
120
+ /**
121
+ * The synthetic execution context (spec §4.1). Pure apart from the dirs it is
122
+ * given: ONE cycle, ordinal 1, the run's own allocator against the bench's
123
+ * `pipeline/`, and `bench: true` so the envelope and the shell env carry W12.
124
+ */
125
+ export function buildBenchCtx({ id, meta, resolved, ports, bindings, dirs, cwd, checkpointRef = null, signal = null, onEvent = () => {} }) {
126
+ const node = { id: 'bench', kind: 'script', key: meta.key };
127
+ const runCtx = { pipelineDir: dirs.pipeline, projectDir: cwd, baseName: 'bench' };
128
+ // W1: outputs, verdict and envelope live in the BENCH's folder, never in a
129
+ // project's plans/reviews store — so every port is allocated as store:'run'.
130
+ const benchPorts = {
131
+ inputs: ports.inputs || [],
132
+ outputs: (ports.outputs || []).filter(Boolean).map((p) => ({ ...p, store: 'run' })),
133
+ verdict: ports.verdict || null,
134
+ };
135
+ const executionId = 'x:bench:1';
136
+ // §4.1 step 2: the run-time facts come from the ONE builder resolveGraph and the
137
+ // resume path use (scriptNodeCtx), fed a synthetic node and the entry's two
138
+ // registry stamps — so a fresh run, a resumed run and a bench cannot drift.
139
+ // Only `mock` is overridden below (W13).
140
+ const sn = scriptNodeCtx(
141
+ { id: node.id, key: meta.key, config: { params: resolved.params, timeoutMs: resolved.timeoutMs } },
142
+ { ...meta, scriptPath: resolved.file, commandResolved: resolved.command },
143
+ );
144
+ return {
145
+ node,
146
+ executionId,
147
+ ordinal: 1,
148
+ cycle: 1,
149
+ pipelineDir: dirs.pipeline,
150
+ pipelineId: `bench-${String(id).slice(-8)}`,
151
+ projectDir: cwd,
152
+ runCtx,
153
+ runRoot: null,
154
+ workspace: undefined,
155
+ repos: null,
156
+ checkpointRef,
157
+ ports: benchPorts,
158
+ outputs: allocateOutputs({ node, ports: benchPorts, executionId, ordinal: 1, runCtx }),
159
+ verdict: allocateVerdict({ node, ports: benchPorts, ordinal: 1, runCtx }),
160
+ bindings,
161
+ trigger: { wireIds: [], freshPorts: Object.keys(bindings) }, // every bound input is fresh
162
+ script: {
163
+ meta,
164
+ runtime: sn.runtime,
165
+ file: sn.file,
166
+ command: sn.command,
167
+ params: sn.params,
168
+ timeoutMs: sn.timeoutMs,
169
+ mock: null, // W13
170
+ },
171
+ claudeOpts: { mock: false }, // W13: always run the program
172
+ bench: true, // W12
173
+ signal,
174
+ onEvent,
175
+ };
176
+ }
177
+
178
+ /** The HEAD of a project checkout, or null when it is not a git repo. */
179
+ function headRef(dir) {
180
+ try {
181
+ return execFileSync('git', ['rev-parse', 'HEAD'], { cwd: dir, stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' }).trim() || null;
182
+ } catch { return null; }
183
+ }
184
+
185
+ /**
186
+ * The bench's working directory (spec §4.1 step 5, W1) and the checkpoint a run
187
+ * would have recorded for it. `scratch` is the bench folder's own empty cwd/
188
+ * (created here); `project` is a REGISTERED project's checkout; `dir` is an
189
+ * arbitrary folder and belongs to the CLI alone (`worca script test --cwd <dir>`,
190
+ * spec §6) — the server never passes allowDirCwd, so a browser can never point a
191
+ * bench run at a folder of its choosing.
192
+ * @param {{kind?: string, projectKey?: string, dir?: string}|null|undefined} cwd
193
+ * @param {{dirs: {cwd: string}, projects?: Function, allowDirCwd?: boolean}} opts
194
+ * @returns {Promise<{cwd: string, checkpointRef: string|null}>}
195
+ * @throws {Error} code BAD_REQUEST
196
+ */
197
+ export async function resolveBenchCwd(cwd, { dirs, projects = listProjects, allowDirCwd = false } = {}) {
198
+ const kind = isObject(cwd) && cwd.kind !== undefined && cwd.kind !== null ? String(cwd.kind) : 'scratch';
199
+ if (kind === 'project') {
200
+ if (typeof cwd.projectKey !== 'string' || !cwd.projectKey) throw benchError('cwd: a project folder needs a projectKey', 'BAD_REQUEST');
201
+ const list = await projects();
202
+ const found = (list || []).find((p) => p && p.key === cwd.projectKey);
203
+ if (!found) throw benchError(`project "${cwd.projectKey}" is not registered`, 'BAD_REQUEST');
204
+ if (!found.exists) throw benchError(`project path does not exist or is not a directory: ${found.path}`, 'BAD_REQUEST');
205
+ return { cwd: found.path, checkpointRef: headRef(found.path) };
206
+ }
207
+ if (kind === 'dir') {
208
+ if (!allowDirCwd) throw benchError('cwd.kind "dir" is not accepted here — pick a registered project', 'BAD_REQUEST');
209
+ if (typeof cwd.dir !== 'string' || !cwd.dir.trim()) throw benchError('cwd: a dir folder needs a dir', 'BAD_REQUEST');
210
+ const dir = resolve(cwd.dir);
211
+ let ok = false;
212
+ try { ok = (await stat(dir)).isDirectory(); } catch { ok = false; }
213
+ if (!ok) throw benchError(`not a directory: ${dir}`, 'BAD_REQUEST');
214
+ return { cwd: dir, checkpointRef: headRef(dir) };
215
+ }
216
+ if (kind !== 'scratch') throw benchError(`cwd.kind must be scratch, project or dir (got "${kind}")`, 'BAD_REQUEST');
217
+ await mkdir(dirs.cwd, { recursive: true });
218
+ return { cwd: dirs.cwd, checkpointRef: null };
219
+ }
220
+
221
+ /** V22's sentences with the bench's node id, so a bad param fails exactly as a run would. */
222
+ function paramErrors(meta, values) {
223
+ const declared = Array.isArray(meta.params) ? meta.params : [];
224
+ const byId = new Map(declared.map((d) => [d.id, d]));
225
+ const errors = [];
226
+ for (const [id, value] of Object.entries(values)) {
227
+ const d = byId.get(id);
228
+ if (!d) {
229
+ errors.push(`script node 'bench' sets unknown param '${id}' — script "${meta.key}" declares `
230
+ + `${declared.length ? declared.map((x) => x.id).join(', ') : 'no params'}`);
231
+ continue;
232
+ }
233
+ const bad = paramValueError(d, value);
234
+ if (bad) errors.push(`script node 'bench' param '${id}': ${bad}`);
235
+ }
236
+ for (const d of declared) {
237
+ if (d.required && values[d.id] === undefined) errors.push(`script node 'bench' is missing required param '${d.id}'`);
238
+ }
239
+ return errors;
240
+ }
241
+
242
+ /** One case's folder set. A single run uses the bench folder itself (spec §4.1);
243
+ * Run all gives each case `case-<id>/` so its outputs survive the next case. */
244
+ function dirsFor(benchDir, caseId) {
245
+ const base = caseId ? join(benchDir, `case-${caseId}`) : benchDir;
246
+ return { base, pipeline: join(base, 'pipeline'), in: join(base, 'in'), cwd: join(base, 'cwd') };
247
+ }
248
+
249
+ /** Read every declared output back, capped inline (the rest via the output route). */
250
+ async function readOutputs(ports, allocated) {
251
+ const out = {};
252
+ for (const port of ports.outputs || []) {
253
+ if (!port) continue;
254
+ if (port.type === 'void') { out[port.id] = { type: 'void' }; continue; }
255
+ const path = allocated[port.id]?.path || null;
256
+ let text = '';
257
+ let bytes = 0;
258
+ let truncated = false;
259
+ if (path) {
260
+ // Only the inline head is read. readFile() would pull the WHOLE file into
261
+ // memory to keep 256 KiB of it — a script that writes a 600 MB log spiked
262
+ // the host by half a gigabyte, and past 2 GiB readFile throws
263
+ // ERR_FS_FILE_TOO_LARGE, which reported the output as empty (bytes: 0).
264
+ try {
265
+ const fh = await open(path, 'r');
266
+ try {
267
+ bytes = (await fh.stat()).size;
268
+ truncated = bytes > BENCH_INLINE_BYTES;
269
+ const want = Math.min(bytes, BENCH_INLINE_BYTES);
270
+ const buf = Buffer.alloc(want);
271
+ let got = 0;
272
+ while (got < want) {
273
+ const { bytesRead } = await fh.read(buf, got, want - got, got);
274
+ if (!bytesRead) break;
275
+ got += bytesRead;
276
+ }
277
+ // Stream mode holds back a code point the byte cap cut in half instead of
278
+ // decoding it as U+FFFD — the head must never show a character the file
279
+ // does not contain. ONLY when the bytes were cut, though (C35): on a
280
+ // COMPLETE output that same hold silently drops the last character of a
281
+ // file that is not valid UTF-8, and `ignoreBOM` keeps the output's own
282
+ // bytes, which the streamed route serves too.
283
+ text = new TextDecoder('utf-8', { ignoreBOM: true }).decode(buf.subarray(0, got), { stream: truncated });
284
+ } finally { await fh.close(); }
285
+ } catch { /* the port did not fire, or nothing wrote it */ }
286
+ }
287
+ out[port.id] = { type: port.type, path, bytes, text, truncated };
288
+ }
289
+ return out;
290
+ }
291
+
292
+ class Bench extends EventEmitter {
293
+ constructor(request, deps = {}) {
294
+ super();
295
+ this.id = `bench_${randomUUID()}`;
296
+ this.request = isObject(request) ? request : {};
297
+ this.deps = deps;
298
+ this.key = typeof this.request.key === 'string' ? this.request.key.trim() : '';
299
+ this.runner = typeof deps.runner === 'function' ? deps.runner : runScriptExecution;
300
+ this.platform = deps.platform || process.platform;
301
+ this.status = 'created';
302
+ this.abort = new AbortController();
303
+ this.benchDir = null;
304
+ }
305
+
306
+ getState() { return { benchId: this.id, key: this.key, status: this.status }; }
307
+
308
+ /** Abort the child (the runner's tree kill) and skip the remaining cases. */
309
+ stop() {
310
+ if (this.status === 'done' || this.status === 'error' || this.status === 'stopped') return;
311
+ this.status = 'stopped';
312
+ try { this.abort.abort(); } catch { /* already aborted */ }
313
+ }
314
+
315
+ _emit(type, payload) { this.emit(type, { benchId: this.id, ...payload }); }
316
+
317
+ /** NEVER throws: exactly one terminal scriptbench-done or scriptbench-error. */
318
+ async run() {
319
+ if (this.status === 'stopped') {
320
+ // stop() landed before run() (a CLI Ctrl-C, a chat-tool timeout holding the
321
+ // bench through deps.onBench): still exactly one terminal event, or
322
+ // runBenchOnce would never settle.
323
+ this._emit('scriptbench-error', { message: 'bench stopped before it started', code: 'STOPPED' });
324
+ return;
325
+ }
326
+ if (this.status !== 'created') return; // a second run() is a no-op
327
+ let held = false;
328
+ try {
329
+ if (!SCRIPT_KEY_RE.test(this.key)) throw benchError(`script not found: ${this.key}`, 'NOT_FOUND');
330
+ // Per-key BEFORE global: re-running the script you are looking at is the
331
+ // common refusal, and "already running" names the fix; the global cap is
332
+ // the one the OTHER tab hits.
333
+ if (activeByKey.has(this.key)) throw benchError(`script "${this.key}" is already running in the bench`, 'BUSY');
334
+ if (activeIds.size >= BENCH_MAX_PARALLEL) {
335
+ throw benchError(`at most ${BENCH_MAX_PARALLEL} bench runs at once — wait for one to finish`, 'BUSY');
336
+ }
337
+ activeIds.add(this.id);
338
+ activeByKey.set(this.key, this.id);
339
+ held = true;
340
+ this.status = 'running';
341
+ const plan = await this._plan(); // resolve first: a refusal leaves no folder behind
342
+ await this._makeDir();
343
+ const result = this.request.all === true ? await this._runAll(plan) : await this._runOne(plan, plan.kase, null);
344
+ if (this.status !== 'stopped') this.status = 'done';
345
+ this._emit('scriptbench-done', { result });
346
+ } catch (e) {
347
+ this.status = 'error';
348
+ this._emit('scriptbench-error', { message: e?.message || String(e), code: e?.code || null });
349
+ } finally {
350
+ if (held) { activeIds.delete(this.id); activeByKey.delete(this.key); }
351
+ }
352
+ }
353
+
354
+ /** W11: one live folder per script key — the previous one is reclaimed here. */
355
+ async _makeDir() {
356
+ const dir = join(benchRoot(this.deps.home || worcaHome()), this.id);
357
+ const prev = dirByKey.get(this.key);
358
+ if (prev && prev !== dir) await rm(prev, { recursive: true, force: true, maxRetries: 3 }).catch(() => {});
359
+ dirByKey.set(this.key, dir);
360
+ await mkdir(dir, { recursive: true });
361
+ this.benchDir = dir;
362
+ }
363
+
364
+ /** Resolve the script (registry entry or draft) and the case, once per bench. */
365
+ async _plan() {
366
+ const registry = isObject(this.deps.registry)
367
+ ? this.deps.registry
368
+ : loadScriptRegistry({ agentKeys: this.deps.agentKeys ?? Object.keys(loadAgentRegistry()) });
369
+ // A registry is a plain object: `constructor`/`toString`/`valueOf`/`hasOwnProperty`
370
+ // pass SCRIPT_KEY_RE and would answer with a Function the bench then benches.
371
+ const entry = Object.hasOwn(registry, this.key) ? registry[this.key] : null;
372
+ const draftReq = isObject(this.request.draft) ? this.request.draft : null;
373
+ let meta = entry;
374
+ if (draftReq) {
375
+ // W10: a draft is the editor's unsaved meta + source. Only the user layer
376
+ // is writable, so a draft of a built-in or a plugin script is refused.
377
+ const origin = String(entry?.origin || 'user');
378
+ if (entry && origin !== 'user') {
379
+ const why = origin.startsWith('plugin:') ? `it belongs to plugin "${origin.slice('plugin:'.length)}"` : 'it is a built-in script';
380
+ throw benchError(`cannot bench a draft of "${this.key}" — ${why}; duplicate it first`, 'BAD_REQUEST');
381
+ }
382
+ // A key the store would refuse on Save must not bench green either: the
383
+ // reserved route segments in any case AND the Windows device stems (C29),
384
+ // read from the store's own gate so the two lists cannot drift apart.
385
+ assertKeyAllowed(this.key);
386
+ // C23: one file holds `Lint` and `lint` on macOS and Windows, so the store
387
+ // refuses a key that differs from an existing one only in case (assertKeyFree).
388
+ const twin = entry ? null : Object.keys(registry).find((k) => k.toLowerCase() === this.key.toLowerCase());
389
+ if (twin) {
390
+ throw benchError(`a script "${twin}" already exists — script keys differ only in case, and one file holds both `
391
+ + 'on macOS and Windows', 'BAD_REQUEST');
392
+ }
393
+ if (!entry && (this.deps.agentKeys ?? Object.keys(loadAgentRegistry())).includes(this.key)) {
394
+ throw benchError(`"${this.key}" is an agent key — scripts and agents share one namespace, so pick another key`, 'BAD_REQUEST');
395
+ }
396
+ // applyFileRules: a win32 program needs its POSIX twin. Without this the draft
397
+ // wrote an EMPTY `.sh` on POSIX, ran it, ignored the sidecar's `command` and
398
+ // reported `clean` — a green bench for a script the Save refuses.
399
+ if (typeof draftReq.sourceWin32 === 'string' && draftReq.sourceWin32.trim()
400
+ && !(typeof draftReq.source === 'string' && draftReq.source.trim())) {
401
+ throw benchError('sourceWin32 needs a shell file — add the default source too', 'BAD_REQUEST');
402
+ }
403
+ // The draft comes off the Overview form, which sends `null` for a cleared key.
404
+ const raw = stripNullKeys({ ...(isObject(draftReq.meta) ? draftReq.meta : {}), key: this.key });
405
+ for (const f of ['origin', 'scriptPath', 'scriptsDir', 'commandResolved', 'frontmatter']) delete raw[f];
406
+ // The store owns `file` on every write path; a draft is no exception.
407
+ const hasSource = typeof draftReq.source === 'string' && draftReq.source.trim() !== '';
408
+ raw.file = hasSource ? sourceFileFor(this.key, raw.runtime) : null;
409
+ const issues = validateScriptMetaV2(raw).errors;
410
+ if (issues.length) throw benchError(issues.join('; '), 'BAD_REQUEST');
411
+ meta = normalizeScriptMeta(raw, { warn: () => {} }).meta;
412
+ } else if (!entry) {
413
+ throw benchError(`script not found: ${this.key}`, 'NOT_FOUND');
414
+ }
415
+ let kase = null;
416
+ const caseId = typeof this.request.caseId === 'string' ? this.request.caseId : '';
417
+ if (caseId || this.request.all === true) {
418
+ const stored = await readScript(this.key);
419
+ const cases = stored ? [...stored.cases, ...stored.userCases] : [];
420
+ if (caseId) {
421
+ kase = cases.find((c) => c.id === caseId) || null;
422
+ if (!kase) throw benchError(`case "${caseId}" not found for script "${this.key}"`, 'NOT_FOUND');
423
+ }
424
+ this.cases = cases;
425
+ }
426
+ return { meta, entry, draft: !!draftReq, draftReq, kase };
427
+ }
428
+
429
+ /** Everything one execution needs: params, ports, inputs, cwd, the draft file. */
430
+ async _prepare(plan, kase, caseId) {
431
+ const { meta } = plan;
432
+ const dirs = dirsFor(this.benchDir, caseId);
433
+ await mkdir(dirs.pipeline, { recursive: true });
434
+ // §4.2: a case supplies everything; the loose request fields are ignored.
435
+ const src = kase || this.request;
436
+ const params = effectiveScriptParams(meta, { params: isObject(src.params) ? src.params : {} });
437
+ const errors = paramErrors(meta, params);
438
+ if (errors.length) throw benchError(errors.join('; '), 'BAD_REQUEST');
439
+ const set = casePortSet(meta, { ports: src.ports ?? null });
440
+ if (set.errors) throw benchError(set.errors.join('; '), 'BAD_REQUEST');
441
+ const ports = { inputs: set.inputs, outputs: set.outputs, verdict: set.verdict };
442
+ const bindings = await writeBenchInputs(dirs.in, ports, isObject(src.inputs) ? src.inputs : {});
443
+ const { cwd, checkpointRef } = await this._cwd(isObject(src.cwd) ? src.cwd : { kind: 'scratch' }, dirs);
444
+ const timeoutMs = Number.isInteger(src.timeoutMs) && src.timeoutMs >= MIN_TIMEOUT_MS
445
+ ? Math.min(src.timeoutMs, MAX_TIMEOUT_MS)
446
+ : (Number.isInteger(meta.timeoutMs) ? meta.timeoutMs : DEFAULT_TIMEOUT_MS);
447
+ // The LAST write. _runOne's finally removes the file once a spec exists; a
448
+ // throw between here and the return would leak it, hence the catch below (W10).
449
+ const draftPath = await this._writeDraft(plan, meta);
450
+ try {
451
+ return this._spec(plan, kase, caseId, { meta, ports, bindings, dirs, cwd, checkpointRef, params, timeoutMs, draftPath });
452
+ } catch (e) {
453
+ if (draftPath) await rm(draftPath, { force: true, maxRetries: 3 }).catch(() => {});
454
+ throw e;
455
+ }
456
+ }
457
+
458
+ /** The synthetic ctx for one prepared execution (split out so _prepare can guard the draft file). */
459
+ _spec(plan, kase, caseId, { meta, ports, bindings, dirs, cwd, checkpointRef, params, timeoutMs, draftPath }) {
460
+ const file = draftPath ?? (plan.draft ? null : plan.entry.scriptPath ?? null);
461
+ // The registry stamps the host-platform command; a draft's own map is
462
+ // resolved here with the SAME reader (a per-platform command is the
463
+ // sidecar's business, so the bench must not ship the raw map to the runner).
464
+ const command = plan.draft
465
+ ? resolvePlatformValue(meta.command, this.platform)
466
+ : (plan.entry.commandResolved ?? null);
467
+ const ctx = buildBenchCtx({
468
+ id: this.id, meta, resolved: { file, command, params, timeoutMs }, ports, bindings, dirs, cwd, checkpointRef,
469
+ signal: this.abort.signal,
470
+ onEvent: (e) => { if (e && e.type === 'text') this._line(caseId, e.text); },
471
+ });
472
+ return { meta, ports, ctx, dirs, draftPath, draft: plan.draft, expect: kase ? kase.expect : null };
473
+ }
474
+
475
+ /** `.bench-<id8>-<basename>` beside the real file, so relative imports still resolve. */
476
+ async _writeDraft(plan, meta) {
477
+ if (!plan.draft) return null;
478
+ const source = typeof plan.draftReq.source === 'string' ? plan.draftReq.source : '';
479
+ const win = typeof plan.draftReq.sourceWin32 === 'string' ? plan.draftReq.sourceWin32 : '';
480
+ if (!source.trim() && !win.trim()) return null; // an inline shell command has no file
481
+ const useWin = this.platform === 'win32' && meta.runtime === 'shell' && win.trim() !== '';
482
+ const basename = sourceFileFor(this.key, meta.runtime, { win32: useWin });
483
+ if (!basename) return null;
484
+ const dir = userScriptsDir();
485
+ if (!dir) throw benchError('cannot resolve the user scripts directory (WORCA_HOME unset?)', 'BAD_REQUEST');
486
+ await mkdir(dir, { recursive: true });
487
+ const path = join(dir, `.bench-${this.id.slice(-8)}-${basename}`);
488
+ await writeFile(path, programText(meta.runtime, useWin ? win : source, { win32: useWin }), 'utf8');
489
+ return path;
490
+ }
491
+
492
+ /** W1: a scratch folder, or a REGISTERED project's real checkout — plus the
493
+ * CLI-only `dir` arm, behind deps.allowDirCwd (resolveBenchCwd). */
494
+ _cwd(cwd, dirs) {
495
+ return resolveBenchCwd(cwd, {
496
+ dirs,
497
+ projects: typeof this.deps.projects === 'function' ? this.deps.projects : listProjects,
498
+ allowDirCwd: this.deps.allowDirCwd === true,
499
+ });
500
+ }
501
+
502
+ _line(caseId, text) {
503
+ this.tail = this.tail || [];
504
+ this.tail.push(text);
505
+ if (this.tail.length > TAIL_LINES) this.tail.shift();
506
+ this._emit('scriptbench-line', { caseId: caseId || null, stream: 'out', text });
507
+ }
508
+
509
+ async _runOne(plan, kase, caseId) {
510
+ this.tail = [];
511
+ const spec = await this._prepare(plan, kase, caseId);
512
+ const started = Date.now();
513
+ try {
514
+ const res = await this.runner(spec.ctx);
515
+ const verdict = res.verdict || null;
516
+ return await this._result(spec, {
517
+ status: hasBlocking(verdict) ? 'blocking' : 'clean',
518
+ exitCode: res.exitCode ?? null,
519
+ runtime: res.runtime || spec.meta.runtime,
520
+ durationMs: res.durationMs ?? (Date.now() - started),
521
+ summary: res.summary || '',
522
+ warnings: res.warnings || [],
523
+ fired: firedOutputs(spec.ports.outputs || [], verdict).map((p) => p.id),
524
+ verdict,
525
+ envelopePath: res.envelopePath || null,
526
+ error: null,
527
+ });
528
+ } catch (e) {
529
+ // An EXECUTION failure is a RESULT — the bench exists to show it. Only the
530
+ // runner's own two non-failure throws are re-read: an abort is a stop, and
531
+ // the timeout sentence (P1a §6.3) is the timeout.
532
+ const message = e?.message || String(e);
533
+ const stopped = this.abort.signal.aborted || e?.name === 'AbortError';
534
+ const status = stopped ? 'stopped' : /^script ".*" timed out after /.test(message) ? 'timeout' : 'error';
535
+ return await this._result(spec, {
536
+ status,
537
+ exitCode: null,
538
+ runtime: spec.meta.runtime,
539
+ durationMs: Date.now() - started,
540
+ summary: '',
541
+ warnings: [],
542
+ fired: [],
543
+ verdict: null,
544
+ envelopePath: null,
545
+ error: { message, tail: [...(this.tail || [])] },
546
+ });
547
+ } finally {
548
+ if (spec.draftPath) await rm(spec.draftPath, { force: true, maxRetries: 3 }).catch(() => {});
549
+ }
550
+ }
551
+
552
+ async _result(spec, partial) {
553
+ const result = {
554
+ ...partial,
555
+ outputs: await readOutputs(spec.ports, spec.ctx.outputs),
556
+ expect: null,
557
+ draft: spec.draft,
558
+ benchDir: spec.dirs.base,
559
+ };
560
+ result.expect = evaluateExpect(spec.expect, result);
561
+ return result;
562
+ }
563
+
564
+ /** The result of a case whose preparation threw (Run all only): an `error`
565
+ * result with the refusal as its message, checked against the expectation
566
+ * like any other so an `expect.verdict: 'error'` case can still pass. */
567
+ _unpreparedResult(plan, kase, e) {
568
+ const result = {
569
+ status: 'error', exitCode: null, runtime: plan.meta.runtime, durationMs: 0, summary: '', warnings: [],
570
+ fired: [], verdict: null, envelopePath: null,
571
+ error: { message: e?.message || String(e), tail: [] },
572
+ outputs: {}, expect: null, draft: plan.draft, benchDir: dirsFor(this.benchDir, kase.id).base,
573
+ };
574
+ result.expect = evaluateExpect(kase.expect, result);
575
+ return result;
576
+ }
577
+
578
+ /** §4.3: every saved case, sequentially, under one benchId; stop skips the rest. */
579
+ async _runAll(plan) {
580
+ const cases = this.cases || [];
581
+ if (!cases.length) throw benchError(`script "${this.key}" has no saved cases`, 'BAD_REQUEST');
582
+ const out = [];
583
+ let passed = 0;
584
+ let failed = 0;
585
+ let unchecked = 0;
586
+ for (const kase of cases) {
587
+ if (this.abort.signal.aborted) break;
588
+ let result;
589
+ try {
590
+ result = await this._runOne(plan, kase, kase.id);
591
+ } catch (e) {
592
+ // A case that cannot even be PREPARED (a param the sidecar no longer
593
+ // declares, a project that was unregistered) is that case's failure,
594
+ // never the batch's: the other cases still run.
595
+ result = this._unpreparedResult(plan, kase, e);
596
+ }
597
+ out.push({ caseId: kase.id, result });
598
+ if (!result.expect) unchecked += 1;
599
+ else if (result.expect.pass) passed += 1;
600
+ else failed += 1;
601
+ }
602
+ return { cases: out, passed, failed, unchecked };
603
+ }
604
+ }
605
+
606
+ /**
607
+ * One bench run. The server wires it onto the WS bus the way wireAgentGen wires
608
+ * an AgentGen; `run()` never throws and emits exactly one terminal event.
609
+ * @param {object} request { key, caseId?, all?, draft?, params?, ports?, inputs?, cwd?, timeoutMs? }
610
+ * @param {object} [deps] { home?, runner?, registry?, agentKeys?, projects?, platform?, onLine?, onBench?,
611
+ * allowDirCwd? } — allowDirCwd admits `cwd: { kind: 'dir', dir }`; the CLI sets it, the server never does.
612
+ */
613
+ export function createBench(request, deps = {}) { return new Bench(request, deps); }
614
+
615
+ /** The same engine without a socket: the CLI (P3) and the chat tool (P4) use this.
616
+ * `deps.onLine({ benchId, caseId, stream, text })` receives every streamed line
617
+ * (the CLI prints them to stderr, the chat tool keeps the tail); `deps.onBench(bench)`
618
+ * hands the caller the live bench so it can `stop()` it (Ctrl-C, a tool timeout). */
619
+ export function runBenchOnce(request, deps = {}) {
620
+ const bench = createBench(request, deps);
621
+ if (typeof deps.onLine === 'function') bench.on('scriptbench-line', deps.onLine);
622
+ if (typeof deps.onBench === 'function') deps.onBench(bench);
623
+ return new Promise((resolve, reject) => {
624
+ bench.once('scriptbench-done', (e) => resolve(e.result));
625
+ bench.once('scriptbench-error', (e) => reject(Object.assign(new Error(e.message), { code: e.code || null })));
626
+ bench.run();
627
+ });
628
+ }