@iowarp/clio-coder 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. package/CHANGELOG.md +407 -0
  2. package/CODE_OF_CONDUCT.md +21 -0
  3. package/CONTRIBUTING.md +224 -0
  4. package/LICENSE +202 -0
  5. package/NOTICE +9 -0
  6. package/README.md +798 -0
  7. package/SECURITY.md +72 -0
  8. package/assets/clio-coder-logo-128.webp +0 -0
  9. package/damage-control-rules.yaml +419 -0
  10. package/dist/acp-UMLFVA3F.js +92 -0
  11. package/dist/agents-Q4MYPMUW.js +91 -0
  12. package/dist/auth-O6HYIJ6J.js +521 -0
  13. package/dist/chunk-262G75JS.js +35 -0
  14. package/dist/chunk-26BZQOAD.js +1281 -0
  15. package/dist/chunk-2J63S4SF.js +508 -0
  16. package/dist/chunk-3DANZDGR.js +717 -0
  17. package/dist/chunk-4UQA7NCT.js +29 -0
  18. package/dist/chunk-527KG6XR.js +497 -0
  19. package/dist/chunk-5LDRNKX2.js +1063 -0
  20. package/dist/chunk-5N2FG33Q.js +25 -0
  21. package/dist/chunk-67MTHP2E.js +135 -0
  22. package/dist/chunk-6CWDTGUC.js +20 -0
  23. package/dist/chunk-7BHLZB3A.js +2115 -0
  24. package/dist/chunk-7RBKDI66.js +348 -0
  25. package/dist/chunk-AMFR5YA3.js +541 -0
  26. package/dist/chunk-BBUH4VAA.js +1224 -0
  27. package/dist/chunk-BYEU76JP.js +899 -0
  28. package/dist/chunk-CLJ5HLUD.js +458 -0
  29. package/dist/chunk-D5YD55AR.js +116 -0
  30. package/dist/chunk-DXQNI4PC.js +61 -0
  31. package/dist/chunk-E3NYWENM.js +1004 -0
  32. package/dist/chunk-GNGDQYDU.js +34688 -0
  33. package/dist/chunk-GOTUR54M.js +9 -0
  34. package/dist/chunk-HBU5MTAM.js +41 -0
  35. package/dist/chunk-HMYNFFY4.js +28 -0
  36. package/dist/chunk-JPOWPFCU.js +1010 -0
  37. package/dist/chunk-JWHCJDCI.js +1215 -0
  38. package/dist/chunk-KBR4MZZR.js +41 -0
  39. package/dist/chunk-KKKPTZLM.js +93 -0
  40. package/dist/chunk-ME6DNWIU.js +66 -0
  41. package/dist/chunk-NI4DEJMC.js +88 -0
  42. package/dist/chunk-O4EJEDHO.js +659 -0
  43. package/dist/chunk-PIDUD6M2.js +31 -0
  44. package/dist/chunk-PS4PFJQP.js +29459 -0
  45. package/dist/chunk-QV47YRF4.js +48 -0
  46. package/dist/chunk-RQDWMVRB.js +279 -0
  47. package/dist/chunk-TFSSEXL6.js +136 -0
  48. package/dist/chunk-TKHQ4DGZ.js +8290 -0
  49. package/dist/chunk-TPOCL34A.js +2876 -0
  50. package/dist/chunk-UGYAX5YI.js +565 -0
  51. package/dist/chunk-UHTSULZS.js +461 -0
  52. package/dist/chunk-UU3R62TT.js +128 -0
  53. package/dist/chunk-UWIJNAOB.js +3906 -0
  54. package/dist/chunk-VOO7NYPP.js +914 -0
  55. package/dist/chunk-VPAWTYLY.js +117 -0
  56. package/dist/chunk-WD6AJM35.js +1216 -0
  57. package/dist/chunk-X3BR7HWV.js +115 -0
  58. package/dist/chunk-X3NE4WVW.js +120 -0
  59. package/dist/chunk-XNISANGE.js +1395 -0
  60. package/dist/chunk-XV4ZJ6ZM.js +3177 -0
  61. package/dist/cli/index.js +236 -0
  62. package/dist/clio-KIQ5SNDS.js +53 -0
  63. package/dist/components-JVHMUBEB.js +653 -0
  64. package/dist/config-ZFCDBMDC.js +372 -0
  65. package/dist/configure-G4E3A2PG.js +27 -0
  66. package/dist/context-CDXTP2MP.js +293 -0
  67. package/dist/context-E3KIFVXI.js +185 -0
  68. package/dist/context-clear-3F4PLXOS.js +102 -0
  69. package/dist/context-index-Q7YSYTR3.js +106 -0
  70. package/dist/docs-YIETIWZI.js +280 -0
  71. package/dist/doctor-M5HJJZOL.js +61 -0
  72. package/dist/domains/agents/builtins/architect.md +33 -0
  73. package/dist/domains/agents/builtins/coder.md +31 -0
  74. package/dist/domains/agents/builtins/context-bootstrap.md +38 -0
  75. package/dist/domains/agents/builtins/debugger.md +30 -0
  76. package/dist/domains/agents/builtins/documenter.md +31 -0
  77. package/dist/domains/agents/builtins/git-master.md +30 -0
  78. package/dist/domains/agents/builtins/provenance.md +30 -0
  79. package/dist/domains/agents/builtins/researcher.md +71 -0
  80. package/dist/domains/agents/builtins/scout.md +42 -0
  81. package/dist/domains/agents/builtins/tester.md +31 -0
  82. package/dist/domains/agents/builtins/verifier.md +30 -0
  83. package/dist/domains/agents/builtins/wiki-writer.md +41 -0
  84. package/dist/eval-B3KZZESM.js +2674 -0
  85. package/dist/evidence-V67CHM35.js +233 -0
  86. package/dist/evolve-YDZSUQYA.js +518 -0
  87. package/dist/extensions-SRG7XCAH.js +207 -0
  88. package/dist/fleet-CA2CRTVG.js +760 -0
  89. package/dist/fleet-preflight-CLIAX7YR.js +21 -0
  90. package/dist/init-2OZDJE2D.js +227 -0
  91. package/dist/memory-3PIQQAKX.js +207 -0
  92. package/dist/models-DY35XI7Y.js +237 -0
  93. package/dist/paths-5OMXW7Z4.js +57 -0
  94. package/dist/preload-KZVHET2B.js +11 -0
  95. package/dist/reset-PIFYNOS3.js +216 -0
  96. package/dist/run-3VSPP24F.js +735 -0
  97. package/dist/share-D36RQCXM.js +241 -0
  98. package/dist/skills-F2MRLELY.js +445 -0
  99. package/dist/skills-eval-E2ZTW4PL.js +932 -0
  100. package/dist/targets-DZMEZAH4.js +977 -0
  101. package/dist/trace-7NYCUI2J.js +250 -0
  102. package/dist/uninstall-AD3JWHBB.js +322 -0
  103. package/dist/upgrade-WYYBKGDY.js +301 -0
  104. package/dist/usage-ULIDAGFF.js +755 -0
  105. package/dist/version-ROZ6CZKH.js +16 -0
  106. package/dist/wiki-generate-PKFIX6OB.js +377 -0
  107. package/dist/worker/entry.js +1739 -0
  108. package/docs/README.md +93 -0
  109. package/docs/acp.md +120 -0
  110. package/docs/alcf-provider.md +72 -0
  111. package/docs/architecture.md +172 -0
  112. package/docs/artifact-versions.md +54 -0
  113. package/docs/built-in-agents.md +265 -0
  114. package/docs/capacity-and-scheduling.md +97 -0
  115. package/docs/commands-and-modes.md +554 -0
  116. package/docs/config-knobs-audit.md +115 -0
  117. package/docs/configuration-and-targets.md +812 -0
  118. package/docs/context-engine.md +236 -0
  119. package/docs/dispatch-architecture-rationale.md +126 -0
  120. package/docs/documentation-coverage.md +46 -0
  121. package/docs/documentation-guide.md +166 -0
  122. package/docs/environment-variables.md +105 -0
  123. package/docs/eval-runner.md +205 -0
  124. package/docs/evals-internal.md +298 -0
  125. package/docs/evidence-and-memory.md +243 -0
  126. package/docs/evolution.md +143 -0
  127. package/docs/exit-codes-and-output.md +74 -0
  128. package/docs/extensions-and-sharing.md +306 -0
  129. package/docs/fleet-demo-runbook.md +179 -0
  130. package/docs/fleet-dispatch.md +591 -0
  131. package/docs/glossary.md +75 -0
  132. package/docs/html/agents_blueprint.html +936 -0
  133. package/docs/html/alcf_blueprint.html +324 -0
  134. package/docs/html/architecture_blueprint.html +850 -0
  135. package/docs/html/commands_blueprint.html +794 -0
  136. package/docs/html/config_knobs_audit_blueprint.html +178 -0
  137. package/docs/html/configuration_blueprint.html +1080 -0
  138. package/docs/html/context_blueprint.html +603 -0
  139. package/docs/html/documentation_blueprint.html +832 -0
  140. package/docs/html/environment_blueprint.html +404 -0
  141. package/docs/html/eval_blueprint.html +743 -0
  142. package/docs/html/evals_internal_blueprint.html +190 -0
  143. package/docs/html/evolution_blueprint.html +674 -0
  144. package/docs/html/extensions_blueprint.html +2065 -0
  145. package/docs/html/fleet_dispatch_blueprint.html +286 -0
  146. package/docs/html/index.html +919 -0
  147. package/docs/html/lifecycle_blueprint.html +723 -0
  148. package/docs/html/memory_blueprint.html +699 -0
  149. package/docs/html/middleware_blueprint.html +664 -0
  150. package/docs/html/models_blueprint.html +2366 -0
  151. package/docs/html/observability_blueprint.html +683 -0
  152. package/docs/html/provider_adapter_blueprint.html +245 -0
  153. package/docs/html/safety_blueprint.html +1386 -0
  154. package/docs/html/shared.css +571 -0
  155. package/docs/html/shared.js +143 -0
  156. package/docs/html/skills_blueprint.html +671 -0
  157. package/docs/html/soak_blueprint.html +182 -0
  158. package/docs/html/tool_usage_blueprint.html +350 -0
  159. package/docs/html/tools_blueprint.html +2249 -0
  160. package/docs/html/trace_blueprint.html +235 -0
  161. package/docs/html/tui_design_blueprint.html +314 -0
  162. package/docs/html/validation_blueprint.html +961 -0
  163. package/docs/html/worker_dispatch_blueprint.html +231 -0
  164. package/docs/installation-and-lifecycle.md +308 -0
  165. package/docs/middleware-and-components.md +148 -0
  166. package/docs/model-catalog.md +189 -0
  167. package/docs/observability.md +233 -0
  168. package/docs/proactive-memory.md +452 -0
  169. package/docs/prompt-envelope-and-tools.md +142 -0
  170. package/docs/provider-adapter-cookbook.md +148 -0
  171. package/docs/release-cut-checklist.md +138 -0
  172. package/docs/safety-model.md +357 -0
  173. package/docs/scientific-validation.md +105 -0
  174. package/docs/session-lifecycle.md +156 -0
  175. package/docs/skills-marketplace.md +46 -0
  176. package/docs/tool-usage.md +527 -0
  177. package/docs/trace-store.md +132 -0
  178. package/docs/troubleshooting.md +33 -0
  179. package/docs/tui-design.md +239 -0
  180. package/docs/worker-dispatch-mechanics.md +242 -0
  181. package/package.json +132 -0
  182. package/skills/README.md +408 -0
  183. package/skills/git/commit-crafting/SKILL.md +79 -0
  184. package/skills/git/commit-crafting/evals.md +92 -0
  185. package/skills/git/create-pr/SKILL.md +116 -0
  186. package/skills/git/create-pr/evals.md +114 -0
  187. package/skills/git/investigate-issue/SKILL.md +139 -0
  188. package/skills/git/investigate-issue/evals.md +94 -0
  189. package/skills/git/resolve-merge-conflicts/SKILL.md +96 -0
  190. package/skills/git/resolve-merge-conflicts/evals.md +58 -0
  191. package/skills/git/review-changes/SKILL.md +103 -0
  192. package/skills/git/review-changes/evals.md +85 -0
  193. package/skills/git/worktree-create/SKILL.md +92 -0
  194. package/skills/git/worktree-create/evals.md +97 -0
  195. package/skills/git/worktree-create/references/worktree-setup.md +66 -0
  196. package/skills/git/worktree-merge/SKILL.md +95 -0
  197. package/skills/git/worktree-merge/evals.md +114 -0
  198. package/skills/skill-marketplace.json +261 -0
  199. package/skills/workflow/cut-it/SKILL.md +86 -0
  200. package/skills/workflow/cut-it/evals.md +42 -0
  201. package/src/domains/agents/builtins/architect.md +33 -0
  202. package/src/domains/agents/builtins/coder.md +31 -0
  203. package/src/domains/agents/builtins/context-bootstrap.md +38 -0
  204. package/src/domains/agents/builtins/debugger.md +30 -0
  205. package/src/domains/agents/builtins/documenter.md +31 -0
  206. package/src/domains/agents/builtins/git-master.md +30 -0
  207. package/src/domains/agents/builtins/provenance.md +30 -0
  208. package/src/domains/agents/builtins/researcher.md +71 -0
  209. package/src/domains/agents/builtins/scout.md +42 -0
  210. package/src/domains/agents/builtins/tester.md +31 -0
  211. package/src/domains/agents/builtins/verifier.md +30 -0
  212. package/src/domains/agents/builtins/wiki-writer.md +41 -0
  213. package/src/domains/agents/fleets/build-review.md +34 -0
  214. package/src/domains/agents/fleets/build-test.md +35 -0
  215. package/src/domains/agents/fleets/sdlc.md +86 -0
  216. package/src/domains/prompts/fragments/identity/clio-worker.md +11 -0
  217. package/src/domains/prompts/fragments/identity/clio.md +26 -0
  218. package/src/domains/prompts/fragments/operating/contract.md +64 -0
  219. package/src/domains/prompts/fragments/safety/auto-edit.md +14 -0
  220. package/src/domains/prompts/fragments/safety/full-auto.md +14 -0
  221. package/src/domains/prompts/fragments/safety/read-only.md +13 -0
  222. package/src/domains/prompts/fragments/safety/suggest.md +13 -0
  223. package/src/domains/prompts/fragments/wiki/page.md +75 -0
  224. package/src/domains/prompts/fragments/wiki/plan.md +48 -0
  225. package/src/domains/providers/models/cloud-models/alcf.yaml +40 -0
  226. package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +993 -0
@@ -0,0 +1,932 @@
1
+ import {
2
+ HEADLESS_PERMISSION_DENIED_MARKER,
3
+ NO_NETWORK_TOOLS_ENV,
4
+ buildEvalEvidence,
5
+ combineBashOutput,
6
+ runBashCommand,
7
+ summarizeEvalResults
8
+ } from "./chunk-GNGDQYDU.js";
9
+ import {
10
+ parseSkillEvals
11
+ } from "./chunk-2J63S4SF.js";
12
+ import "./chunk-RQDWMVRB.js";
13
+ import {
14
+ ZERO_EVAL_HARNESS_METRICS,
15
+ evalClioProvenance,
16
+ evalEnvironmentProvenance,
17
+ writeEvalArtifact
18
+ } from "./chunk-7BHLZB3A.js";
19
+ import "./chunk-TPOCL34A.js";
20
+ import "./chunk-E3NYWENM.js";
21
+ import "./chunk-XV4ZJ6ZM.js";
22
+ import {
23
+ discoverMarketplaceSkills,
24
+ loadSkills
25
+ } from "./chunk-26BZQOAD.js";
26
+ import "./chunk-WD6AJM35.js";
27
+ import "./chunk-QV47YRF4.js";
28
+ import "./chunk-CLJ5HLUD.js";
29
+ import "./chunk-BBUH4VAA.js";
30
+ import "./chunk-GOTUR54M.js";
31
+ import "./chunk-PIDUD6M2.js";
32
+ import {
33
+ formatColumns,
34
+ printError
35
+ } from "./chunk-67MTHP2E.js";
36
+ import "./chunk-TKHQ4DGZ.js";
37
+ import "./chunk-D5YD55AR.js";
38
+ import "./chunk-JWHCJDCI.js";
39
+ import "./chunk-UHTSULZS.js";
40
+ import "./chunk-5LDRNKX2.js";
41
+ import "./chunk-UGYAX5YI.js";
42
+ import "./chunk-5N2FG33Q.js";
43
+ import "./chunk-VPAWTYLY.js";
44
+ import {
45
+ clioDataDir,
46
+ clioStateDir
47
+ } from "./chunk-X3NE4WVW.js";
48
+ import "./chunk-DXQNI4PC.js";
49
+
50
+ // src/cli/skills-eval.ts
51
+ import { spawn } from "node:child_process";
52
+ import { createHash } from "node:crypto";
53
+ import { existsSync } from "node:fs";
54
+ import { cp, mkdir, mkdtemp, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
55
+ import { tmpdir } from "node:os";
56
+ import { join, resolve } from "node:path";
57
+ var DEFAULT_RUN_TIMEOUT_MS = 6e5;
58
+ var ARM_AUTONOMY = "full-auto";
59
+ var CHILD_OUTPUT_LIMIT = 4e6;
60
+ var PREVIEW_MAX_CHARS = 300;
61
+ var TRANSCRIPT_HEAD_CHARS = 9e3;
62
+ var TRANSCRIPT_TAIL_CHARS = 4e3;
63
+ var SKILL_EVAL_SIDECAR = "skill-eval.json";
64
+ async function runSkillsEvalCommand(nameOrPath, options) {
65
+ const resolved = resolveSkillBaseDir(nameOrPath, process.cwd());
66
+ if (resolved.baseDir === null) {
67
+ printError(resolved.error ?? `unknown skill: ${nameOrPath}`);
68
+ return 2;
69
+ }
70
+ const list = loadSkills({ disableDiscovery: true, explicitSkillPaths: [resolved.baseDir] });
71
+ const skill = list.items[0];
72
+ if (skill === void 0) {
73
+ printError(`skill did not load from ${resolved.baseDir}: ${list.diagnostics.map((d) => d.message).join("; ")}`);
74
+ return 2;
75
+ }
76
+ process.stderr.write(
77
+ `clio-coder skills eval: ${skill.name} from ${resolved.origin ?? "unknown"} at ${resolved.baseDir} (sha256 ${skill.normalizedHash.slice(0, 12)}\u2026)
78
+ `
79
+ );
80
+ const evalsPath = join(resolved.baseDir, "evals.md");
81
+ let evalsRaw;
82
+ try {
83
+ evalsRaw = await readFile(evalsPath, "utf8");
84
+ } catch {
85
+ printError(`skill ${skill.name} has no evals.md at ${evalsPath}`);
86
+ return 2;
87
+ }
88
+ const parsed = parseSkillEvals(evalsRaw);
89
+ for (const diagnostic of parsed.diagnostics) {
90
+ process.stderr.write(`clio-coder skills eval: ${diagnostic}
91
+ `);
92
+ }
93
+ const matcher = options.scenario === void 0 ? null : scenarioMatcher(options.scenario);
94
+ if (options.scenario !== void 0 && matcher === null) {
95
+ printError(`invalid --scenario "${options.scenario}": use a scenario id like S1 or a bare number`);
96
+ return 2;
97
+ }
98
+ const scenarios = matcher === null ? parsed.scenarios : parsed.scenarios.filter(matcher);
99
+ if (scenarios.length === 0) {
100
+ printError(
101
+ matcher === null ? `no parseable scenarios in ${evalsPath}` : `scenario ${options.scenario} not found in ${evalsPath} (have: ${parsed.scenarios.map((s) => s.id).join(", ") || "none"})`
102
+ );
103
+ return 2;
104
+ }
105
+ process.stderr.write(`clio-coder skills eval: ${describeArmPolicy(options.allowNetwork)}
106
+ `);
107
+ const childEnv = evalChildEnv(options.allowNetwork);
108
+ const timeoutMs = options.timeoutSeconds !== void 0 ? options.timeoutSeconds * 1e3 : DEFAULT_RUN_TIMEOUT_MS;
109
+ const workspaceOverride = await resolveWorkspaceOverride(options.workspace);
110
+ if (typeof workspaceOverride !== "string" && workspaceOverride !== null) {
111
+ printError(workspaceOverride.error);
112
+ return 2;
113
+ }
114
+ const startedAt = (/* @__PURE__ */ new Date()).toISOString();
115
+ const outcomes = [];
116
+ for (const scenario of scenarios) {
117
+ process.stderr.write(`clio-coder skills eval: ${skill.name} ${scenario.id} baseline/treatment/judge...
118
+ `);
119
+ outcomes.push(
120
+ await runScenario(
121
+ skill.name,
122
+ resolved.baseDir,
123
+ scenario,
124
+ options.target,
125
+ timeoutMs,
126
+ workspaceOverride,
127
+ options.trustFixtures,
128
+ childEnv
129
+ )
130
+ );
131
+ }
132
+ const endedAt = (/* @__PURE__ */ new Date()).toISOString();
133
+ const artifact = synthesizeArtifact(skill.name, evalsPath, evalsRaw, startedAt, endedAt, outcomes, options.target);
134
+ let evidenceId = null;
135
+ let evidenceDirectory = null;
136
+ const evidenceErrors = [];
137
+ try {
138
+ await writeEvalArtifact(clioDataDir(), artifact);
139
+ const built = await buildEvalEvidence({
140
+ dataDir: clioDataDir(),
141
+ stateDir: clioStateDir(),
142
+ artifact,
143
+ sidecars: [SKILL_EVAL_SIDECAR]
144
+ });
145
+ evidenceId = built.evidenceId;
146
+ evidenceDirectory = built.directory;
147
+ await writeFile(
148
+ join(built.directory, SKILL_EVAL_SIDECAR),
149
+ `${JSON.stringify(sidecar(skill.name, artifact.evalId, outcomes, options.allowNetwork), null, 2)}
150
+ `,
151
+ "utf8"
152
+ );
153
+ } catch (error) {
154
+ evidenceErrors.push(error instanceof Error ? error.message : String(error));
155
+ }
156
+ for (const message of evidenceErrors) {
157
+ process.stderr.write(`clio-coder skills eval: evidence build failed: ${message}
158
+ `);
159
+ }
160
+ if (options.json) {
161
+ for (const outcome of outcomes) {
162
+ for (const bullet of outcome.bullets) {
163
+ process.stdout.write(
164
+ `${JSON.stringify({
165
+ schema: "experimental",
166
+ kind: "skill-eval-bullet",
167
+ skill: skill.name,
168
+ scenario: outcome.scenario.id,
169
+ title: outcome.scenario.title,
170
+ bullet: bullet.index,
171
+ expected: bullet.text,
172
+ verdict: bullet.verdict,
173
+ reason: bullet.reason,
174
+ network: networkPolicyLabel(options.allowNetwork),
175
+ autonomy: ARM_AUTONOMY,
176
+ baselineSessionId: outcome.baseline?.sessionId ?? null,
177
+ treatmentSessionId: outcome.treatment?.sessionId ?? null,
178
+ judgeSessionId: outcome.judge?.sessionId ?? null,
179
+ evalId: artifact.evalId,
180
+ evidenceId
181
+ })}
182
+ `
183
+ );
184
+ }
185
+ }
186
+ } else {
187
+ printHumanReport(skill.name, outcomes, artifact.evalId, evidenceId, evidenceDirectory, options.allowNetwork);
188
+ }
189
+ const anyFailure = outcomes.some(
190
+ (outcome) => outcome.bullets.some((bullet) => bullet.verdict === "fail" || bullet.verdict === "error")
191
+ );
192
+ const anyUnmeasured = outcomes.some((outcome) => outcome.bullets.some((bullet) => bullet.verdict === "unmeasured"));
193
+ if (anyFailure) return 1;
194
+ return anyUnmeasured || evidenceErrors.length > 0 ? 3 : 0;
195
+ }
196
+ function networkPolicyLabel(allowNetwork) {
197
+ return allowNetwork ? "allowed" : "hermetic";
198
+ }
199
+ function describeArmPolicy(allowNetwork) {
200
+ const network = allowNetwork ? "network allowed (--allow-network), so arm runs keep the web tools" : "network hermetic, so the web tools are stripped from every arm run (pass --allow-network to keep them)";
201
+ return `policy: autonomy ${ARM_AUTONOMY} for the baseline and treatment arms in per-run disposable workspaces; ${network}`;
202
+ }
203
+ function describeArmPolicyOutcome(allowNetwork) {
204
+ const network = allowNetwork ? "network: allowed (--allow-network); arm runs kept the web tools" : "network: hermetic; the web tools were stripped from every arm run";
205
+ return `policy: the baseline and treatment arms ran at autonomy ${ARM_AUTONOMY}; ${network}`;
206
+ }
207
+ function armRunArgs(arm, prompt, options = {}) {
208
+ const args = ["run", "--json", "--json-events", "terminal", "--no-skills"];
209
+ if (arm !== "judge") args.push("--autonomy", ARM_AUTONOMY);
210
+ if (options.skillBaseDir !== void 0) args.push("--skill", options.skillBaseDir);
211
+ if (options.target !== void 0) args.push("--target", options.target);
212
+ args.push(prompt);
213
+ return args;
214
+ }
215
+ function permissionWallReason(arm, run) {
216
+ const haystack = `${run.transcript}
217
+ ${run.finalText}`;
218
+ let blocked = 0;
219
+ let at = haystack.indexOf(HEADLESS_PERMISSION_DENIED_MARKER);
220
+ while (at >= 0) {
221
+ blocked += 1;
222
+ at = haystack.indexOf(HEADLESS_PERMISSION_DENIED_MARKER, at + HEADLESS_PERMISSION_DENIED_MARKER.length);
223
+ }
224
+ if (blocked === 0) return null;
225
+ return `${arm} arm transcript carries the headless permission wall ${blocked} time(s) ("${HEADLESS_PERMISSION_DENIED_MARKER}"): this run measured the harness's own autonomy gate, not the skill`;
226
+ }
227
+ function unmeasuredBullets(scenario, reason) {
228
+ return scenario.expected.map((text, index) => ({ index: index + 1, text, verdict: "unmeasured", reason }));
229
+ }
230
+ function evalChildEnv(allowNetwork, env = process.env) {
231
+ const childEnv = { ...env };
232
+ if (allowNetwork) Reflect.deleteProperty(childEnv, NO_NETWORK_TOOLS_ENV);
233
+ else childEnv[NO_NETWORK_TOOLS_ENV] = "1";
234
+ return childEnv;
235
+ }
236
+ function resolveSkillBaseDir(nameOrPath, cwd) {
237
+ const asPath = resolve(nameOrPath);
238
+ if (existsSync(join(asPath, "SKILL.md"))) return { baseDir: asPath, origin: "path" };
239
+ const discovered = loadSkills({ cwd }).items.find((skill) => skill.name === nameOrPath);
240
+ if (discovered !== void 0) {
241
+ return { baseDir: discovered.baseDir, origin: `${discovered.source}/${discovered.scope}` };
242
+ }
243
+ const marketplace = discoverMarketplaceSkills({ cwd });
244
+ const entry = marketplace.skills.find((skill) => skill.name === nameOrPath && skill.origin === "catalog");
245
+ if (entry !== void 0) return { baseDir: entry.sourceUrl, origin: "catalog" };
246
+ return {
247
+ baseDir: null,
248
+ error: `skill not found: ${nameOrPath} (expected an installed skill, a catalog skill name, or a directory containing SKILL.md)`
249
+ };
250
+ }
251
+ function scenarioMatcher(value) {
252
+ const trimmed = value.trim();
253
+ const full = /^([A-Za-z])(\d+)$/.exec(trimmed);
254
+ if (full !== null) {
255
+ const id = `${(full[1] ?? "").toUpperCase()}${Number.parseInt(full[2] ?? "", 10)}`;
256
+ return (scenario) => scenario.id === id;
257
+ }
258
+ const bare = /^(\d+)$/.exec(trimmed);
259
+ if (bare !== null) {
260
+ const wanted = Number.parseInt(bare[1] ?? "", 10);
261
+ return (scenario) => scenario.number === wanted;
262
+ }
263
+ return null;
264
+ }
265
+ async function resolveWorkspaceOverride(workspace) {
266
+ if (workspace === void 0) return null;
267
+ const resolved = resolve(workspace);
268
+ try {
269
+ const info = await stat(resolved);
270
+ if (!info.isDirectory()) return { error: `--workspace is not a directory: ${resolved}` };
271
+ return resolved;
272
+ } catch (error) {
273
+ const detail = error instanceof Error ? error.message : String(error);
274
+ return { error: `--workspace is not readable: ${resolved}: ${detail}` };
275
+ }
276
+ }
277
+ async function runScenario(skillName, skillBaseDir, scenario, target, timeoutMs, workspaceOverride, trustFixtures, childEnv) {
278
+ const scenarioStart = Date.now();
279
+ const workspace = await mkdtemp(join(tmpdir(), "clio-skill-eval-seed-"));
280
+ let runWorkspaces = null;
281
+ try {
282
+ if (workspaceOverride !== null) await copyWorkspace(workspaceOverride, workspace);
283
+ const fixtureError = await runFixtureCommands(scenario, workspace, timeoutMs, trustFixtures);
284
+ if (fixtureError !== null) {
285
+ return await completeScenarioOutcome({
286
+ scenario,
287
+ bullets: errorBullets(scenario, fixtureError),
288
+ baseline: null,
289
+ treatment: null,
290
+ judge: null,
291
+ workspace,
292
+ wallTimeMs: Date.now() - scenarioStart,
293
+ infraError: fixtureError
294
+ });
295
+ }
296
+ runWorkspaces = await materializeSkillEvalWorkspaces(workspace);
297
+ const baseline = await captureHeadlessRun(
298
+ armRunArgs("baseline", scenario.setup, { target }),
299
+ runWorkspaces.baseline,
300
+ timeoutMs,
301
+ childEnv
302
+ );
303
+ const treatment = await captureHeadlessRun(
304
+ armRunArgs("treatment", `/skill:${skillName} ${scenario.setup}`, { target, skillBaseDir }),
305
+ runWorkspaces.treatment,
306
+ timeoutMs,
307
+ childEnv
308
+ );
309
+ const infra = runInfraError("baseline", baseline) ?? runInfraError("treatment", treatment);
310
+ if (infra !== null) {
311
+ return await completeScenarioOutcome({
312
+ scenario,
313
+ bullets: errorBullets(scenario, infra),
314
+ baseline,
315
+ treatment,
316
+ judge: null,
317
+ workspace,
318
+ wallTimeMs: Date.now() - scenarioStart,
319
+ infraError: infra
320
+ });
321
+ }
322
+ const wall = permissionWallReason("baseline", baseline) ?? permissionWallReason("treatment", treatment);
323
+ if (wall !== null) {
324
+ return await completeScenarioOutcome({
325
+ scenario,
326
+ bullets: unmeasuredBullets(scenario, wall),
327
+ baseline,
328
+ treatment,
329
+ judge: null,
330
+ workspace,
331
+ wallTimeMs: Date.now() - scenarioStart,
332
+ infraError: wall
333
+ });
334
+ }
335
+ const judge = await captureHeadlessRun(
336
+ armRunArgs("judge", judgePrompt(scenario, baseline.transcript, treatment.transcript), { target }),
337
+ runWorkspaces.judge,
338
+ timeoutMs,
339
+ childEnv
340
+ );
341
+ const judgeInfra = runInfraError("judge", judge);
342
+ if (judgeInfra !== null) {
343
+ return await completeScenarioOutcome({
344
+ scenario,
345
+ bullets: errorBullets(scenario, judgeInfra),
346
+ baseline,
347
+ treatment,
348
+ judge,
349
+ workspace,
350
+ wallTimeMs: Date.now() - scenarioStart,
351
+ infraError: judgeInfra
352
+ });
353
+ }
354
+ const bullets = parseJudgeVerdicts(scenario, judge);
355
+ return await completeScenarioOutcome({
356
+ scenario,
357
+ bullets,
358
+ baseline,
359
+ treatment,
360
+ judge,
361
+ workspace,
362
+ wallTimeMs: Date.now() - scenarioStart,
363
+ infraError: null
364
+ });
365
+ } finally {
366
+ try {
367
+ if (runWorkspaces !== null) await runWorkspaces.cleanup();
368
+ } finally {
369
+ await rm(workspace, { recursive: true, force: true });
370
+ }
371
+ }
372
+ }
373
+ async function armWorkspace(created) {
374
+ const root = await mkdtemp(join(tmpdir(), "clio-skill-eval-"));
375
+ created.push(root);
376
+ const workspace = join(root, "workspace");
377
+ await mkdir(workspace, { recursive: true });
378
+ return workspace;
379
+ }
380
+ async function materializeSkillEvalWorkspaces(seedWorkspace) {
381
+ const created = [];
382
+ try {
383
+ const baseline = await armWorkspace(created);
384
+ const treatment = await armWorkspace(created);
385
+ const judge = await armWorkspace(created);
386
+ await Promise.all([copyWorkspace(seedWorkspace, baseline), copyWorkspace(seedWorkspace, treatment)]);
387
+ return {
388
+ baseline,
389
+ treatment,
390
+ judge,
391
+ cleanup: async () => {
392
+ await Promise.all(created.map((path) => rm(path, { recursive: true, force: true })));
393
+ }
394
+ };
395
+ } catch (error) {
396
+ await Promise.all(created.map((path) => rm(path, { recursive: true, force: true })));
397
+ throw error;
398
+ }
399
+ }
400
+ async function copyWorkspace(source, destination) {
401
+ await cp(source, destination, {
402
+ recursive: true,
403
+ preserveTimestamps: true,
404
+ verbatimSymlinks: true
405
+ });
406
+ }
407
+ async function completeScenarioOutcome(outcome) {
408
+ return {
409
+ ...outcome,
410
+ usage: await usageForCapturedRuns([outcome.baseline, outcome.treatment, outcome.judge])
411
+ };
412
+ }
413
+ async function runFixtureCommands(scenario, workspace, timeoutMs, trustFixtures) {
414
+ const commands = scenario.fixtureCommands?.trim();
415
+ if (commands === void 0 || commands.length === 0) return null;
416
+ if (!trustFixtures) {
417
+ return "scenario declares fixture commands, which are real shell from the skill's evals.md; review them and rerun with --trust-fixtures";
418
+ }
419
+ const validationError = validateFixtureCommands(commands);
420
+ if (validationError !== null) return `fixture setup rejected: ${validationError}`;
421
+ const result = await runBashCommand(commands, {
422
+ cwd: workspace,
423
+ timeoutMs: Math.min(timeoutMs, 12e4)
424
+ });
425
+ const output = combineBashOutput(result).trim();
426
+ if (result.timedOut) return `fixture setup timed out${output.length > 0 ? `: ${truncate(output, 300)}` : ""}`;
427
+ if (result.outputCapped) {
428
+ return `fixture setup output exceeded limit${output.length > 0 ? `: ${truncate(output, 300)}` : ""}`;
429
+ }
430
+ if (result.error !== null) {
431
+ return `fixture setup failed${output.length > 0 ? `: ${truncate(output, 300)}` : `: ${result.error.message}`}`;
432
+ }
433
+ return null;
434
+ }
435
+ function validateFixtureCommands(commands) {
436
+ const checks = [
437
+ {
438
+ pattern: /(^|[\s;&|(<>='"`])\/(?!dev\/null(?:\s|$))(?=\S)/,
439
+ reason: "absolute paths are not allowed in fixture commands"
440
+ },
441
+ {
442
+ pattern: /(^|[\s;&|()'"`/=])\.\.(?=$|[/\s;&|()'"`])/,
443
+ reason: "parent-directory path segments are not allowed in fixture commands"
444
+ },
445
+ {
446
+ pattern: /(^|[\s;&|()])~(?=$|[/\s;&|()])/,
447
+ reason: "home-directory expansion is not allowed in fixture commands"
448
+ },
449
+ {
450
+ pattern: /\$(?:\{(?:HOME|CLIO_CODER_HOME|CLIO_CODER_CONFIG_DIR|CLIO_CODER_DATA_DIR|CLIO_CODER_STATE_DIR|CLIO_CODER_CACHE_DIR)\}|(?:HOME|CLIO_CODER_HOME|CLIO_CODER_CONFIG_DIR|CLIO_CODER_DATA_DIR|CLIO_CODER_STATE_DIR|CLIO_CODER_CACHE_DIR)\b)/,
451
+ reason: "home and Clio directory environment variables are not allowed in fixture commands"
452
+ },
453
+ {
454
+ pattern: /(^|[\s;&|()])(?:cd|pushd|popd)\b/,
455
+ reason: "directory-changing commands are not allowed in fixture commands"
456
+ },
457
+ {
458
+ pattern: /(^|[\s;&|()])(?:sudo|su|ssh|scp|rsync)\b/,
459
+ reason: "privilege-changing or remote commands are not allowed in fixture commands"
460
+ }
461
+ ];
462
+ for (const check of checks) {
463
+ if (check.pattern.test(commands)) return check.reason;
464
+ }
465
+ return null;
466
+ }
467
+ async function usageForCapturedRuns(runs) {
468
+ const sessionIds = new Set(
469
+ runs.flatMap((run) => run?.sessionId !== null && run?.sessionId !== void 0 ? [run.sessionId] : [])
470
+ );
471
+ const usage = {
472
+ tokens: 0,
473
+ costUsd: 0,
474
+ harness: { ...ZERO_EVAL_HARNESS_METRICS }
475
+ };
476
+ if (sessionIds.size === 0) return usage;
477
+ const receiptRoot = join(clioStateDir(), "receipts");
478
+ let files;
479
+ try {
480
+ files = (await readdir(receiptRoot)).filter((name) => name.endsWith(".json"));
481
+ } catch {
482
+ return usage;
483
+ }
484
+ for (const file of files) {
485
+ let parsed;
486
+ try {
487
+ parsed = JSON.parse(await readFile(join(receiptRoot, file), "utf8"));
488
+ } catch {
489
+ continue;
490
+ }
491
+ if (!isRecord(parsed) || typeof parsed.sessionId !== "string" || !sessionIds.has(parsed.sessionId)) continue;
492
+ usage.tokens += readNumber(parsed.tokenCount);
493
+ usage.costUsd += readNumber(parsed.costUsd);
494
+ usage.harness.receiptCount += 1;
495
+ usage.harness.toolCalls += readNumber(parsed.toolCalls);
496
+ if (isRecord(parsed.safety) && isRecord(parsed.safety.decisions)) {
497
+ usage.harness.safetyBlocks += readNumber(parsed.safety.decisions.blocked);
498
+ }
499
+ }
500
+ return usage;
501
+ }
502
+ function runInfraError(label, run) {
503
+ if (run.timedOut) return `${label} run timed out`;
504
+ if (run.exitCode !== 0) {
505
+ const detail = run.stderr.trim().split("\n").at(-1) ?? "";
506
+ return `${label} run exited ${run.exitCode}${detail.length > 0 ? `: ${detail}` : ""}`;
507
+ }
508
+ if (run.transcript.trim().length === 0 && run.finalText.trim().length === 0) {
509
+ return `${label} run produced no transcript`;
510
+ }
511
+ return null;
512
+ }
513
+ function errorBullets(scenario, reason) {
514
+ return scenario.expected.map((text, index) => ({ index: index + 1, text, verdict: "error", reason }));
515
+ }
516
+ function captureHeadlessRun(args, cwd, timeoutMs, env) {
517
+ const startedMs = Date.now();
518
+ return new Promise((resolvePromise) => {
519
+ let stdout = "";
520
+ let pendingStdout = "";
521
+ let stderr = "";
522
+ let timedOut = false;
523
+ let settled = false;
524
+ const keepLine = (line) => {
525
+ if (line.length === 0) return;
526
+ if (stdout.length < CHILD_OUTPUT_LIMIT) stdout += `${line}
527
+ `;
528
+ };
529
+ const child = spawn(process.execPath, [process.argv[1] ?? "", ...args], {
530
+ cwd,
531
+ stdio: ["ignore", "pipe", "pipe"],
532
+ env
533
+ });
534
+ child.stdout.setEncoding("utf8");
535
+ child.stderr.setEncoding("utf8");
536
+ child.stdout.on("data", (chunk) => {
537
+ pendingStdout += chunk;
538
+ const lines = pendingStdout.split("\n");
539
+ pendingStdout = lines.pop() ?? "";
540
+ for (const line of lines) keepLine(line);
541
+ });
542
+ child.stderr.on("data", (chunk) => {
543
+ if (stderr.length < CHILD_OUTPUT_LIMIT) stderr += chunk;
544
+ });
545
+ let killTimer;
546
+ const timer = setTimeout(() => {
547
+ timedOut = true;
548
+ child.kill("SIGTERM");
549
+ killTimer = setTimeout(() => child.kill("SIGKILL"), 5e3);
550
+ }, timeoutMs);
551
+ const finish = (exitCode) => {
552
+ if (settled) return;
553
+ settled = true;
554
+ clearTimeout(timer);
555
+ if (killTimer !== void 0) clearTimeout(killTimer);
556
+ keepLine(pendingStdout);
557
+ pendingStdout = "";
558
+ const parsedRun = parseRunStdout(stdout);
559
+ resolvePromise({
560
+ ...parsedRun,
561
+ exitCode,
562
+ timedOut,
563
+ wallTimeMs: Date.now() - startedMs,
564
+ stderr
565
+ });
566
+ };
567
+ child.on("error", (error) => {
568
+ stderr += `
569
+ ${error.message}`;
570
+ finish(1);
571
+ });
572
+ child.on("close", (code) => finish(typeof code === "number" ? code : timedOut ? 124 : 1));
573
+ });
574
+ }
575
+ function loadsSkillBody(tool, args) {
576
+ if (tool !== "context" || !isRecord(args)) return false;
577
+ return args.scope === "skills" && typeof args.name === "string" && args.name.trim().length > 0;
578
+ }
579
+ function parseRunStdout(stdout) {
580
+ let sessionId = null;
581
+ const lines = [];
582
+ let finalText = "";
583
+ let sawJson = false;
584
+ const skillBodyCallIds = /* @__PURE__ */ new Set();
585
+ for (const line of stdout.split("\n")) {
586
+ if (line.trim().length === 0) continue;
587
+ let event;
588
+ try {
589
+ event = JSON.parse(line);
590
+ } catch {
591
+ continue;
592
+ }
593
+ if (!isRecord(event)) continue;
594
+ sawJson = true;
595
+ if (event.type === "session" && typeof event.id === "string") {
596
+ sessionId = event.id;
597
+ continue;
598
+ }
599
+ if (event.type === "tool_execution_start") {
600
+ const tool = readString(event.toolName) ?? readString(event.tool) ?? "tool";
601
+ const args = event.args ?? event.arguments ?? event.input;
602
+ const callId = readString(event.toolCallId);
603
+ if (callId !== null && loadsSkillBody(tool, args)) skillBodyCallIds.add(callId);
604
+ lines.push(`TOOL ${tool} args=${preview(args)}`);
605
+ if (tool === "artifact" && isRecord(args) && typeof args.content === "string") {
606
+ lines.push(`TERMINAL ${tool} content:
607
+ ${args.content}`);
608
+ }
609
+ continue;
610
+ }
611
+ if (event.type === "tool_execution_end") {
612
+ const tool = readString(event.toolName) ?? readString(event.tool) ?? "tool";
613
+ const status = event.isError === true ? "error" : "ok";
614
+ const callId = readString(event.toolCallId);
615
+ if (callId !== null && skillBodyCallIds.has(callId)) {
616
+ lines.push(`RESULT ${tool} ${status}: <skill body withheld from the judge>`);
617
+ continue;
618
+ }
619
+ lines.push(`RESULT ${tool} ${status}: ${preview(event.result)}`);
620
+ continue;
621
+ }
622
+ if (event.type === "message_end" && isRecord(event.message) && event.message.role === "assistant") {
623
+ const content = Array.isArray(event.message.content) ? event.message.content : [];
624
+ const text = content.filter((item) => {
625
+ return isRecord(item) && item.type === "text" && typeof item.text === "string";
626
+ }).map((item) => item.text).join("").trim();
627
+ if (text.length > 0) {
628
+ lines.push(`ASSISTANT: ${text}`);
629
+ finalText = text;
630
+ }
631
+ }
632
+ }
633
+ if (!sawJson) {
634
+ const text = stdout.trim();
635
+ return { sessionId: null, transcript: text, finalText: text };
636
+ }
637
+ return { sessionId, transcript: elide(lines.join("\n")), finalText };
638
+ }
639
+ function judgePrompt(scenario, baselineTranscript, treatmentTranscript) {
640
+ const bullets = scenario.expected.map((text, index) => `${index + 1}. ${text}`).join("\n");
641
+ return [
642
+ "You are scoring a skill evaluation. Two transcripts follow: BASELINE ran without the skill, TREATMENT ran with the skill loaded.",
643
+ "Score each EXPECTED bullet strictly against the TREATMENT transcript; the baseline exists only for gap context.",
644
+ "A bullet passes only if the treatment transcript observably satisfies it; anything unverifiable from the transcript fails.",
645
+ 'Reply with STRICT JSON only, no prose and no code fences, exactly: {"bullets":[{"index":1,"pass":true,"reason":"<= 25 words"}]}',
646
+ `Include one entry per bullet, indexes 1 through ${scenario.expected.length} in order.`,
647
+ "Do not use any tools. Respond with the JSON verdict directly.",
648
+ "",
649
+ `SCENARIO ${scenario.id} - ${scenario.title}`,
650
+ `SETUP: ${scenario.setup}`,
651
+ "",
652
+ "EXPECTED BULLETS:",
653
+ bullets,
654
+ "",
655
+ "BASELINE TRANSCRIPT:",
656
+ baselineTranscript.length > 0 ? baselineTranscript : "(empty)",
657
+ "",
658
+ "TREATMENT TRANSCRIPT:",
659
+ treatmentTranscript.length > 0 ? treatmentTranscript : "(empty)"
660
+ ].join("\n");
661
+ }
662
+ function judgeVerdictAbsenceReason(judge) {
663
+ const sawBulletsKey = `${judge.finalText}
664
+ ${judge.transcript}`.includes('"bullets"');
665
+ return sawBulletsKey ? "judge response truncated mid-verdict (no complete bullets object): bullet not scored" : "judge response carried no parseable verdict object: bullet not scored";
666
+ }
667
+ function parseJudgeVerdicts(scenario, judge) {
668
+ const parsed = extractBulletsObject(judge.finalText) ?? extractBulletsObject(judge.transcript);
669
+ if (parsed === null) {
670
+ const reason = judgeVerdictAbsenceReason(judge);
671
+ return scenario.expected.map((text, i) => ({ index: i + 1, text, verdict: "unmeasured", reason }));
672
+ }
673
+ const entries = /* @__PURE__ */ new Map();
674
+ if (Array.isArray(parsed.bullets)) {
675
+ for (const item of parsed.bullets) {
676
+ if (!isRecord(item)) continue;
677
+ const index = typeof item.index === "number" ? item.index : Number.parseInt(String(item.index), 10);
678
+ if (!Number.isInteger(index)) continue;
679
+ entries.set(index, {
680
+ pass: item.pass === true,
681
+ reason: readString(item.reason) ?? ""
682
+ });
683
+ }
684
+ }
685
+ return scenario.expected.map((text, i) => {
686
+ const index = i + 1;
687
+ const entry = entries.get(index);
688
+ if (entry === void 0) {
689
+ return {
690
+ index,
691
+ text,
692
+ verdict: "unmeasured",
693
+ reason: "judge output omitted this bullet: bullet not scored"
694
+ };
695
+ }
696
+ return { index, text, verdict: entry.pass ? "pass" : "fail", reason: entry.reason };
697
+ });
698
+ }
699
+ function extractBulletsObject(text) {
700
+ const stripped = text.replace(/```(?:json)?/g, "");
701
+ const bulletsAt = stripped.indexOf('"bullets"');
702
+ if (bulletsAt < 0) return null;
703
+ let start = stripped.lastIndexOf("{", bulletsAt);
704
+ while (start >= 0) {
705
+ const candidate = balancedJsonSlice(stripped, start);
706
+ if (candidate !== null) {
707
+ try {
708
+ const parsed = JSON.parse(candidate);
709
+ if (isRecord(parsed) && Array.isArray(parsed.bullets)) return parsed;
710
+ } catch {
711
+ }
712
+ }
713
+ if (start === 0) break;
714
+ start = stripped.lastIndexOf("{", start - 1);
715
+ }
716
+ return null;
717
+ }
718
+ function balancedJsonSlice(text, start) {
719
+ let depth = 0;
720
+ let inString = false;
721
+ let escaped = false;
722
+ for (let i = start; i < text.length; i += 1) {
723
+ const ch = text[i];
724
+ if (escaped) {
725
+ escaped = false;
726
+ continue;
727
+ }
728
+ if (ch === "\\") {
729
+ if (inString) escaped = true;
730
+ continue;
731
+ }
732
+ if (ch === '"') {
733
+ inString = !inString;
734
+ continue;
735
+ }
736
+ if (inString) continue;
737
+ if (ch === "{") depth += 1;
738
+ else if (ch === "}") {
739
+ depth -= 1;
740
+ if (depth === 0) return text.slice(start, i + 1);
741
+ }
742
+ }
743
+ return null;
744
+ }
745
+ function synthesizeArtifact(skillName, evalsPath, evalsRaw, startedAt, endedAt, outcomes, target) {
746
+ const contentHash = createHash("sha256").update(evalsRaw, "utf8").digest("hex");
747
+ const stamp = startedAt.replace(/[-:.]/g, "");
748
+ const evalId = `skill-${skillName}-${stamp}-${contentHash.slice(0, 8)}`;
749
+ const results = outcomes.map((outcome) => {
750
+ const failed = outcome.bullets.some((bullet) => bullet.verdict === "fail" || bullet.verdict === "error");
751
+ const unmeasured = outcome.bullets.some((bullet) => bullet.verdict === "unmeasured");
752
+ const pass = outcome.bullets.length > 0 && !failed && !unmeasured;
753
+ const record = {
754
+ taskId: outcome.scenario.id,
755
+ runId: `${evalId}-${outcome.scenario.id}`,
756
+ repeatIndex: 0,
757
+ cwd: outcome.workspace,
758
+ prompt: outcome.scenario.setup,
759
+ tags: [
760
+ "skill-eval",
761
+ `skill:${skillName}`,
762
+ ...unmeasured ? ["scenario:unmeasured"] : [],
763
+ ...outcome.bullets.map((bullet) => `bullet-${bullet.index}:${bullet.verdict}`)
764
+ ],
765
+ pass,
766
+ // 3, not 1, when the scenario carries an unscored bullet and no failed
767
+ // one: a reader who sees exitCode 1 with failureClass verifier_failed
768
+ // reads "this skill failed its rubric", and an unscored bullet is the
769
+ // judge's silence, not the skill's answer.
770
+ exitCode: pass ? 0 : failed ? 1 : 3,
771
+ tokens: outcome.usage.tokens,
772
+ costUsd: outcome.usage.costUsd,
773
+ wallTimeMs: outcome.wallTimeMs,
774
+ harness: outcome.usage.harness,
775
+ commands: []
776
+ };
777
+ if (failed) record.failureClass = "verifier_failed";
778
+ return record;
779
+ });
780
+ return {
781
+ version: 1,
782
+ evalId,
783
+ taskFile: evalsPath,
784
+ taskFileHash: contentHash,
785
+ clio: evalClioProvenance(),
786
+ environment: evalEnvironmentProvenance(),
787
+ target: target ?? null,
788
+ model: null,
789
+ thinking: null,
790
+ paths: { taskFile: evalsPath, receipts: [], sessionLedgers: [] },
791
+ repeat: 1,
792
+ startedAt,
793
+ endedAt,
794
+ summary: summarizeEvalResults(results),
795
+ results
796
+ };
797
+ }
798
+ function sidecar(skillName, evalId, outcomes, allowNetwork) {
799
+ return {
800
+ version: 1,
801
+ schema: "experimental",
802
+ kind: "skill-eval",
803
+ skill: skillName,
804
+ evalId,
805
+ network: networkPolicyLabel(allowNetwork),
806
+ autonomy: ARM_AUTONOMY,
807
+ deltas: [
808
+ "bullet verdicts are judge-scored from run transcripts, not command exit codes; the evals-domain artifact carries scenario-level records with empty command lists",
809
+ "a bullet the judge never scored is recorded unmeasured, not failed: its scenario record is pass:false with exitCode 3 and no failureClass",
810
+ "an arm whose transcript carries the headless permission wall is recorded unmeasured with an infraError: the harness's own gate is not a verdict about the skill",
811
+ "tokens and cost are rolled up from headless main-agent receipts when those receipts are present",
812
+ "this sidecar is additive and is registered in overview.json files[]"
813
+ ],
814
+ scenarios: outcomes.map((outcome) => ({
815
+ id: outcome.scenario.id,
816
+ title: outcome.scenario.title,
817
+ setup: outcome.scenario.setup,
818
+ fixtureCommands: outcome.scenario.fixtureCommands ?? null,
819
+ infraError: outcome.infraError,
820
+ usage: outcome.usage,
821
+ bullets: outcome.bullets,
822
+ baseline: sidecarRun(outcome.baseline),
823
+ treatment: sidecarRun(outcome.treatment),
824
+ judge: sidecarRun(outcome.judge)
825
+ }))
826
+ };
827
+ }
828
+ function sidecarRun(run) {
829
+ if (run === null) return null;
830
+ return {
831
+ sessionId: run.sessionId,
832
+ exitCode: run.exitCode,
833
+ timedOut: run.timedOut,
834
+ wallTimeMs: run.wallTimeMs,
835
+ transcript: run.transcript,
836
+ stderrTail: run.stderr.slice(-500)
837
+ };
838
+ }
839
+ function printHumanReport(skillName, outcomes, evalId, evidenceId, evidenceDirectory, allowNetwork) {
840
+ const rows = [["scenario", "bullet", "verdict", "expected"]];
841
+ for (const outcome of outcomes) {
842
+ for (const bullet of outcome.bullets) {
843
+ rows.push([outcome.scenario.id, String(bullet.index), bullet.verdict, truncate(bullet.text, 76)]);
844
+ }
845
+ }
846
+ process.stdout.write(formatColumns(rows));
847
+ const total = outcomes.reduce((sum, outcome) => sum + outcome.bullets.length, 0);
848
+ const passed = outcomes.reduce(
849
+ (sum, outcome) => sum + outcome.bullets.filter((bullet) => bullet.verdict === "pass").length,
850
+ 0
851
+ );
852
+ const unmeasured = outcomes.reduce(
853
+ (sum, outcome) => sum + outcome.bullets.filter((bullet) => bullet.verdict === "unmeasured").length,
854
+ 0
855
+ );
856
+ const unmeasuredNote = unmeasured > 0 ? `, ${unmeasured} unmeasured` : "";
857
+ process.stdout.write(
858
+ `
859
+ ${skillName}: ${passed}/${total} bullets passed${unmeasuredNote} (judge-scored, experimental)
860
+ `
861
+ );
862
+ if (unmeasured > 0) {
863
+ process.stdout.write(
864
+ unmeasured === total ? "no bullet was scored, so this run is not evidence about the skill either way\n" : `${unmeasured} of ${total} bullets were never scored; unmeasured is not a failed bullet
865
+ `
866
+ );
867
+ }
868
+ for (const outcome of outcomes) {
869
+ if (outcome.infraError !== null) {
870
+ process.stdout.write(`${outcome.scenario.id}: infra error: ${outcome.infraError}
871
+ `);
872
+ continue;
873
+ }
874
+ for (const bullet of outcome.bullets.filter((item) => item.verdict !== "pass")) {
875
+ process.stdout.write(
876
+ `${outcome.scenario.id} bullet ${bullet.index} ${bullet.verdict}${bullet.reason.length > 0 ? `: ${bullet.reason}` : ""}
877
+ `
878
+ );
879
+ }
880
+ }
881
+ process.stdout.write(`${describeArmPolicyOutcome(allowNetwork)}
882
+ `);
883
+ process.stdout.write(`eval artifact: ${evalId}
884
+ `);
885
+ if (evidenceId !== null && evidenceDirectory !== null) {
886
+ process.stdout.write(`evidence: ${evidenceId} at ${evidenceDirectory} (per-bullet detail in skill-eval.json)
887
+ `);
888
+ }
889
+ }
890
+ function preview(value) {
891
+ if (value === void 0) return "";
892
+ const text = typeof value === "string" ? value : safeStringify(value);
893
+ return truncate(text.replace(/\s+/g, " ").trim(), PREVIEW_MAX_CHARS);
894
+ }
895
+ function safeStringify(value) {
896
+ try {
897
+ return JSON.stringify(value) ?? "";
898
+ } catch {
899
+ return String(value);
900
+ }
901
+ }
902
+ function elide(text) {
903
+ if (text.length <= TRANSCRIPT_HEAD_CHARS + TRANSCRIPT_TAIL_CHARS) return text;
904
+ return `${text.slice(0, TRANSCRIPT_HEAD_CHARS)}
905
+ [... transcript elided ...]
906
+ ${text.slice(-TRANSCRIPT_TAIL_CHARS)}`;
907
+ }
908
+ function truncate(text, maxChars) {
909
+ if (text.length <= maxChars) return text;
910
+ return `${text.slice(0, Math.max(0, maxChars - 3))}...`;
911
+ }
912
+ function readString(value) {
913
+ return typeof value === "string" && value.length > 0 ? value : null;
914
+ }
915
+ function readNumber(value) {
916
+ return typeof value === "number" && Number.isFinite(value) ? value : 0;
917
+ }
918
+ function isRecord(value) {
919
+ return typeof value === "object" && value !== null && !Array.isArray(value);
920
+ }
921
+ export {
922
+ armRunArgs,
923
+ evalChildEnv,
924
+ extractBulletsObject,
925
+ materializeSkillEvalWorkspaces,
926
+ parseJudgeVerdicts,
927
+ parseRunStdout,
928
+ permissionWallReason,
929
+ resolveSkillBaseDir,
930
+ runSkillsEvalCommand
931
+ };
932
+ //# sourceMappingURL=skills-eval-E2ZTW4PL.js.map